Code Coverage
 
Lines
Functions and Methods
Classes and Traits
Total
66.41% covered (warning)
66.41%
425 / 640
68.00% covered (warning)
68.00%
34 / 50
CRAP
0.00% covered (danger)
0.00%
0 / 1
Sanitizer
66.51% covered (warning)
66.51%
425 / 639
68.00% covered (warning)
68.00%
34 / 50
1592.11
0.00% covered (danger)
0.00%
0 / 1
 getAttribsRegex
100.00% covered (success)
100.00%
11 / 11
100.00% covered (success)
100.00%
1 / 1
2
 getAttribNameRegex
100.00% covered (success)
100.00%
5 / 5
100.00% covered (success)
100.00%
1 / 1
2
 getRecognizedTagData
40.00% covered (danger)
40.00%
24 / 60
0.00% covered (danger)
0.00%
0 / 1
21.82
 internalRemoveHtmlTags
96.43% covered (success)
96.43%
27 / 28
0.00% covered (danger)
0.00%
0 / 1
12
 removeSomeTags
100.00% covered (success)
100.00%
29 / 29
100.00% covered (success)
100.00%
1 / 1
1
 removeHTMLcomments
70.59% covered (warning)
70.59%
12 / 17
0.00% covered (danger)
0.00%
0 / 1
9.63
 validateTag
77.78% covered (warning)
77.78%
7 / 9
0.00% covered (danger)
0.00%
0 / 1
8.70
 validateTagAttributes
100.00% covered (success)
100.00%
2 / 2
100.00% covered (success)
100.00%
1 / 1
1
 validateAttributes
91.49% covered (success)
91.49%
43 / 47
0.00% covered (danger)
0.00%
0 / 1
36.80
 isReservedDataAttribute
100.00% covered (success)
100.00%
1 / 1
100.00% covered (success)
100.00%
1 / 1
1
 mergeAttributes
0.00% covered (danger)
0.00%
0 / 8
0.00% covered (danger)
0.00%
0 / 1
42
 normalizeCss
100.00% covered (success)
100.00%
18 / 18
100.00% covered (success)
100.00%
1 / 1
4
 checkCss
100.00% covered (success)
100.00%
7 / 7
100.00% covered (success)
100.00%
1 / 1
4
 cssDecodeCallback
80.00% covered (warning)
80.00%
8 / 10
0.00% covered (danger)
0.00%
0 / 1
8.51
 fixTagAttributes
85.71% covered (warning)
85.71%
6 / 7
0.00% covered (danger)
0.00%
0 / 1
3.03
 encodeAttribute
100.00% covered (success)
100.00%
7 / 7
100.00% covered (success)
100.00%
1 / 1
1
 armorFrenchSpaces
100.00% covered (success)
100.00%
6 / 6
100.00% covered (success)
100.00%
1 / 1
1
 safeEncodeAttribute
100.00% covered (success)
100.00%
25 / 25
100.00% covered (success)
100.00%
1 / 1
1
 escapeIdForAttribute
100.00% covered (success)
100.00%
6 / 6
100.00% covered (success)
100.00%
1 / 1
3
 escapeIdForLink
100.00% covered (success)
100.00%
5 / 5
100.00% covered (success)
100.00%
1 / 1
2
 escapeIdForExternalInterwiki
100.00% covered (success)
100.00%
2 / 2
100.00% covered (success)
100.00%
1 / 1
1
 escapeIdInternalUrl
100.00% covered (success)
100.00%
4 / 4
100.00% covered (success)
100.00%
1 / 1
2
 escapeIdInternal
100.00% covered (success)
100.00%
14 / 14
100.00% covered (success)
100.00%
1 / 1
4
 escapeIdReferenceListInternal
100.00% covered (success)
100.00%
5 / 5
100.00% covered (success)
100.00%
1 / 1
2
 escapeClass
0.00% covered (danger)
0.00%
0 / 4
0.00% covered (danger)
0.00%
0 / 1
2
 escapeCombiningChar
100.00% covered (success)
100.00%
3 / 3
100.00% covered (success)
100.00%
1 / 1
1
 escapeHtmlAllowEntities
100.00% covered (success)
100.00%
3 / 3
100.00% covered (success)
100.00%
1 / 1
1
 decodeTagAttributes
100.00% covered (success)
100.00%
19 / 19
100.00% covered (success)
100.00%
1 / 1
5
 safeEncodeTagAttributes
100.00% covered (success)
100.00%
7 / 7
100.00% covered (success)
100.00%
1 / 1
3
 getTagAttributeCallback
88.89% covered (warning)
88.89%
8 / 9
0.00% covered (danger)
0.00%
0 / 1
5.03
 normalizeWhitespace
60.00% covered (warning)
60.00%
3 / 5
0.00% covered (danger)
0.00%
0 / 1
2.26
 normalizeSectionNameWhitespace
60.00% covered (warning)
60.00%
3 / 5
0.00% covered (danger)
0.00%
0 / 1
2.26
 normalizeCharReferences
100.00% covered (success)
100.00%
5 / 5
100.00% covered (success)
100.00%
1 / 1
1
 normalizeCharReferencesCallback
100.00% covered (success)
100.00%
10 / 10
100.00% covered (success)
100.00%
1 / 1
5
 normalizeEntity
100.00% covered (success)
100.00%
9 / 9
100.00% covered (success)
100.00%
1 / 1
4
 decCharReference
100.00% covered (success)
100.00%
4 / 4
100.00% covered (success)
100.00%
1 / 1
2
 hexCharReference
100.00% covered (success)
100.00%
4 / 4
100.00% covered (success)
100.00%
1 / 1
3
 validateCodepoint
100.00% covered (success)
100.00%
6 / 6
100.00% covered (success)
100.00%
1 / 1
10
 decodeCharReferences
100.00% covered (success)
100.00%
5 / 5
100.00% covered (success)
100.00%
1 / 1
1
 decodeCharReferencesAndNormalize
100.00% covered (success)
100.00%
8 / 8
100.00% covered (success)
100.00%
1 / 1
2
 decodeCharReferencesCallback
100.00% covered (success)
100.00%
10 / 10
100.00% covered (success)
100.00%
1 / 1
5
 decodeChar
100.00% covered (success)
100.00%
3 / 3
100.00% covered (success)
100.00%
1 / 1
2
 decodeEntity
75.00% covered (warning)
75.00%
3 / 4
0.00% covered (danger)
0.00%
0 / 1
2.06
 attributesAllowedInternal
100.00% covered (success)
100.00%
2 / 2
100.00% covered (success)
100.00%
1 / 1
1
 setupAttributesAllowedInternal
2.21% covered (danger)
2.21%
3 / 136
0.00% covered (danger)
0.00%
0 / 1
5.74
 stripAllTags
100.00% covered (success)
100.00%
9 / 9
100.00% covered (success)
100.00%
1 / 1
1
 hackDocType
0.00% covered (danger)
0.00%
0 / 11
0.00% covered (danger)
0.00%
0 / 1
20
 stripIDNs
100.00% covered (success)
100.00%
1 / 1
100.00% covered (success)
100.00%
1 / 1
1
 cleanUrl
100.00% covered (success)
100.00%
12 / 12
100.00% covered (success)
100.00%
1 / 1
4
 validateEmail
91.67% covered (success)
91.67%
11 / 12
0.00% covered (danger)
0.00%
0 / 1
2.00
1<?php
2declare( strict_types = 1 );
3
4/**
5 * HTML sanitizer for %MediaWiki.
6 *
7 * Copyright © 2002-2005 Brooke Vibber <bvibber@wikimedia.org> et al
8 * https://www.mediawiki.org/
9 *
10 * @license GPL-2.0-or-later
11 * @file
12 * @ingroup Parser
13 */
14
15namespace MediaWiki\Parser;
16
17use InvalidArgumentException;
18use LogicException;
19use MediaWiki\HookContainer\HookRunner;
20use MediaWiki\MediaWikiServices;
21use MediaWiki\Tidy\RemexCompatFormatter;
22use UnexpectedValueException;
23use Wikimedia\RemexHtml\HTMLData;
24use Wikimedia\RemexHtml\Serializer\Serializer as RemexSerializer;
25use Wikimedia\RemexHtml\Tokenizer\Tokenizer as RemexTokenizer;
26use Wikimedia\RemexHtml\TreeBuilder\Dispatcher as RemexDispatcher;
27use Wikimedia\RemexHtml\TreeBuilder\TreeBuilder as RemexTreeBuilder;
28use Wikimedia\StringUtils\StringUtils;
29
30/**
31 * HTML sanitizer for MediaWiki
32 * @ingroup Parser
33 */
34class Sanitizer {
35    /**
36     * Regular expression to match various types of character references in
37     * Sanitizer::normalizeCharReferences and Sanitizer::decodeCharReferences.
38     * Note that HTML5 allows some named entities to omit the trailing
39     * semicolon; wikitext entities *must* have a trailing semicolon.
40     */
41    private const CHAR_REFS_REGEX =
42        '/&([A-Za-z0-9\x80-\xff]+;)
43        |&\#([0-9]+);
44        |&\#[xX]([0-9A-Fa-f]+);
45        |&/x';
46
47    private const INSECURE_RE = '! expression
48        | accelerator\s*:
49        | -o-link\s*:
50        | -o-link-source\s*:
51        | -o-replace\s*:
52        | url\s*\(
53        | src\s*\(
54        | image\s*\(
55        | image-set\s*\(
56        | attr\s*\([^)]+[\s,]+url
57    !ix';
58
59    /**
60     * Acceptable tag name charset from HTML5 parsing spec
61     * https://www.w3.org/TR/html5/syntax.html#tag-open-state
62     */
63    private const ELEMENT_BITS_REGEX = '!^(/?)([A-Za-z][^\t\n\v />\0]*+)([^>]*?)(/?>)([^<]*)$!';
64
65    /**
66     * Pattern matching evil uris like javascript:
67     * WARNING: DO NOT use this in any place that actually requires denying
68     * certain URIs for security reasons. There are NUMEROUS[1] ways to bypass
69     * pattern-based deny lists; the only way to be secure from javascript:
70     * uri based xss vectors is to allow only things that you know are safe
71     * and deny everything else.
72     * [1]: http://ha.ckers.org/xss.html
73     */
74    private const EVIL_URI_PATTERN = '!(^|\s|\*/\s*)(javascript|vbscript)([^\w]|$)!i';
75    private const XMLNS_ATTRIBUTE_PATTERN = "/^xmlns:[:A-Z_a-z-.0-9]+$/";
76
77    /**
78     * Tells escapeUrlForHtml() to encode the ID using the wiki's primary encoding.
79     *
80     * @since 1.30
81     */
82    public const ID_PRIMARY = 0;
83
84    /**
85     * Tells escapeUrlForHtml() to encode the ID using the fallback encoding, or return false
86     * if no fallback is configured.
87     *
88     * @since 1.30
89     */
90    public const ID_FALLBACK = 1;
91
92    /** Characters that will be ignored in IDNs.
93     * https://datatracker.ietf.org/doc/html/rfc8264#section-9.13
94     * https://www.unicode.org/Public/UCD/latest/ucd/DerivedCoreProperties.txt
95     * Strip them before further processing so deny lists and such work.
96     */
97    private const IDN_RE_G = "/
98                \\s|      # general whitespace
99                \u{00AD}|               # SOFT HYPHEN
100                \u{034F}|               # COMBINING GRAPHEME JOINER
101                \u{061C}|               # ARABIC LETTER MARK
102                [\u{115F}-\u{1160}]|    # HANGUL CHOSEONG FILLER..
103                            # HANGUL JUNGSEONG FILLER
104                [\u{17B4}-\u{17B5}]|    # KHMER VOWEL INHERENT AQ..
105                            # KHMER VOWEL INHERENT AA
106                [\u{180B}-\u{180D}]|    # MONGOLIAN FREE VARIATION SELECTOR ONE..
107                            # MONGOLIAN FREE VARIATION SELECTOR THREE
108                \u{180E}|               # MONGOLIAN VOWEL SEPARATOR
109                [\u{200B}-\u{200F}]|    # ZERO WIDTH SPACE..
110                            # RIGHT-TO-LEFT MARK
111                [\u{202A}-\u{202E}]|    # LEFT-TO-RIGHT EMBEDDING..
112                            # RIGHT-TO-LEFT OVERRIDE
113                [\u{2060}-\u{2064}]|    # WORD JOINER..
114                            # INVISIBLE PLUS
115                \u{2065}|               # <reserved-2065>
116                [\u{2066}-\u{206F}]|    # LEFT-TO-RIGHT ISOLATE..
117                            # NOMINAL DIGIT SHAPES
118                \u{3164}|               # HANGUL FILLER
119                [\u{FE00}-\u{FE0F}]|    # VARIATION SELECTOR-1..
120                            # VARIATION SELECTOR-16
121                \u{FEFF}|               # ZERO WIDTH NO-BREAK SPACE
122                \u{FFA0}|               # HALFWIDTH HANGUL FILLER
123                [\u{FFF0}-\u{FFF8}]|    # <reserved-FFF0>..
124                            # <reserved-FFF8>
125                [\u{1BCA0}-\u{1BCA3}]|  # SHORTHAND FORMAT LETTER OVERLAP..
126                            # SHORTHAND FORMAT UP STEP
127                [\u{1D173}-\u{1D17A}]|  # MUSICAL SYMBOL BEGIN BEAM..
128                            # MUSICAL SYMBOL END PHRASE
129                \u{E0000}|              # <reserved-E0000>
130                \u{E0001}|              # LANGUAGE TAG
131                [\u{E0002}-\u{E001F}]|  # <reserved-E0002>..
132                            # <reserved-E001F>
133                [\u{E0020}-\u{E007F}]|  # TAG SPACE..
134                            # CANCEL TAG
135                [\u{E0080}-\u{E00FF}]|  # <reserved-E0080>..
136                            # <reserved-E00FF>
137                [\u{E0100}-\u{E01EF}]|  # VARIATION SELECTOR-17..
138                            # VARIATION SELECTOR-256
139                [\u{E01F0}-\u{E0FFF}]|  # <reserved-E01F0>..
140                            # <reserved-E0FFF>
141                /xuD";
142
143    /**
144     * Character entity aliases accepted by MediaWiki in wikitext.
145     * These are not part of the HTML standard.
146     */
147    private const MW_ENTITY_ALIASES = [
148        'רלמ;' => 'rlm;',
149        'رلم;' => 'rlm;',
150    ];
151
152    /**
153     * Lazy-initialised attributes regex, see getAttribsRegex()
154     */
155    private static ?string $attribsRegex = null;
156
157    /**
158     * Regular expression to match HTML/XML attribute pairs within a tag.
159     * Based on https://www.w3.org/TR/html5/syntax.html#before-attribute-name-state
160     * Used in Sanitizer::decodeTagAttributes
161     */
162    private static function getAttribsRegex(): string {
163        if ( self::$attribsRegex === null ) {
164            $spaceChars = '\x09\x0a\x0c\x0d\x20';
165            $space = "[{$spaceChars}]";
166            $attrib = "[^{$spaceChars}\/>=]";
167            $attribFirst = "(?:{$attrib}|=)";
168            self::$attribsRegex =
169                "/({$attribFirst}{$attrib}*)
170                    ($space*=$space*
171                    (?:
172                        # The attribute value: quoted or alone
173                        \"([^\"]*)(?:\"|\$)
174                        | '([^']*)(?:'|\$)
175                        | (((?!$space|>).)*)
176                    )
177                )?/sxu";
178        }
179        return self::$attribsRegex;
180    }
181
182    /**
183     * Lazy-initialised attribute name regex, see getAttribNameRegex()
184     */
185    private static ?string $attribNameRegex = null;
186
187    /**
188     * Used in Sanitizer::decodeTagAttributes to filter attributes.
189     */
190    private static function getAttribNameRegex(): string {
191        if ( self::$attribNameRegex === null ) {
192            $attribFirst = "[:_\p{L}\p{N}]";
193            $attrib = "[:_\.\-\p{L}\p{N}]";
194            self::$attribNameRegex = "/^({$attribFirst}{$attrib}*)$/sxu";
195        }
196        return self::$attribNameRegex;
197    }
198
199    /**
200     * Return the various lists of recognized tags
201     * @param string[] $extratags For any extra tags to include
202     * @param string[] $removetags For any tags (default or extra) to exclude
203     * @return array
204     * @internal
205     */
206    public static function getRecognizedTagData( array $extratags = [], array $removetags = [] ): array {
207        static $commonCase, $staticInitialised = false;
208        $isCommonCase = ( $extratags === [] && $removetags === [] );
209        if ( $staticInitialised && $isCommonCase && $commonCase ) {
210            return $commonCase;
211        }
212
213        static $htmlpairsStatic, $htmlsingle, $htmlsingleonly, $htmlnest, $tabletags,
214            $htmllist, $listtags, $htmlsingleallowed, $htmlelementsStatic;
215
216        if ( !$staticInitialised ) {
217            $htmlpairsStatic = [ # Tags that must be closed
218                'b', 'bdi', 'del', 'i', 'ins', 'u', 'font', 'big', 'small', 'sub', 'sup', 'h1',
219                'h2', 'h3', 'h4', 'h5', 'h6', 'cite', 'code', 'em', 's',
220                'strike', 'strong', 'tt', 'var', 'div', 'center',
221                'blockquote', 'ol', 'ul', 'dl', 'table', 'caption', 'pre',
222                'ruby', 'rb', 'rp', 'rt', 'rtc', 'p', 'span', 'abbr', 'dfn',
223                'kbd', 'samp', 'data', 'time', 'mark'
224            ];
225            # These tags can be self-closed. For tags not also on
226            # $htmlsingleonly, a self-closed tag will be emitted as
227            # an empty element (open-tag/close-tag pair).
228            $htmlsingle = [
229                'br', 'wbr', 'hr', 'li', 'dt', 'dd', 'meta', 'link'
230            ];
231
232            # Elements that cannot have close tags. This is (not coincidentally)
233            # also the list of tags for which the HTML 5 parsing algorithm
234            # requires you to "acknowledge the token's self-closing flag", i.e.
235            # a self-closing tag like <br/> is not an HTML 5 parse error only
236            # for this list.
237            $htmlsingleonly = [
238                'br', 'wbr', 'hr', 'meta', 'link'
239            ];
240
241            $htmlnest = [ # Tags that can be nested--??
242                'table', 'tr', 'td', 'th', 'div', 'blockquote', 'ol', 'ul',
243                'li', 'dl', 'dt', 'dd', 'font', 'big', 'small', 'sub', 'sup', 'span',
244                'var', 'kbd', 'samp', 'em', 'strong', 'q', 'ruby', 'bdo'
245            ];
246            $tabletags = [ # Can only appear inside table, we will close them
247                'td', 'th', 'tr',
248            ];
249            $htmllist = [ # Tags used by list
250                'ul', 'ol',
251            ];
252            $listtags = [ # Tags that can appear in a list
253                'li',
254            ];
255
256            $htmlsingleallowed = array_unique( array_merge( $htmlsingle, $tabletags ) );
257            $htmlelementsStatic = array_unique( array_merge( $htmlsingle, $htmlpairsStatic, $htmlnest ) );
258
259            # Convert them all to hashtables for faster lookup
260            $vars = [ 'htmlpairsStatic', 'htmlsingle', 'htmlsingleonly', 'htmlnest', 'tabletags',
261                'htmllist', 'listtags', 'htmlsingleallowed', 'htmlelementsStatic' ];
262            foreach ( $vars as $var ) {
263                $$var = array_fill_keys( $$var, true );
264            }
265            $staticInitialised = true;
266        }
267
268        # Populate $htmlpairs and $htmlelements with the $extratags and $removetags arrays
269        $extratags = array_fill_keys( $extratags, true );
270        $removetags = array_fill_keys( $removetags, true );
271        $htmlpairs = array_merge( $extratags, $htmlpairsStatic );
272        $htmlelements = array_diff_key( array_merge( $extratags, $htmlelementsStatic ), $removetags );
273
274        $result = [
275            'htmlpairs' => $htmlpairs,
276            'htmlsingle' => $htmlsingle,
277            'htmlsingleonly' => $htmlsingleonly,
278            'htmlnest' => $htmlnest,
279            'tabletags' => $tabletags,
280            'htmllist' => $htmllist,
281            'listtags' => $listtags,
282            'htmlsingleallowed' => $htmlsingleallowed,
283            'htmlelements' => $htmlelements,
284        ];
285        if ( $isCommonCase ) {
286            $commonCase = $result;
287        }
288        return $result;
289    }
290
291    /**
292     * Cleans up HTML, removes dangerous tags and attributes, and
293     * removes HTML comments; BEWARE there may be unmatched HTML
294     * tags in the result.
295     *
296     * @note Callers are recommended to use `::removeSomeTags()` instead
297     * of this method.  `Sanitizer::removeSomeTags()` is safer and will
298     * always return well-formed HTML; however, it is significantly
299     * slower (especially for short strings where setup costs
300     * predominate).  This method is for internal use by the legacy parser
301     * where we know the result will be cleaned up in a subsequent tidy pass.
302     *
303     * @param string $text Original string; see T268353 for why untainted.
304     * @param-taint $text none
305     * @param callable|null $processCallback Callback to do any variable or
306     *   parameter replacements in HTML attribute values.
307     *   This argument should be considered @internal.
308     * @param-taint $processCallback exec_shell
309     * @param array|bool $args Arguments for the processing callback
310     * @param-taint $args none
311     * @param array $extratags For any extra tags to include
312     * @param-taint $extratags tainted
313     * @param array $removetags For any tags (default or extra) to exclude
314     * @param-taint $removetags none
315     * @return string
316     * @return-taint escaped
317     * @internal
318     */
319    public static function internalRemoveHtmlTags( string $text, ?callable $processCallback = null,
320        $args = [], array $extratags = [], array $removetags = []
321    ): string {
322        $tagData = self::getRecognizedTagData( $extratags, $removetags );
323        $htmlsingle = $tagData['htmlsingle'];
324        $htmlsingleonly = $tagData['htmlsingleonly'];
325        $htmlelements = $tagData['htmlelements'];
326
327        # Remove HTML comments
328        $text = self::removeHTMLcomments( $text );
329        $bits = explode( '<', $text );