Code Coverage |
||||||||||
Lines |
Functions and Methods |
Classes and Traits |
||||||||
| Total | |
66.41% |
425 / 640 |
|
68.00% |
34 / 50 |
CRAP | |
0.00% |
0 / 1 |
| Sanitizer | |
66.51% |
425 / 639 |
|
68.00% |
34 / 50 |
1592.11 | |
0.00% |
0 / 1 |
| getAttribsRegex | |
100.00% |
11 / 11 |
|
100.00% |
1 / 1 |
2 | |||
| getAttribNameRegex | |
100.00% |
5 / 5 |
|
100.00% |
1 / 1 |
2 | |||
| getRecognizedTagData | |
40.00% |
24 / 60 |
|
0.00% |
0 / 1 |
21.82 | |||
| internalRemoveHtmlTags | |
96.43% |
27 / 28 |
|
0.00% |
0 / 1 |
12 | |||
| removeSomeTags | |
100.00% |
29 / 29 |
|
100.00% |
1 / 1 |
1 | |||
| removeHTMLcomments | |
70.59% |
12 / 17 |
|
0.00% |
0 / 1 |
9.63 | |||
| validateTag | |
77.78% |
7 / 9 |
|
0.00% |
0 / 1 |
8.70 | |||
| validateTagAttributes | |
100.00% |
2 / 2 |
|
100.00% |
1 / 1 |
1 | |||
| validateAttributes | |
91.49% |
43 / 47 |
|
0.00% |
0 / 1 |
36.80 | |||
| isReservedDataAttribute | |
100.00% |
1 / 1 |
|
100.00% |
1 / 1 |
1 | |||
| mergeAttributes | |
0.00% |
0 / 8 |
|
0.00% |
0 / 1 |
42 | |||
| normalizeCss | |
100.00% |
18 / 18 |
|
100.00% |
1 / 1 |
4 | |||
| checkCss | |
100.00% |
7 / 7 |
|
100.00% |
1 / 1 |
4 | |||
| cssDecodeCallback | |
80.00% |
8 / 10 |
|
0.00% |
0 / 1 |
8.51 | |||
| fixTagAttributes | |
85.71% |
6 / 7 |
|
0.00% |
0 / 1 |
3.03 | |||
| encodeAttribute | |
100.00% |
7 / 7 |
|
100.00% |
1 / 1 |
1 | |||
| armorFrenchSpaces | |
100.00% |
6 / 6 |
|
100.00% |
1 / 1 |
1 | |||
| safeEncodeAttribute | |
100.00% |
25 / 25 |
|
100.00% |
1 / 1 |
1 | |||
| escapeIdForAttribute | |
100.00% |
6 / 6 |
|
100.00% |
1 / 1 |
3 | |||
| escapeIdForLink | |
100.00% |
5 / 5 |
|
100.00% |
1 / 1 |
2 | |||
| escapeIdForExternalInterwiki | |
100.00% |
2 / 2 |
|
100.00% |
1 / 1 |
1 | |||
| escapeIdInternalUrl | |
100.00% |
4 / 4 |
|
100.00% |
1 / 1 |
2 | |||
| escapeIdInternal | |
100.00% |
14 / 14 |
|
100.00% |
1 / 1 |
4 | |||
| escapeIdReferenceListInternal | |
100.00% |
5 / 5 |
|
100.00% |
1 / 1 |
2 | |||
| escapeClass | |
0.00% |
0 / 4 |
|
0.00% |
0 / 1 |
2 | |||
| escapeCombiningChar | |
100.00% |
3 / 3 |
|
100.00% |
1 / 1 |
1 | |||
| escapeHtmlAllowEntities | |
100.00% |
3 / 3 |
|
100.00% |
1 / 1 |
1 | |||
| decodeTagAttributes | |
100.00% |
19 / 19 |
|
100.00% |
1 / 1 |
5 | |||
| safeEncodeTagAttributes | |
100.00% |
7 / 7 |
|
100.00% |
1 / 1 |
3 | |||
| getTagAttributeCallback | |
88.89% |
8 / 9 |
|
0.00% |
0 / 1 |
5.03 | |||
| normalizeWhitespace | |
60.00% |
3 / 5 |
|
0.00% |
0 / 1 |
2.26 | |||
| normalizeSectionNameWhitespace | |
60.00% |
3 / 5 |
|
0.00% |
0 / 1 |
2.26 | |||
| normalizeCharReferences | |
100.00% |
5 / 5 |
|
100.00% |
1 / 1 |
1 | |||
| normalizeCharReferencesCallback | |
100.00% |
10 / 10 |
|
100.00% |
1 / 1 |
5 | |||
| normalizeEntity | |
100.00% |
9 / 9 |
|
100.00% |
1 / 1 |
4 | |||
| decCharReference | |
100.00% |
4 / 4 |
|
100.00% |
1 / 1 |
2 | |||
| hexCharReference | |
100.00% |
4 / 4 |
|
100.00% |
1 / 1 |
3 | |||
| validateCodepoint | |
100.00% |
6 / 6 |
|
100.00% |
1 / 1 |
10 | |||
| decodeCharReferences | |
100.00% |
5 / 5 |
|
100.00% |
1 / 1 |
1 | |||
| decodeCharReferencesAndNormalize | |
100.00% |
8 / 8 |
|
100.00% |
1 / 1 |
2 | |||
| decodeCharReferencesCallback | |
100.00% |
10 / 10 |
|
100.00% |
1 / 1 |
5 | |||
| decodeChar | |
100.00% |
3 / 3 |
|
100.00% |
1 / 1 |
2 | |||
| decodeEntity | |
75.00% |
3 / 4 |
|
0.00% |
0 / 1 |
2.06 | |||
| attributesAllowedInternal | |
100.00% |
2 / 2 |
|
100.00% |
1 / 1 |
1 | |||
| setupAttributesAllowedInternal | |
2.21% |
3 / 136 |
|
0.00% |
0 / 1 |
5.74 | |||
| stripAllTags | |
100.00% |
9 / 9 |
|
100.00% |
1 / 1 |
1 | |||
| hackDocType | |
0.00% |
0 / 11 |
|
0.00% |
0 / 1 |
20 | |||
| stripIDNs | |
100.00% |
1 / 1 |
|
100.00% |
1 / 1 |
1 | |||
| cleanUrl | |
100.00% |
12 / 12 |
|
100.00% |
1 / 1 |
4 | |||
| validateEmail | |
91.67% |
11 / 12 |
|
0.00% |
0 / 1 |
2.00 | |||
| 1 | <?php |
| 2 | declare( strict_types = 1 ); |
| 3 | |
| 4 | /** |
| 5 | * HTML sanitizer for %MediaWiki. |
| 6 | * |
| 7 | * Copyright © 2002-2005 Brooke Vibber <bvibber@wikimedia.org> et al |
| 8 | * https://www.mediawiki.org/ |
| 9 | * |
| 10 | * @license GPL-2.0-or-later |
| 11 | * @file |
| 12 | * @ingroup Parser |
| 13 | */ |
| 14 | |
| 15 | namespace MediaWiki\Parser; |
| 16 | |
| 17 | use InvalidArgumentException; |
| 18 | use LogicException; |
| 19 | use MediaWiki\HookContainer\HookRunner; |
| 20 | use MediaWiki\MediaWikiServices; |
| 21 | use MediaWiki\Tidy\RemexCompatFormatter; |
| 22 | use UnexpectedValueException; |
| 23 | use Wikimedia\RemexHtml\HTMLData; |
| 24 | use Wikimedia\RemexHtml\Serializer\Serializer as RemexSerializer; |
| 25 | use Wikimedia\RemexHtml\Tokenizer\Tokenizer as RemexTokenizer; |
| 26 | use Wikimedia\RemexHtml\TreeBuilder\Dispatcher as RemexDispatcher; |
| 27 | use Wikimedia\RemexHtml\TreeBuilder\TreeBuilder as RemexTreeBuilder; |
| 28 | use Wikimedia\StringUtils\StringUtils; |
| 29 | |
| 30 | /** |
| 31 | * HTML sanitizer for MediaWiki |
| 32 | * @ingroup Parser |
| 33 | */ |
| 34 | class Sanitizer { |
| 35 | /** |
| 36 | * Regular expression to match various types of character references in |
| 37 | * Sanitizer::normalizeCharReferences and Sanitizer::decodeCharReferences. |
| 38 | * Note that HTML5 allows some named entities to omit the trailing |
| 39 | * semicolon; wikitext entities *must* have a trailing semicolon. |
| 40 | */ |
| 41 | private const CHAR_REFS_REGEX = |
| 42 | '/&([A-Za-z0-9\x80-\xff]+;) |
| 43 | |&\#([0-9]+); |
| 44 | |&\#[xX]([0-9A-Fa-f]+); |
| 45 | |&/x'; |
| 46 | |
| 47 | private const INSECURE_RE = '! expression |
| 48 | | accelerator\s*: |
| 49 | | -o-link\s*: |
| 50 | | -o-link-source\s*: |
| 51 | | -o-replace\s*: |
| 52 | | url\s*\( |
| 53 | | src\s*\( |
| 54 | | image\s*\( |
| 55 | | image-set\s*\( |
| 56 | | attr\s*\([^)]+[\s,]+url |
| 57 | !ix'; |
| 58 | |
| 59 | /** |
| 60 | * Acceptable tag name charset from HTML5 parsing spec |
| 61 | * https://www.w3.org/TR/html5/syntax.html#tag-open-state |
| 62 | */ |
| 63 | private const ELEMENT_BITS_REGEX = '!^(/?)([A-Za-z][^\t\n\v />\0]*+)([^>]*?)(/?>)([^<]*)$!'; |
| 64 | |
| 65 | /** |
| 66 | * Pattern matching evil uris like javascript: |
| 67 | * WARNING: DO NOT use this in any place that actually requires denying |
| 68 | * certain URIs for security reasons. There are NUMEROUS[1] ways to bypass |
| 69 | * pattern-based deny lists; the only way to be secure from javascript: |
| 70 | * uri based xss vectors is to allow only things that you know are safe |
| 71 | * and deny everything else. |
| 72 | * [1]: http://ha.ckers.org/xss.html |
| 73 | */ |
| 74 | private const EVIL_URI_PATTERN = '!(^|\s|\*/\s*)(javascript|vbscript)([^\w]|$)!i'; |
| 75 | private const XMLNS_ATTRIBUTE_PATTERN = "/^xmlns:[:A-Z_a-z-.0-9]+$/"; |
| 76 | |
| 77 | /** |
| 78 | * Tells escapeUrlForHtml() to encode the ID using the wiki's primary encoding. |
| 79 | * |
| 80 | * @since 1.30 |
| 81 | */ |
| 82 | public const ID_PRIMARY = 0; |
| 83 | |
| 84 | /** |
| 85 | * Tells escapeUrlForHtml() to encode the ID using the fallback encoding, or return false |
| 86 | * if no fallback is configured. |
| 87 | * |
| 88 | * @since 1.30 |
| 89 | */ |
| 90 | public const ID_FALLBACK = 1; |
| 91 | |
| 92 | /** Characters that will be ignored in IDNs. |
| 93 | * https://datatracker.ietf.org/doc/html/rfc8264#section-9.13 |
| 94 | * https://www.unicode.org/Public/UCD/latest/ucd/DerivedCoreProperties.txt |
| 95 | * Strip them before further processing so deny lists and such work. |
| 96 | */ |
| 97 | private const IDN_RE_G = "/ |
| 98 | \\s| # general whitespace |
| 99 | \u{00AD}| # SOFT HYPHEN |
| 100 | \u{034F}| # COMBINING GRAPHEME JOINER |
| 101 | \u{061C}| # ARABIC LETTER MARK |
| 102 | [\u{115F}-\u{1160}]| # HANGUL CHOSEONG FILLER.. |
| 103 | # HANGUL JUNGSEONG FILLER |
| 104 | [\u{17B4}-\u{17B5}]| # KHMER VOWEL INHERENT AQ.. |
| 105 | # KHMER VOWEL INHERENT AA |
| 106 | [\u{180B}-\u{180D}]| # MONGOLIAN FREE VARIATION SELECTOR ONE.. |
| 107 | # MONGOLIAN FREE VARIATION SELECTOR THREE |
| 108 | \u{180E}| # MONGOLIAN VOWEL SEPARATOR |
| 109 | [\u{200B}-\u{200F}]| # ZERO WIDTH SPACE.. |
| 110 | # RIGHT-TO-LEFT MARK |
| 111 | [\u{202A}-\u{202E}]| # LEFT-TO-RIGHT EMBEDDING.. |
| 112 | # RIGHT-TO-LEFT OVERRIDE |
| 113 | [\u{2060}-\u{2064}]| # WORD JOINER.. |
| 114 | # INVISIBLE PLUS |
| 115 | \u{2065}| # <reserved-2065> |
| 116 | [\u{2066}-\u{206F}]| # LEFT-TO-RIGHT ISOLATE.. |
| 117 | # NOMINAL DIGIT SHAPES |
| 118 | \u{3164}| # HANGUL FILLER |
| 119 | [\u{FE00}-\u{FE0F}]| # VARIATION SELECTOR-1.. |
| 120 | # VARIATION SELECTOR-16 |
| 121 | \u{FEFF}| # ZERO WIDTH NO-BREAK SPACE |
| 122 | \u{FFA0}| # HALFWIDTH HANGUL FILLER |
| 123 | [\u{FFF0}-\u{FFF8}]| # <reserved-FFF0>.. |
| 124 | # <reserved-FFF8> |
| 125 | [\u{1BCA0}-\u{1BCA3}]| # SHORTHAND FORMAT LETTER OVERLAP.. |
| 126 | # SHORTHAND FORMAT UP STEP |
| 127 | [\u{1D173}-\u{1D17A}]| # MUSICAL SYMBOL BEGIN BEAM.. |
| 128 | # MUSICAL SYMBOL END PHRASE |
| 129 | \u{E0000}| # <reserved-E0000> |
| 130 | \u{E0001}| # LANGUAGE TAG |
| 131 | [\u{E0002}-\u{E001F}]| # <reserved-E0002>.. |
| 132 | # <reserved-E001F> |
| 133 | [\u{E0020}-\u{E007F}]| # TAG SPACE.. |
| 134 | # CANCEL TAG |
| 135 | [\u{E0080}-\u{E00FF}]| # <reserved-E0080>.. |
| 136 | # <reserved-E00FF> |
| 137 | [\u{E0100}-\u{E01EF}]| # VARIATION SELECTOR-17.. |
| 138 | # VARIATION SELECTOR-256 |
| 139 | [\u{E01F0}-\u{E0FFF}]| # <reserved-E01F0>.. |
| 140 | # <reserved-E0FFF> |
| 141 | /xuD"; |
| 142 | |
| 143 | /** |
| 144 | * Character entity aliases accepted by MediaWiki in wikitext. |
| 145 | * These are not part of the HTML standard. |
| 146 | */ |
| 147 | private const MW_ENTITY_ALIASES = [ |
| 148 | 'רלמ;' => 'rlm;', |
| 149 | 'رلم;' => 'rlm;', |
| 150 | ]; |
| 151 | |
| 152 | /** |
| 153 | * Lazy-initialised attributes regex, see getAttribsRegex() |
| 154 | */ |
| 155 | private static ?string $attribsRegex = null; |
| 156 | |
| 157 | /** |
| 158 | * Regular expression to match HTML/XML attribute pairs within a tag. |
| 159 | * Based on https://www.w3.org/TR/html5/syntax.html#before-attribute-name-state |
| 160 | * Used in Sanitizer::decodeTagAttributes |
| 161 | */ |
| 162 | private static function getAttribsRegex(): string { |
| 163 | if ( self::$attribsRegex === null ) { |
| 164 | $spaceChars = '\x09\x0a\x0c\x0d\x20'; |
| 165 | $space = "[{$spaceChars}]"; |
| 166 | $attrib = "[^{$spaceChars}\/>=]"; |
| 167 | $attribFirst = "(?:{$attrib}|=)"; |
| 168 | self::$attribsRegex = |
| 169 | "/({$attribFirst}{$attrib}*) |
| 170 | ($space*=$space* |
| 171 | (?: |
| 172 | # The attribute value: quoted or alone |
| 173 | \"([^\"]*)(?:\"|\$) |
| 174 | | '([^']*)(?:'|\$) |
| 175 | | (((?!$space|>).)*) |
| 176 | ) |
| 177 | )?/sxu"; |
| 178 | } |
| 179 | return self::$attribsRegex; |
| 180 | } |
| 181 | |
| 182 | /** |
| 183 | * Lazy-initialised attribute name regex, see getAttribNameRegex() |
| 184 | */ |
| 185 | private static ?string $attribNameRegex = null; |
| 186 | |
| 187 | /** |
| 188 | * Used in Sanitizer::decodeTagAttributes to filter attributes. |
| 189 | */ |
| 190 | private static function getAttribNameRegex(): string { |
| 191 | if ( self::$attribNameRegex === null ) { |
| 192 | $attribFirst = "[:_\p{L}\p{N}]"; |
| 193 | $attrib = "[:_\.\-\p{L}\p{N}]"; |
| 194 | self::$attribNameRegex = "/^({$attribFirst}{$attrib}*)$/sxu"; |
| 195 | } |
| 196 | return self::$attribNameRegex; |
| 197 | } |
| 198 | |
| 199 | /** |
| 200 | * Return the various lists of recognized tags |
| 201 | * @param string[] $extratags For any extra tags to include |
| 202 | * @param string[] $removetags For any tags (default or extra) to exclude |
| 203 | * @return array |
| 204 | * @internal |
| 205 | */ |
| 206 | public static function getRecognizedTagData( array $extratags = [], array $removetags = [] ): array { |
| 207 | static $commonCase, $staticInitialised = false; |
| 208 | $isCommonCase = ( $extratags === [] && $removetags === [] ); |
| 209 | if ( $staticInitialised && $isCommonCase && $commonCase ) { |
| 210 | return $commonCase; |
| 211 | } |
| 212 | |
| 213 | static $htmlpairsStatic, $htmlsingle, $htmlsingleonly, $htmlnest, $tabletags, |
| 214 | $htmllist, $listtags, $htmlsingleallowed, $htmlelementsStatic; |
| 215 | |
| 216 | if ( !$staticInitialised ) { |
| 217 | $htmlpairsStatic = [ # Tags that must be closed |
| 218 | 'b', 'bdi', 'del', 'i', 'ins', 'u', 'font', 'big', 'small', 'sub', 'sup', 'h1', |
| 219 | 'h2', 'h3', 'h4', 'h5', 'h6', 'cite', 'code', 'em', 's', |
| 220 | 'strike', 'strong', 'tt', 'var', 'div', 'center', |
| 221 | 'blockquote', 'ol', 'ul', 'dl', 'table', 'caption', 'pre', |
| 222 | 'ruby', 'rb', 'rp', 'rt', 'rtc', 'p', 'span', 'abbr', 'dfn', |
| 223 | 'kbd', 'samp', 'data', 'time', 'mark' |
| 224 | ]; |
| 225 | # These tags can be self-closed. For tags not also on |
| 226 | # $htmlsingleonly, a self-closed tag will be emitted as |
| 227 | # an empty element (open-tag/close-tag pair). |
| 228 | $htmlsingle = [ |
| 229 | 'br', 'wbr', 'hr', 'li', 'dt', 'dd', 'meta', 'link' |
| 230 | ]; |
| 231 | |
| 232 | # Elements that cannot have close tags. This is (not coincidentally) |
| 233 | # also the list of tags for which the HTML 5 parsing algorithm |
| 234 | # requires you to "acknowledge the token's self-closing flag", i.e. |
| 235 | # a self-closing tag like <br/> is not an HTML 5 parse error only |
| 236 | # for this list. |
| 237 | $htmlsingleonly = [ |
| 238 | 'br', 'wbr', 'hr', 'meta', 'link' |
| 239 | ]; |
| 240 | |
| 241 | $htmlnest = [ # Tags that can be nested--?? |
| 242 | 'table', 'tr', 'td', 'th', 'div', 'blockquote', 'ol', 'ul', |
| 243 | 'li', 'dl', 'dt', 'dd', 'font', 'big', 'small', 'sub', 'sup', 'span', |
| 244 | 'var', 'kbd', 'samp', 'em', 'strong', 'q', 'ruby', 'bdo' |
| 245 | ]; |
| 246 | $tabletags = [ # Can only appear inside table, we will close them |
| 247 | 'td', 'th', 'tr', |
| 248 | ]; |
| 249 | $htmllist = [ # Tags used by list |
| 250 | 'ul', 'ol', |
| 251 | ]; |
| 252 | $listtags = [ # Tags that can appear in a list |
| 253 | 'li', |
| 254 | ]; |
| 255 | |
| 256 | $htmlsingleallowed = array_unique( array_merge( $htmlsingle, $tabletags ) ); |
| 257 | $htmlelementsStatic = array_unique( array_merge( $htmlsingle, $htmlpairsStatic, $htmlnest ) ); |
| 258 | |
| 259 | # Convert them all to hashtables for faster lookup |
| 260 | $vars = [ 'htmlpairsStatic', 'htmlsingle', 'htmlsingleonly', 'htmlnest', 'tabletags', |
| 261 | 'htmllist', 'listtags', 'htmlsingleallowed', 'htmlelementsStatic' ]; |
| 262 | foreach ( $vars as $var ) { |
| 263 | $$var = array_fill_keys( $$var, true ); |
| 264 | } |
| 265 | $staticInitialised = true; |
| 266 | } |
| 267 | |
| 268 | # Populate $htmlpairs and $htmlelements with the $extratags and $removetags arrays |
| 269 | $extratags = array_fill_keys( $extratags, true ); |
| 270 | $removetags = array_fill_keys( $removetags, true ); |
| 271 | $htmlpairs = array_merge( $extratags, $htmlpairsStatic ); |
| 272 | $htmlelements = array_diff_key( array_merge( $extratags, $htmlelementsStatic ), $removetags ); |
| 273 | |
| 274 | $result = [ |
| 275 | 'htmlpairs' => $htmlpairs, |
| 276 | 'htmlsingle' => $htmlsingle, |
| 277 | 'htmlsingleonly' => $htmlsingleonly, |
| 278 | 'htmlnest' => $htmlnest, |
| 279 | 'tabletags' => $tabletags, |
| 280 | 'htmllist' => $htmllist, |
| 281 | 'listtags' => $listtags, |
| 282 | 'htmlsingleallowed' => $htmlsingleallowed, |
| 283 | 'htmlelements' => $htmlelements, |
| 284 | ]; |
| 285 | if ( $isCommonCase ) { |
| 286 | $commonCase = $result; |
| 287 | } |
| 288 | return $result; |
| 289 | } |
| 290 | |
| 291 | /** |
| 292 | * Cleans up HTML, removes dangerous tags and attributes, and |
| 293 | * removes HTML comments; BEWARE there may be unmatched HTML |
| 294 | * tags in the result. |
| 295 | * |
| 296 | * @note Callers are recommended to use `::removeSomeTags()` instead |
| 297 | * of this method. `Sanitizer::removeSomeTags()` is safer and will |
| 298 | * always return well-formed HTML; however, it is significantly |
| 299 | * slower (especially for short strings where setup costs |
| 300 | * predominate). This method is for internal use by the legacy parser |
| 301 | * where we know the result will be cleaned up in a subsequent tidy pass. |
| 302 | * |
| 303 | * @param string $text Original string; see T268353 for why untainted. |
| 304 | * @param-taint $text none |
| 305 | * @param callable|null $processCallback Callback to do any variable or |
| 306 | * parameter replacements in HTML attribute values. |
| 307 | * This argument should be considered @internal. |
| 308 | * @param-taint $processCallback exec_shell |
| 309 | * @param array|bool $args Arguments for the processing callback |
| 310 | * @param-taint $args none |
| 311 | * @param array $extratags For any extra tags to include |
| 312 | * @param-taint $extratags tainted |
| 313 | * @param array $removetags For any tags (default or extra) to exclude |
| 314 | * @param-taint $removetags none |
| 315 | * @return string |
| 316 | * @return-taint escaped |
| 317 | * @internal |
| 318 | */ |
| 319 | public static function internalRemoveHtmlTags( string $text, ?callable $processCallback = null, |
| 320 | $args = [], array $extratags = [], array $removetags = [] |
| 321 | ): string { |
| 322 | $tagData = self::getRecognizedTagData( $extratags, $removetags ); |
| 323 | $htmlsingle = $tagData['htmlsingle']; |
| 324 | $htmlsingleonly = $tagData['htmlsingleonly']; |
| 325 | $htmlelements = $tagData['htmlelements']; |
| 326 | |
| 327 | # Remove HTML comments |
| 328 | $text = self::removeHTMLcomments( $text ); |
| 329 | $bits = explode( '<', $text ); |