Skip to content

Commit 717e02d

Browse files
committed
HTML API: Ensure attribute value prefix matches the full search text.
`WP_HTML_Decoder::attribute_starts_with()` incorrectly reported a match when the attribute value was exhausted before the search text. Developed in WordPress#12380. Props jonsurrell, dmsnell. See #65372. git-svn-id: https://develop.svn.wordpress.org/trunk@62711 602fd350-edb4-49c9-b593-d223f7449a82
1 parent 446655f commit 717e02d

2 files changed

Lines changed: 168 additions & 5 deletions

File tree

src/wp-includes/html-api/class-wp-html-decoder.php

Lines changed: 55 additions & 5 deletions
Original file line numberDiff line numberDiff line change
@@ -17,6 +17,9 @@ class WP_HTML_Decoder {
1717
* of how it might be encoded in HTML. For instance, `http:` could be represented as `http:`
1818
* or as `http:` or as `http:` or as `http:`, or in many other ways.
1919
*
20+
* This is equivalent to a byte-prefix test against the decoded attribute value, without
21+
* the need to allocate and decode the full string.
22+
*
2023
* Example:
2124
*
2225
* $value = 'http://wordpress.org/';
@@ -53,24 +56,71 @@ public static function attribute_starts_with( $haystack, $search_text, $case_sen
5356
return false;
5457
}
5558

56-
// If there's no character reference but the character do match, then it could still match.
59+
// If there's no character reference but the characters do match, then it could still match.
5760
if ( null === $next_chunk && $chars_match ) {
5861
++$haystack_at;
5962
++$search_at;
6063
continue;
6164
}
6265

63-
// If there is a character reference, then the decoded value must exactly match what follows in the search string.
64-
if ( 0 !== substr_compare( $search_text, $next_chunk, $search_at, strlen( $next_chunk ), $loose_case ) ) {
66+
/**
67+
* The decoded character reference in `$next_chunk` must be compared with the
68+
* corresponding `$search_text` bytes checking for matching prefixes. The remaining
69+
* search text may be shorter than the decoded chunk, in which case a partial match
70+
* satisfies the prefix. Otherwise, if the decoded chunk is fully matched, the
71+
* comparison must continue after advancing the appropriate byte lengths: the character
72+
* reference token length in the haystack and the decoded chunk length in the
73+
* search text.
74+
*
75+
* For example, consider searches that have reached the character reference
76+
* `fj` (7 bytes), decoded into the 2-byte chunk `fj`:
77+
*
78+
* $haystack_at
79+
* │
80+
* │ ┌─after matching `fj` continue here
81+
* │ │ (advance by $token_length, 7 bytes)
82+
* ↓ ↓
83+
* Haystack: startfjord
84+
* ╰──┬──╯
85+
* fj - the decoded chunk, tested against the search text.
86+
*
87+
* $search_at
88+
* │
89+
* │ ┌─after matching `fj` continue here
90+
* │ │ (advance by $match_length, 2 bytes)
91+
* ↓ ↓
92+
* Search A: startfjord Compare 2 bytes: `fj` matches,
93+
* continue matching at `o`.
94+
*
95+
* $search_at
96+
* ↓
97+
* Search B: startf Compare 1 byte: `f` matches and the
98+
* search text is exhausted — prefix confirmed.
99+
*
100+
* $search_at
101+
* ↓
102+
* Search C: startfr Compare 2 bytes: `fj` differs
103+
* from `fr`, no match is possible.
104+
*
105+
* The `min()` is required in both directions: Search A fails if the
106+
* comparison length comes from the search text, Search B if it comes
107+
* from the chunk.
108+
*
109+
* After a match each cursor must advance by the appropriate length, the haystack
110+
* cursor by the character reference token length, and the search cursor by the
111+
* matched length.
112+
*/
113+
$match_length = min( strlen( $next_chunk ), $search_length - $search_at );
114+
if ( 0 !== substr_compare( $search_text, $next_chunk, $search_at, $match_length, $loose_case ) ) {
65115
return false;
66116
}
67117

68118
// The character reference matched, so continue checking.
69119
$haystack_at += $token_length;
70-
$search_at += strlen( $next_chunk );
120+
$search_at += $match_length;
71121
}
72122

73-
return true;
123+
return $search_at === $search_length;
74124
}
75125

76126
/**

tests/phpunit/tests/html-api/wpHtmlDecoder.php

Lines changed: 113 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -346,6 +346,119 @@ public static function data_case_variants_of_attribute_prefixes() {
346346
}
347347
}
348348

349+
/**
350+
* Ensures that `attribute_starts_with` checks the full search string.
351+
*
352+
* @ticket 65372
353+
*
354+
* @dataProvider data_attribute_starts_with_search_string_boundaries
355+
*
356+
* @param string $attribute_value Raw attribute value from HTML string.
357+
* @param string $search_string Prefix contained or not contained in encoded attribute value.
358+
* @param string $case_sensitivity Whether to search with ASCII case sensitivity;
359+
* 'ascii-case-insensitive' or 'case-sensitive'.
360+
* @param bool $is_match Whether the search string is a prefix for the attribute value.
361+
*/
362+
public function test_attribute_starts_with_checks_search_string_boundaries(
363+
string $attribute_value,
364+
string $search_string,
365+
string $case_sensitivity,
366+
bool $is_match
367+
): void {
368+
if ( $is_match ) {
369+
$this->assertTrue(
370+
WP_HTML_Decoder::attribute_starts_with( $attribute_value, $search_string, $case_sensitivity ),
371+
'Should have matched attribute prefix.'
372+
);
373+
} else {
374+
$this->assertFalse(
375+
WP_HTML_Decoder::attribute_starts_with( $attribute_value, $search_string, $case_sensitivity ),
376+
'Should not have matched attribute with prefix.'
377+
);
378+
}
379+
}
380+
381+
/**
382+
* Data provider.
383+
*
384+
* @return Generator<string, array{string, string, string, bool}> Test cases.
385+
*/
386+
public static function data_attribute_starts_with_search_string_boundaries(): Generator {
387+
yield 'Empty attribute does not match non-empty prefix' => array( '', 'http', 'case-sensitive', false );
388+
yield 'Short attribute does not match longer prefix' => array(
389+
'java',
390+
'javascript',
391+
'case-sensitive',
392+
false,
393+
);
394+
yield 'Attribute ending in a character reference does not match a longer prefix' => array(
395+
'&amp;',
396+
'&&',
397+
'case-sensitive',
398+
false,
399+
);
400+
yield 'Longer attribute matches shorter prefix' => array(
401+
'javascript',
402+
'java',
403+
'case-sensitive',
404+
true,
405+
);
406+
yield "&fjlig; (decodes to 2-codepoint 'fj') starts with f" => array(
407+
'&fjlig; is literally "f" followed by "j"',
408+
'f',
409+
'case-sensitive',
410+
true,
411+
);
412+
yield "&nvlt; (decodes to 2-codepoint '<⃒') starts with '<'" => array(
413+
'&nvlt;script>',
414+
'<',
415+
'case-sensitive',
416+
true,
417+
);
418+
yield "Combining character references (¬̸) full match on '¬̸' prefix" => array(
419+
'&not;&#x338; A negated not?',
420+
'¬̸',
421+
'case-sensitive',
422+
true,
423+
);
424+
yield "Combining character references (¬̸) partial match on '¬' prefix" => array(
425+
'&not;&#x338; A negated not?',
426+
'¬',
427+
'case-sensitive',
428+
true,
429+
);
430+
yield 'Search A: prefix continues past a decoded character reference' => array(
431+
'start&fjlig;ord',
432+
'startfjord',
433+
'case-sensitive',
434+
true,
435+
);
436+
yield 'Search B: prefix ends part-way through a decoded character reference' => array(
437+
'start&fjlig;ord',
438+
'startf',
439+
'case-sensitive',
440+
true,
441+
);
442+
yield 'Search C: prefix mismatches within a decoded character reference' => array(
443+
'start&fjlig;ord',
444+
'startfr',
445+
'case-sensitive',
446+
false,
447+
);
448+
yield 'ASCII-case-insensitive prefix ends part-way through a decoded character reference' => array(
449+
'start&fjlig;ord',
450+
'STARTF',
451+
'ascii-case-insensitive',
452+
true,
453+
);
454+
yield 'ASCII-case-insensitive prefix mismatches within a decoded character reference' => array(
455+
'start&fjlig;ord',
456+
'STARTFR',
457+
'ascii-case-insensitive',
458+
false,
459+
);
460+
}
461+
349462
/**
350463
* Ensures that `attribute_starts_with` respects the case sensitivity argument.
351464
*

0 commit comments

Comments
 (0)