diff --git a/Libraries/LibJS/Runtime/GlobalObject.cpp b/Libraries/LibJS/Runtime/GlobalObject.cpp index b74ee06761..907995ac0c 100644 --- a/Libraries/LibJS/Runtime/GlobalObject.cpp +++ b/Libraries/LibJS/Runtime/GlobalObject.cpp @@ -588,36 +588,43 @@ JS_DEFINE_NATIVE_FUNCTION(GlobalObject::escape) JS_DEFINE_NATIVE_FUNCTION(GlobalObject::unescape) { // 1. Set string to ? ToString(string). - auto string = TRY(vm.argument(0).to_byte_string(vm)); + auto string = TRY(vm.argument(0).to_utf16_string(vm)); // 2. Let length be the length of string. - ssize_t length = string.length(); + ssize_t length = string.length_in_code_units(); // 3. Let R be the empty String. - StringBuilder unescaped(length); + Utf16StringBuilder unescaped(length); // 4. Let k be 0. // 5. Repeat, while k ≠ length, for (auto k = 0; k < length; ++k) { // a. Let c be the code unit at index k within string. - u32 code_point = string[k]; + u16 code_unit = string.code_unit_at(k); // b. If c is the code unit 0x0025 (PERCENT SIGN), then - if (code_point == '%') { + if (code_unit == '%') { // i. Let hexEscape be the empty String. // ii. Let skip be 0. // iii. If k ≤ length - 6 and the code unit at index k + 1 within string is the code unit 0x0075 (LATIN SMALL LETTER U), then - if (k <= length - 6 && string[k + 1] == 'u' && is_ascii_hex_digit(string[k + 2]) && is_ascii_hex_digit(string[k + 3]) && is_ascii_hex_digit(string[k + 4]) && is_ascii_hex_digit(string[k + 5])) { + if (k <= length - 6 + && string.code_unit_at(k + 1) == 'u' + && is_ascii_hex_digit(string.code_unit_at(k + 2)) + && is_ascii_hex_digit(string.code_unit_at(k + 3)) + && is_ascii_hex_digit(string.code_unit_at(k + 4)) + && is_ascii_hex_digit(string.code_unit_at(k + 5))) { // 1. Set hexEscape to the substring of string from k + 2 to k + 6. - code_point = (parse_ascii_hex_digit(string[k + 2]) << 12) | (parse_ascii_hex_digit(string[k + 3]) << 8) | (parse_ascii_hex_digit(string[k + 4]) << 4) | parse_ascii_hex_digit(string[k + 5]); + code_unit = (parse_ascii_hex_digit(string.code_unit_at(k + 2)) << 12) | (parse_ascii_hex_digit(string.code_unit_at(k + 3)) << 8) | (parse_ascii_hex_digit(string.code_unit_at(k + 4)) << 4) | parse_ascii_hex_digit(string.code_unit_at(k + 5)); // 2. Set skip to 5. k += 5; } // iv. Else if k ≤ length - 3, then - else if (k <= length - 3 && is_ascii_hex_digit(string[k + 1]) && is_ascii_hex_digit(string[k + 2])) { + else if (k <= length - 3 + && is_ascii_hex_digit(string.code_unit_at(k + 1)) + && is_ascii_hex_digit(string.code_unit_at(k + 2))) { // 1. Set hexEscape to the substring of string from k + 1 to k + 3. - code_point = (parse_ascii_hex_digit(string[k + 1]) << 4) | parse_ascii_hex_digit(string[k + 2]); + code_unit = (parse_ascii_hex_digit(string.code_unit_at(k + 1)) << 4) | parse_ascii_hex_digit(string.code_unit_at(k + 2)); // 2. Set skip to 2. k += 2; @@ -632,13 +639,13 @@ JS_DEFINE_NATIVE_FUNCTION(GlobalObject::unescape) } // c. Set R to the string-concatenation of R and c. - unescaped.append_code_point(code_point); + unescaped.append_code_unit(code_unit); // d. Set k to k + 1. } // 6. Return R. - return PrimitiveString::create(vm, unescaped.to_byte_string()); + return PrimitiveString::create(vm, unescaped.to_string()); } } diff --git a/Libraries/LibJS/Runtime/RegExpPrototype.cpp b/Libraries/LibJS/Runtime/RegExpPrototype.cpp index 274fc2cc76..64c2630c25 100644 --- a/Libraries/LibJS/Runtime/RegExpPrototype.cpp +++ b/Libraries/LibJS/Runtime/RegExpPrototype.cpp @@ -78,9 +78,23 @@ static ThrowCompletionOr increment_last_index(VM& vm, Object& regexp_objec } // FIXME: Add an eviction policy to bound the size of this cache. +struct RegexCacheKey { + Utf16String pattern; + RegExpObject::Flags flags; + + bool operator==(RegexCacheKey const&) const = default; +}; + +struct RegexCacheKeyTraits : public Traits { + static unsigned hash(RegexCacheKey const& key) + { + return pair_int_hash(key.pattern.hash(), to_underlying(key.flags)); + } +}; + static auto& regex_cache() { - static NeverDestroyed>> cache; + static NeverDestroyed, RegexCacheKeyTraits>> cache; return *cache; } @@ -93,13 +107,7 @@ static regex::ECMAScriptRegex const* get_or_compile_regex(RegExpObject& regexp_o auto const& pattern = regexp_object.pattern(); auto flag_bits = regexp_object.flag_bits(); - // Build a cache key from pattern + flag bits. - StringBuilder key_builder; - key_builder.append('/'); - key_builder.append(pattern.utf16_view()); - key_builder.append('/'); - key_builder.append_code_point(static_cast(flag_bits)); - auto cache_key = key_builder.to_string_without_validation(); + RegexCacheKey cache_key { pattern, flag_bits }; if (auto it = regex_cache().find(cache_key); it != regex_cache().end()) { auto* ptr = it->value.ptr(); @@ -110,7 +118,7 @@ static regex::ECMAScriptRegex const* get_or_compile_regex(RegExpObject& regexp_o bool unicode = has_flag(flag_bits, RegExpObject::Flags::Unicode); bool unicode_sets = has_flag(flag_bits, RegExpObject::Flags::UnicodeSets); - // Parse the pattern from UTF-16 source to UTF-8 with escape normalization. + // Normalize non-ASCII code units to ASCII escapes before compiling the pattern. auto parsed_pattern = parse_regex_pattern(pattern.utf16_view(), unicode, unicode_sets); if (parsed_pattern.is_error()) return nullptr; @@ -132,7 +140,7 @@ static regex::ECMAScriptRegex const* get_or_compile_regex(RegExpObject& regexp_o auto owned = make(compiled.release_value()); auto* ptr = owned.ptr(); - regex_cache().set(cache_key, move(owned)); + regex_cache().set(move(cache_key), move(owned)); regexp_object.set_cached_regex(ptr); return ptr; } diff --git a/Tests/LibJS/Runtime/builtins/functions/escapeUnescape.js b/Tests/LibJS/Runtime/builtins/functions/escapeUnescape.js index 69862fa426..d78f4e5d7a 100644 --- a/Tests/LibJS/Runtime/builtins/functions/escapeUnescape.js +++ b/Tests/LibJS/Runtime/builtins/functions/escapeUnescape.js @@ -16,6 +16,9 @@ test("unescape", () => { ["%E4%F6%FC", "äöü"], ["%u0107", "ć"], ["@*_+-./", "@*_+-./"], + ["äöü", "äöü"], + ["%uD834%uDF06", "\ud834\udf06"], + ["%uD800", "\ud800"], ].forEach(test => { expect(unescape(test[0])).toBe(test[1]); });