From b624276e72db99983841e3aaf3df9fed2c07e174 Mon Sep 17 00:00:00 2001 From: Shannon Booth Date: Sun, 5 Apr 2026 16:41:54 +0200 Subject: [PATCH] LibURL: Remove C++ URLPattern implementation --- Libraries/LibURL/CMakeLists.txt | 10 - Libraries/LibURL/Pattern/Canonicalization.cpp | 262 ------- Libraries/LibURL/Pattern/Canonicalization.h | 25 - Libraries/LibURL/Pattern/Component.cpp | 340 --------- Libraries/LibURL/Pattern/Component.h | 56 -- .../Pattern/ConstructorStringParser.cpp | 713 ------------------ .../LibURL/Pattern/ConstructorStringParser.h | 106 --- Libraries/LibURL/Pattern/Init.cpp | 367 --------- Libraries/LibURL/Pattern/Init.h | 45 -- Libraries/LibURL/Pattern/Options.cpp | 41 - Libraries/LibURL/Pattern/Options.h | 30 - Libraries/LibURL/Pattern/Part.cpp | 64 -- Libraries/LibURL/Pattern/Part.h | 79 -- Libraries/LibURL/Pattern/Pattern.cpp | 434 ----------- Libraries/LibURL/Pattern/Pattern.h | 95 --- Libraries/LibURL/Pattern/PatternError.h | 23 - Libraries/LibURL/Pattern/PatternParser.cpp | 409 ---------- Libraries/LibURL/Pattern/PatternParser.h | 74 -- Libraries/LibURL/Pattern/String.cpp | 310 -------- Libraries/LibURL/Pattern/String.h | 23 - Libraries/LibURL/Pattern/Tokenizer.cpp | 440 ----------- Libraries/LibURL/Pattern/Tokenizer.h | 117 --- Tests/LibURL/CMakeLists.txt | 1 - .../TestURLPatternConstructorStringParser.cpp | 174 ----- 24 files changed, 4238 deletions(-) delete mode 100644 Libraries/LibURL/Pattern/Canonicalization.cpp delete mode 100644 Libraries/LibURL/Pattern/Canonicalization.h delete mode 100644 Libraries/LibURL/Pattern/Component.cpp delete mode 100644 Libraries/LibURL/Pattern/Component.h delete mode 100644 Libraries/LibURL/Pattern/ConstructorStringParser.cpp delete mode 100644 Libraries/LibURL/Pattern/ConstructorStringParser.h delete mode 100644 Libraries/LibURL/Pattern/Init.cpp delete mode 100644 Libraries/LibURL/Pattern/Init.h delete mode 100644 Libraries/LibURL/Pattern/Options.cpp delete mode 100644 Libraries/LibURL/Pattern/Options.h delete mode 100644 Libraries/LibURL/Pattern/Part.cpp delete mode 100644 Libraries/LibURL/Pattern/Part.h delete mode 100644 Libraries/LibURL/Pattern/Pattern.cpp delete mode 100644 Libraries/LibURL/Pattern/Pattern.h delete mode 100644 Libraries/LibURL/Pattern/PatternError.h delete mode 100644 Libraries/LibURL/Pattern/PatternParser.cpp delete mode 100644 Libraries/LibURL/Pattern/PatternParser.h delete mode 100644 Libraries/LibURL/Pattern/String.cpp delete mode 100644 Libraries/LibURL/Pattern/String.h delete mode 100644 Libraries/LibURL/Pattern/Tokenizer.cpp delete mode 100644 Libraries/LibURL/Pattern/Tokenizer.h delete mode 100644 Tests/LibURL/TestURLPatternConstructorStringParser.cpp diff --git a/Libraries/LibURL/CMakeLists.txt b/Libraries/LibURL/CMakeLists.txt index 659a69f85a..a996a95a02 100644 --- a/Libraries/LibURL/CMakeLists.txt +++ b/Libraries/LibURL/CMakeLists.txt @@ -8,16 +8,6 @@ set(SOURCES Site.cpp URL.cpp ${PUBLIC_SUFFIX_SOURCES} - Pattern/Canonicalization.cpp - Pattern/Component.cpp - Pattern/ConstructorStringParser.cpp - Pattern/Init.cpp - Pattern/Options.cpp - Pattern/Part.cpp - Pattern/Pattern.cpp - Pattern/PatternParser.cpp - Pattern/String.cpp - Pattern/Tokenizer.cpp ) ladybird_lib(LibURL url) diff --git a/Libraries/LibURL/Pattern/Canonicalization.cpp b/Libraries/LibURL/Pattern/Canonicalization.cpp deleted file mode 100644 index e278df3166..0000000000 --- a/Libraries/LibURL/Pattern/Canonicalization.cpp +++ /dev/null @@ -1,262 +0,0 @@ -/* - * Copyright (c) 2025, Shannon Booth - * - * SPDX-License-Identifier: BSD-2-Clause - */ - -#include -#include - -namespace URL::Pattern { - -// https://urlpattern.spec.whatwg.org/#url-pattern-create-a-dummy-url -static URL create_a_dummy_url() -{ - // 1. Let dummyInput be "https://dummy.invalid/". - // 2. Return the result of running the basic URL parser on dummyInput. - return Parser::basic_parse("https://dummy.invalid/"sv).release_value(); -} - -// https://urlpattern.spec.whatwg.org/#canonicalize-a-protocol -PatternErrorOr canonicalize_a_protocol(String const& value) -{ - // 1. If value is the empty string, return value. - if (value.is_empty()) - return value; - - // 2. Let parseResult be the result of running the basic URL parser given value followed by "://dummy.invalid/". - // NOTE: Note, state override is not used here because it enforces restrictions that are only appropriate for the - // protocol setter. Instead we use the protocol to parse a dummy URL using the normal parsing entry point. - auto parse_result = Parser::basic_parse(MUST(String::formatted("{}://dummy.invalid", value))); - - // 4. If parseResult is failure, then throw a TypeError. - if (!parse_result.has_value()) - return ErrorInfo { "Failed to canonicalize URL protocol string"_string }; - - // 5. Return parseResult’s scheme. - return parse_result->scheme(); -} - -// https://urlpattern.spec.whatwg.org/#canonicalize-a-username -String canonicalize_a_username(String const& value) -{ - // 1. If value is the empty string, return value. - if (value.is_empty()) - return value; - - // 2. Let dummyURL be the result of creating a dummy URL. - auto dummy_url = create_a_dummy_url(); - - // 3. Set the username given dummyURL and value. - dummy_url.set_username(value); - - // 4. Return dummyURL’s username. - return dummy_url.username(); -} - -// https://urlpattern.spec.whatwg.org/#canonicalize-a-password -String canonicalize_a_password(String const& value) -{ - // 1. If value is the empty string, return value. - if (value.is_empty()) - return value; - - // 2. Let dummyURL be the result of creating a dummy URL. - auto dummy_url = create_a_dummy_url(); - - // 3. Set the password given dummyURL and value. - dummy_url.set_password(value); - - // 4. Return dummyURL’s password. - return dummy_url.password(); -} - -// https://urlpattern.spec.whatwg.org/#canonicalize-a-hostname -PatternErrorOr canonicalize_a_hostname(String const& value) -{ - // 1. If value is the empty string, return value. - if (value.is_empty()) - return value; - - // 2. Let dummyURL be the result of creating a dummy URL. - auto dummy_url = create_a_dummy_url(); - - // 3. Let parseResult be the result of running the basic URL parser given value with dummyURL - // as url and hostname state as state override. - auto parse_result = Parser::basic_parse(value, {}, &dummy_url, Parser::State::Hostname); - - // 4. If parseResult is failure, then throw a TypeError. - if (!parse_result.has_value()) - return ErrorInfo { "Failed to canonicalize URL hostname string"_string }; - - // 5. Return dummyURL’s host, serialized, or empty string if it is null. - if (!dummy_url.host().has_value()) - return String {}; - return dummy_url.host()->serialize(); -} - -// https://urlpattern.spec.whatwg.org/#canonicalize-an-ipv6-hostname -PatternErrorOr canonicalize_an_ipv6_hostname(String const& value) -{ - // 1. Let result be the empty string. - StringBuilder result; - - // 2. For each code point in value interpreted as a list of code points: - for (auto code_point : value.code_points()) { - // 1. If all of the following are true: - // * code point is not an ASCII hex digit; - // * code point is not U+005B ([); - // * code point is not U+005D (]); and - // * code point is not U+003A (:), - // then throw a TypeError. - if (!is_ascii_hex_digit(code_point) - && code_point != '[' - && code_point != ']' - && code_point != ':') { - return ErrorInfo { "Failed to canonicalize IPv6 hostname string"_string }; - } - - // 2. Append the result of running ASCII lowercase given code point to the end of result. - result.append(to_ascii_lowercase(code_point)); - } - - // 3. Return result. - return result.to_string_without_validation(); -} - -// https://urlpattern.spec.whatwg.org/#canonicalize-a-port -PatternErrorOr canonicalize_a_port(String const& port_value, Optional const& protocol_value) -{ - // 1. If portValue is the empty string, return portValue. - if (port_value.is_empty()) - return port_value; - - // 2. Let dummyURL be the result of creating a dummy URL. - auto dummy_url = create_a_dummy_url(); - - // 3. If protocolValue was given, then set dummyURL’s scheme to protocolValue. - // NOTE: Note, we set the URL record's scheme in order for the basic URL parser to - // recognize and normalize default port values. - if (protocol_value.has_value()) - dummy_url.set_scheme(protocol_value.value()); - - // 4. Let parseResult be the result of running basic URL parser given portValue with dummyURL - // as url and port state as state override. - auto parse_result = Parser::basic_parse(port_value, {}, &dummy_url, Parser::State::Port); - - // 4. If parseResult is failure, then throw a TypeError. - if (!parse_result.has_value()) - return ErrorInfo { "Failed to canonicalize port string"_string }; - - // 5. Return dummyURL’s port, serialized, or empty string if it is null. - if (!dummy_url.port().has_value()) - return String {}; - return String::number(*dummy_url.port()); -} - -// https://urlpattern.spec.whatwg.org/#canonicalize-a-pathname -String canonicalize_a_pathname(String const& value) -{ - // 1. If value is the empty string, then return value. - if (value.is_empty()) - return value; - - // 2. Let leading slash be true if the first code point in value is U+002F (/) and otherwise false. - bool leading_slash = value.bytes()[0] == '/'; - - // 3. Let modified value be "/-" if leading slash is false and otherwise the empty string. - StringBuilder modified_value; - if (!leading_slash) - modified_value.append("/-"sv); - - // 4. Append value to the end of modified value. - modified_value.append(value); - - // 5. Let dummyURL be the result of creating a dummy URL. - auto dummy_url = create_a_dummy_url(); - - // 6. Empty dummyURL’s path. - dummy_url.set_paths({}); - - // 7. Run basic URL parser given modified value with dummyURL as url and path start state as state override. - (void)Parser::basic_parse(modified_value.string_view(), {}, &dummy_url, Parser::State::PathStart); - - // 8. Let result be the result of URL path serializing dummyURL. - auto result = dummy_url.serialize_path(); - - // 9. If leading slash is false, then set result to the code point substring from 2 to the end of the string within result. - if (!leading_slash) - result = MUST(String::from_utf8(result.code_points().unicode_substring_view(2).as_string())); - - // 10. Return result. - return result; -} - -// https://urlpattern.spec.whatwg.org/#canonicalize-an-opaque-pathname -PatternErrorOr canonicalize_an_opaque_pathname(String const& value) -{ - // 1. If value is the empty string, return value. - if (value.is_empty()) - return value; - - // 2. Let dummyURL be the result of creating a dummy URL. - auto dummy_url = create_a_dummy_url(); - - // 3. Set dummyURL’s path to the empty string. - dummy_url.set_paths({ "" }); - dummy_url.set_has_an_opaque_path(true); - - // 4. Let parseResult be the result of running URL parsing given value with dummyURL as url and opaque path state as state override. - auto parse_result = Parser::basic_parse(value, {}, &dummy_url, Parser::State::OpaquePath); - - // 5. If parseResult is failure, then throw a TypeError. - if (!parse_result.has_value()) - return ErrorInfo { "Failed to canonicalize opaque pathname string"_string }; - - // 6. Return the result of URL path serializing dummyURL. - return dummy_url.serialize_path(); -} - -// https://urlpattern.spec.whatwg.org/#canonicalize-a-search -String canonicalize_a_search(String const& value) -{ - // 1. If value is the empty string, return value. - if (value.is_empty()) - return value; - - // 2. Let dummyURL be the result of creating a dummy URL. - auto dummy_url = create_a_dummy_url(); - - // 3. Set dummyURL’s query to the empty string. - dummy_url.set_query(String {}); - - // 4. Run basic URL parser given value with dummyURL as url and query state as state override. - (void)Parser::basic_parse(value, {}, &dummy_url, Parser::State::Query); - - // 5. Return dummyURL’s query. - VERIFY(dummy_url.query().has_value()); - return *dummy_url.query(); -} - -// https://urlpattern.spec.whatwg.org/#canonicalize-a-hash -String canonicalize_a_hash(String const& value) -{ - // 1. If value is the empty string, return value. - if (value.is_empty()) - return value; - - // 2. Let dummyURL be the result of creating a dummy URL. - auto dummy_url = create_a_dummy_url(); - - // 3. Set dummyURL’s fragment to the empty string. - dummy_url.set_fragment(String {}); - - // 4. Run basic URL parser given value with dummyURL as url and fragment state as state override. - (void)Parser::basic_parse(value, {}, &dummy_url, Parser::State::Fragment); - - // 5. Return dummyURL’s fragment. - VERIFY(dummy_url.fragment().has_value()); - return *dummy_url.fragment(); -} - -} diff --git a/Libraries/LibURL/Pattern/Canonicalization.h b/Libraries/LibURL/Pattern/Canonicalization.h deleted file mode 100644 index ef4b9d93d5..0000000000 --- a/Libraries/LibURL/Pattern/Canonicalization.h +++ /dev/null @@ -1,25 +0,0 @@ -/* - * Copyright (c) 2025, Shannon Booth - * - * SPDX-License-Identifier: BSD-2-Clause - */ - -#pragma once - -#include -#include - -namespace URL::Pattern { - -PatternErrorOr canonicalize_a_protocol(String const&); -String canonicalize_a_username(String const&); -String canonicalize_a_password(String const&); -PatternErrorOr canonicalize_a_hostname(String const&); -PatternErrorOr canonicalize_an_ipv6_hostname(String const&); -PatternErrorOr canonicalize_a_port(String const&, Optional const& protocol_value = {}); -String canonicalize_a_pathname(String const&); -PatternErrorOr canonicalize_an_opaque_pathname(String const&); -String canonicalize_a_search(String const&); -String canonicalize_a_hash(String const&); - -} diff --git a/Libraries/LibURL/Pattern/Component.cpp b/Libraries/LibURL/Pattern/Component.cpp deleted file mode 100644 index 668de358ed..0000000000 --- a/Libraries/LibURL/Pattern/Component.cpp +++ /dev/null @@ -1,340 +0,0 @@ -/* - * Copyright (c) 2025-2026, Shannon Booth - * - * SPDX-License-Identifier: BSD-2-Clause - */ - -#include -#include -#include -#include -#include -#include - -namespace URL::Pattern { - -// https://urlpattern.spec.whatwg.org/#protocol-component-matches-a-special-scheme -bool protocol_component_matches_a_special_scheme(Component const& protocol_component) -{ - // 1. Let special scheme list be a list populated with all of the special schemes. - // 2. For each scheme of special scheme list: - for (StringView scheme : special_schemes()) { - // 1. Let test result be RegExpBuiltinExec(protocol component’s regular expression, scheme). - auto test_result = protocol_component.matches(scheme); - - // 2. If test result is not null, then return true. - if (test_result) - return true; - } - - // 3. Return false. - return false; -} - -// https://urlpattern.spec.whatwg.org/#generate-a-regular-expression-and-name-list -struct RegularExpressionAndNameList { - String regular_expression; - Vector name_list; -}; - -static RegularExpressionAndNameList generate_a_regular_expression_and_name_list(Vector const& part_list, Options const& options) -{ - // 1. Let result be "^". - StringBuilder result; - result.append('^'); - - // 2. Let name list be a new list. - Vector name_list; - - // 3. For each part of part list: - for (auto const& part : part_list) { - // 1. If part’s type is "fixed-text": - if (part.type == Part::Type::FixedText) { - // 1. If part’s modifier is "none", then append the result of running escape a regexp string given part’s - // value to the end of result. - if (part.modifier == Part::Modifier::None) { - result.append(escape_a_regexp_string(part.value)); - } - // 2. Otherwise: - else { - // 1. Append "(?:" to the end of result. - result.append("(?:"sv); - - // 2. Append the result of running escape a regexp string given part’s value to the end of result. - result.append(escape_a_regexp_string(part.value)); - - // 3. Append ")" to the end of result. - result.append(')'); - - // 4. Append the result of running convert a modifier to a string given part’s modifier to the end of result. - result.append(Part::convert_modifier_to_string(part.modifier)); - } - - // 3. Continue. - continue; - } - - // 2. Assert: part’s name is not the empty string. - VERIFY(!part.name.is_empty()); - - // 3. Append part’s name to name list. - name_list.append(part.name); - - // 4. Let regexp value be part’s value. - auto regexp_value = part.value; - - // 5. If part’s type is "segment-wildcard", then set regexp value to the result of running generate a segment wildcard regexp given options. - if (part.type == Part::Type::SegmentWildcard) { - regexp_value = generate_a_segment_wildcard_regexp(options); - } - // 6. Otherwise if part’s type is "full-wildcard", then set regexp value to full wildcard regexp value. - else if (part.type == Part::Type::FullWildcard) { - regexp_value = MUST(String::from_utf8(full_wildcard_regexp_value)); - } - - // 7. If part’s prefix is the empty string and part’s suffix is the empty string: - if (part.prefix.is_empty() && part.suffix.is_empty()) { - // 1. If part’s modifier is "none" or "optional", then: - if (part.modifier == Part::Modifier::None || part.modifier == Part::Modifier::Optional) { - // 1. Append "(" to the end of result. - result.append('('); - - // 2. Append regexp value to the end of result. - result.append(regexp_value); - - // 3. Append ")" to the end of result. - result.append(')'); - - // 4. Append the result of running convert a modifier to a string given part’s modifier to the end of result. - result.append(Part::convert_modifier_to_string(part.modifier)); - } - // 2. Otherwise: - else { - // 1. Append "((?:" to the end of result. - result.append("((?:"sv); - - // 2. Append regexp value to the end of result. - result.append(regexp_value); - - // 3. Append ")" to the end of result. - result.append(')'); - - // 4. Append the result of running convert a modifier to a string given part’s modifier to the end of result. - result.append(Part::convert_modifier_to_string(part.modifier)); - - // 5. Append ")" to the end of result. - result.append(')'); - } - - // 3. Continue. - continue; - } - - // 8. If part’s modifier is "none" or "optional": - if (part.modifier == Part::Modifier::None || part.modifier == Part::Modifier::Optional) { - // 1. Append "(?:" to the end of result. - result.append("(?:"sv); - - // 2. Append the result of running escape a regexp string given part’s prefix to the end of result. - result.append(escape_a_regexp_string(part.prefix)); - - // 3. Append "(" to the end of result. - result.append('('); - - // 4. Append regexp value to the end of result. - result.append(regexp_value); - - // 5. Append ")" to the end of result. - result.append(')'); - - // 6. Append the result of running escape a regexp string given part’s suffix to the end of result. - result.append(escape_a_regexp_string(part.suffix)); - - // 7. Append ")" to the end of result. - result.append(')'); - - // 8. Append the result of running convert a modifier to a string given part’s modifier to the end of result. - result.append(Part::convert_modifier_to_string(part.modifier)); - - // 9. Continue. - continue; - } - - // 9. Assert: part’s modifier is "zero-or-more" or "one-or-more". - VERIFY(part.modifier == Part::Modifier::ZeroOrMore || part.modifier == Part::Modifier::OneOrMore); - - // 10. Assert: part’s prefix is not the empty string or part’s suffix is not the empty string. - VERIFY(!part.prefix.is_empty() || !part.suffix.is_empty()); - - // 11. Append "(?:" to the end of result. - result.append("(?:"sv); - - // 12. Append the result of running escape a regexp string given part’s prefix to the end of result. - result.append(escape_a_regexp_string(part.prefix)); - - // 13. Append "((?:" to the end of result. - result.append("((?:"sv); - - // 14. Append regexp value to the end of result. - result.append(regexp_value); - - // 15. Append ")(?:" to the end of result. - result.append(")(?:"sv); - - // 16. Append the result of running escape a regexp string given part’s suffix to the end of result. - result.append(escape_a_regexp_string(part.suffix)); - - // 17. Append the result of running escape a regexp string given part’s prefix to the end of result. - result.append(escape_a_regexp_string(part.prefix)); - - // 18. Append "(?:" to the end of result. - result.append("(?:"sv); - - // 19. Append regexp value to the end of result. - result.append(regexp_value); - - // 20. Append "))*)" to the end of result. - result.append("))*)"sv); - - // 21. Append the result of running escape a regexp string given part’s suffix to the end of result. - result.append(escape_a_regexp_string(part.suffix)); - - // 22. Append ")" to the end of result. - result.append(')'); - - // 23. If part’s modifier is "zero-or-more" then append "?" to the end of result. - if (part.modifier == Part::Modifier::ZeroOrMore) - result.append('?'); - } - - // 4. Append "$" to the end of result. - result.append('$'); - - // 5. Return (result, name list). - return { result.to_string_without_validation(), move(name_list) }; -} - -// https://urlpattern.spec.whatwg.org/#compile-a-component -PatternErrorOr Component::compile(Utf8View const& input, PatternParser::EncodingCallback encoding_callback, Options const& options) -{ - // 1. Let part list be the result of running parse a pattern string given input, options, and encoding callback. - auto part_list = TRY(PatternParser::parse(input, options, move(encoding_callback))); - - // 2. Let (regular expression string, name list) be the result of running generate a regular expression and name - // list given part list and options. - auto [regular_expression_string, name_list] = generate_a_regular_expression_and_name_list(part_list, options); - - // 3. Let flags be an empty string. - // NOTE: These flags match the flags for the empty string of the LibJS RegExp implementation. - regex::ECMAScriptCompileFlags flags {}; - - // 4. If options’s ignore case is true then set flags to "vi". - if (options.ignore_case) { - flags.unicode_sets = true; - flags.ignore_case = true; - } - // 5. Otherwise set flags to "v" - else { - flags.unicode_sets = true; - } - - // 6. Let regular expression be RegExpCreate(regular expression string, flags). If this throws an exception, catch - // it, and throw a TypeError. - auto regex = regex::ECMAScriptRegex::compile(regular_expression_string.bytes_as_string_view(), flags); - if (regex.is_error()) - return ErrorInfo { MUST(String::formatted("RegExp compile error: {}", regex.release_error())) }; - - // 7. Let pattern string be the result of running generate a pattern string given part list and options. - auto pattern_string = generate_a_pattern_string(part_list, options); - - // 8. Let has regexp groups be false. - bool has_regexp_groups = false; - - // 9. For each part of part list: - for (auto const& part : part_list) { - // 1. If part’s type is "regexp", then set has regexp groups to true. - if (part.type == Part::Type::Regexp) { - has_regexp_groups = true; - break; - } - } - - // 10. Return a new component whose pattern string is pattern string, regular expression is regular expression, - // group name list is name list, and has regexp groups is has regexp groups. - return Component { - .pattern_string = move(pattern_string), - .regular_expression = adopt_own(*new regex::ECMAScriptRegex(regex.release_value())), - .group_name_list = move(name_list), - .has_regexp_groups = has_regexp_groups, - }; -} - -Component::ExecutionResult Component::execute(String const& input) const -{ - auto utf16_input = Utf16String::from_utf8(input); - auto match_result = regular_expression->exec(utf16_input.utf16_view(), 0); - if (match_result != regex::MatchResult::Match) - return {}; - - ExecutionResult result; - result.success = true; - result.captures.ensure_capacity(group_name_list.size()); - for (size_t index = 1; index <= group_name_list.size(); ++index) { - auto start = regular_expression->capture_slot(index * 2); - auto end = regular_expression->capture_slot(index * 2 + 1); - if (start < 0 || end < 0) { - result.captures.append({}); - continue; - } - - auto capture = utf16_input.substring_view(static_cast(start), static_cast(end - start)); - result.captures.append(MUST(capture.to_utf8())); - } - - return result; -} - -bool Component::matches(StringView input) const -{ - auto utf16_input = Utf16String::from_utf8(input); - return regular_expression->test(utf16_input.utf16_view(), 0) == regex::MatchResult::Match; -} - -// https://urlpattern.spec.whatwg.org/#create-a-component-match-result -Component::Result Component::create_match_result(String const& input, ExecutionResult const& exec_result) const -{ - // 1. Let result be a new URLPatternComponentResult. - Component::Result result; - - // 2. Set result["input"] to input. - result.input = input; - - // 3. Let groups be a record. - OrderedHashMap> groups; - - // 4. Let index be 1. - // 5. While index is less than or equal to component’s group name list’s size: - VERIFY(exec_result.captures.size() == group_name_list.size()); - for (size_t index = 1; index <= group_name_list.size(); ++index) { - // 1. Let name be component’s group name list[index − 1]. - auto name = group_name_list[index - 1]; - - // 2. Let value be Get(execResult, ToString(index)). - // 3. Set groups[name] to value. - auto const& capture = exec_result.captures[index - 1]; - if (!capture.has_value()) - groups.set(name, Empty {}); - else - groups.set(name, *capture); - - // 4. Increment index by 1. - } - - // 6. Set result["groups"] to groups. - result.groups = move(groups); - - // 7. Return result. - return result; -} - -} diff --git a/Libraries/LibURL/Pattern/Component.h b/Libraries/LibURL/Pattern/Component.h deleted file mode 100644 index be1f9dffc3..0000000000 --- a/Libraries/LibURL/Pattern/Component.h +++ /dev/null @@ -1,56 +0,0 @@ -/* - * Copyright (c) 2025, Shannon Booth - * - * SPDX-License-Identifier: BSD-2-Clause - */ - -#pragma once - -#include -#include -#include -#include -#include -#include - -namespace URL::Pattern { - -// https://urlpattern.spec.whatwg.org/#component -struct Component { - static PatternErrorOr compile(Utf8View const& input, PatternParser::EncodingCallback, Options const&); - - // https://urlpattern.spec.whatwg.org/#dictdef-urlpatterncomponentresult - struct Result { - String input; - OrderedHashMap> groups; - }; - - struct ExecutionResult { - bool success { false }; - Vector> captures; - }; - - Result create_match_result(String const& input, ExecutionResult const& exec_result) const; - ExecutionResult execute(String const& input) const; - bool matches(StringView input) const; - - // https://urlpattern.spec.whatwg.org/#component-pattern-string - // pattern string, a well formed pattern string - String pattern_string; - - // https://urlpattern.spec.whatwg.org/#component-regular-expression - // regular expression, a RegExp - OwnPtr regular_expression; - - // https://urlpattern.spec.whatwg.org/#component-group-name-list - // group name list, a list of strings - Vector group_name_list; - - // https://urlpattern.spec.whatwg.org/#component-has-regexp-groups - // has regexp groups, a boolean - bool has_regexp_groups {}; -}; - -bool protocol_component_matches_a_special_scheme(Component const& protocol_component); - -} diff --git a/Libraries/LibURL/Pattern/ConstructorStringParser.cpp b/Libraries/LibURL/Pattern/ConstructorStringParser.cpp deleted file mode 100644 index 32bf7afa97..0000000000 --- a/Libraries/LibURL/Pattern/ConstructorStringParser.cpp +++ /dev/null @@ -1,713 +0,0 @@ -/* - * Copyright (c) 2025, Shannon Booth - * - * SPDX-License-Identifier: BSD-2-Clause - */ - -#include -#include -#include -#include -#include - -namespace URL::Pattern { - -StringView ConstructorStringParser::state_to_string() const -{ - switch (m_state) { - case State::Initial: - return "Initial"sv; - case State::Protocol: - return "Protocol"sv; - case State::Authority: - return "Authority"sv; - case State::Username: - return "Username"sv; - case State::Password: - return "Password"sv; - case State::Hostname: - return "Hostname"sv; - case State::Port: - return "Port"sv; - case State::Pathname: - return "Pathname"sv; - case State::Search: - return "Search"sv; - case State::Hash: - return "Hash"sv; - case State::Done: - return "Done"sv; - } - VERIFY_NOT_REACHED(); -} - -ConstructorStringParser::ConstructorStringParser(Utf8View const& input, Vector token_list) - : m_input(input) - , m_token_list(move(token_list)) -{ -} - -// https://urlpattern.spec.whatwg.org/#parse-a-constructor-string -PatternErrorOr ConstructorStringParser::parse(Utf8View const& input) -{ - // 1. Let parser be a new constructor string parser whose input is input and token list is the result of running - // tokenize given input and "lenient". - ConstructorStringParser parser { input, TRY(Tokenizer::tokenize(input, Tokenizer::Policy::Lenient)) }; - - // 2. While parser’s token index is less than parser’s token list size: - while (parser.m_token_index < parser.m_token_list.size()) { - dbgln_if(URL_PATTERN_DEBUG, "{}\t| Token@{} (group depth {}) -> {}", parser.state_to_string(), - parser.m_token_index, parser.m_group_depth, parser.m_token_list[parser.m_token_index].to_string()); - // 1. Set parser’s token increment to 1. - parser.m_token_increment = 1; - - // NOTE: On every iteration of the parse loop the parser’s token index will be incremented by its token - // increment value. Typically this means incrementing by 1, but at certain times it is set to zero. - // The token increment is then always reset back to 1 at the top of the loop. - - // 2. If parser’s token list[parser’s token index]'s type is "end" then: - if (parser.m_token_list[parser.m_token_index].type == Token::Type::End) { - // 1. If parser’s state is "init": - if (parser.m_state == State::Initial) { - // NOTE: If we reached the end of the string in the "init" state, then we failed to find a protocol - // terminator and this has to be a relative URLPattern constructor string. - - // 1. Run rewind given parser. - parser.rewind(); - - // NOTE: We next determine at which component the relative pattern begins. Relative pathnames are - // most common, but URLs and URLPattern constructor strings can begin with the search or hash - // components as well. - - // 2. If the result of running is a hash prefix given parser is true, then run change state given parser, - // "hash" and 1. - if (parser.is_a_hash_prefix()) { - parser.change_state(State::Hash, 1); - } - // 3. Otherwise if the result of running is a search prefix given parser is true: - else if (parser.is_a_search_prefix()) { - // 1. Run change state given parser, "search" and 1. - parser.change_state(State::Search, 1); - } - // 4. Otherwise: - else { - // 1. Run change state given parser, "pathname" and 0. - parser.change_state(State::Pathname, 0); - } - - // 5. Increment parser’s token index by parser’s token increment. - parser.m_token_index += parser.m_token_increment; - - // 6. Continue. - continue; - } - - // 2. If parser’s state is "authority": - if (parser.m_state == State::Authority) { - // NOTE: If we reached the end of the string in the "authority" state, then we failed to find an - // "@". Therefore there is no username or password. - - // 1. Run rewind and set state given parser, and "hostname". - parser.rewind_and_set_state(State::Hostname); - - // 2. Increment parser’s token index by parser’s token increment. - parser.m_token_index += parser.m_token_increment; - - // 3. Continue. - continue; - } - - // 3. Run change state given parser, "done" and 0. - parser.change_state(State::Done, 0); - - // 4. Break. - break; - } - - // 3. If the result of running is a group open given parser is true: - if (parser.is_a_group_open()) { - // NOTE: We ignore all code points within "{ ... }" pattern groupings. It would not make sense to allow - // a URL component boundary to lie within a grouping; e.g. "https://example.c{om/fo}o". While not - // supported within well formed pattern strings, we handle nested groupings here to avoid parser - // confusion. - // - // It is not necessary to perform this logic for regexp or named groups since those values are collapsed into - // individual tokens by the tokenize algorithm. - - // 1. Increment parser’s group depth by 1. - ++parser.m_group_depth; - - // 2. Increment parser’s token index by parser’s token increment. - parser.m_token_index += parser.m_token_increment; - - // 3. Continue. - continue; - } - - // 4. If parser’s group depth is greater than 0: - if (parser.m_group_depth > 0) { - // 1. If the result of running is a group close given parser is true, then decrement parser’s group depth by 1. - if (parser.is_a_group_close()) { - VERIFY(parser.m_group_depth != 0); - --parser.m_group_depth; - } - // 2. Otherwise: - else { - // 1. Increment parser’s token index by parser’s token increment. - parser.m_token_index += parser.m_token_increment; - - // 2. Continue. - continue; - } - } - - // 5. Switch on parser’s state and run the associated steps: - switch (parser.m_state) { - // -> "init", https://urlpattern.spec.whatwg.org/#ref-for-constructor-string-parser-state-init%E2%91%A2 - case State::Initial: { - // 1. If the result of running is a protocol suffix given parser is true: - if (parser.is_a_protocol_suffix()) { - // 1. Run rewind and set state given parser and "protocol". - parser.rewind_and_set_state(State::Protocol); - } - break; - } - // -> "protocol", https://urlpattern.spec.whatwg.org/#ref-for-constructor-string-parser-state-protocol%E2%91%A0 - case State::Protocol: { - // 1. If the result of running is a protocol suffix given parser is true: - if (parser.is_a_protocol_suffix()) { - // 1. Run compute protocol matches a special scheme flag given parser. - TRY(parser.compute_protocol_matches_a_special_scheme_flag()); - - // NOTE: We need to eagerly compile the protocol component to determine if it matches any special - // schemes. If it does then certain special rules apply. It determines if the pathname - // defaults to a "/" and also whether we will look for the username, password, hostname, and - // port components. Authority slashes can also cause us to look for these components as well. - // Otherwise we treat this as an "opaque path URL" and go straight to the pathname component. - - // 2. Let next state be "pathname". - auto next_state = State::Pathname; - - // 3. Let skip be 1. - u32 skip = 1; - - // 4. If the result of running next is authority slashes given parser is true: - if (parser.next_is_authority_slashes()) { - // 1. Set next state to "authority". - next_state = State::Authority; - - // 2. Set skip to 3. - skip = 3; - } - // 5. Otherwise if parser’s protocol matches a special scheme flag is true, then set next state to "authority". - else if (parser.m_protocol_matches_a_special_scheme) { - next_state = State::Authority; - } - - // 6. Run change state given parser, next state, and skip. - parser.change_state(next_state, skip); - } - break; - } - // -> "authority", https://urlpattern.spec.whatwg.org/#ref-for-constructor-string-parser-state-authority%E2%91%A3 - case State::Authority: { - // 1. If the result of running is an identity terminator given parser is true, then run rewind and set state - // given parser and "username". - if (parser.is_an_identity_terminator()) { - parser.rewind_and_set_state(State::Username); - } - // 2. Otherwise if any of the following are true: - // * the result of running is a pathname start given parser; - // * the result of running is a search prefix given parser; or - // * the result of running is a hash prefix given parser, - // then run rewind and set state given parser and "hostname". - else if (parser.is_a_pathname_start() - || parser.is_a_search_prefix() - || parser.is_a_hash_prefix()) { - parser.rewind_and_set_state(State::Hostname); - } - break; - } - // -> "username", https://urlpattern.spec.whatwg.org/#ref-for-constructor-string-parser-state-username%E2%91%A0 - case State::Username: { - // 1. If the result of running is a password prefix given parser is true, then run change state given - // parser, "password", and 1. - if (parser.is_a_password_prefix()) { - parser.change_state(State::Password, 1); - } - // 2. Otherwise if the result of running is an identity terminator given parser is true, then run change - // state given parser, "hostname", and 1. - else if (parser.is_an_identity_terminator()) { - parser.change_state(State::Hostname, 1); - } - break; - } - // -> "password", https://urlpattern.spec.whatwg.org/#ref-for-constructor-string-parser-state-password%E2%91%A0 - case State::Password: { - // 1. If the result of running is an identity terminator given parser is true, then run change state - // given parser, "hostname", and 1. - if (parser.is_an_identity_terminator()) - parser.change_state(State::Hostname, 1); - break; - } - // -> "hostname", https://urlpattern.spec.whatwg.org/#ref-for-constructor-string-parser-state-hostname%E2%91%A3 - case State::Hostname: { - // 1. If the result of running is an IPv6 open given parser is true, then increment parser’s hostname - // IPv6 bracket depth by 1. - if (parser.is_an_ipv6_open()) { - ++parser.m_hostname_ipv6_bracket_depth; - } - // 2. Otherwise if the result of running is an IPv6 close given parser is true, then decrement parser’s - // hostname IPv6 bracket depth by 1. - else if (parser.is_an_ipv6_close()) { - VERIFY(parser.m_hostname_ipv6_bracket_depth != 0); - --parser.m_hostname_ipv6_bracket_depth; - } - // 3. Otherwise if the result of running is a port prefix given parser is true and parser’s hostname IPv6 - // bracket depth is zero, then run change state given parser, "port", and 1. - else if (parser.is_a_port_prefix() && parser.m_hostname_ipv6_bracket_depth == 0) { - parser.change_state(State::Port, 1); - } - // 4. Otherwise if the result of running is a pathname start given parser is true, then run change state - // given parser, "pathname", and 0. - else if (parser.is_a_pathname_start()) { - parser.change_state(State::Pathname, 0); - } - // 5. Otherwise if the result of running is a search prefix given parser is true, then run change state - // given parser, "search", and 1. - else if (parser.is_a_search_prefix()) { - parser.change_state(State::Search, 1); - } - // 6. Otherwise if the result of running is a hash prefix given parser is true, then run change state - // given parser, "hash", and 1. - else if (parser.is_a_hash_prefix()) { - parser.change_state(State::Hash, 1); - } - break; - } - // -> "port", https://urlpattern.spec.whatwg.org/#ref-for-constructor-string-parser-state-port%E2%91%A0 - case State::Port: { - // 1. If the result of running is a pathname start given parser is true, then run change state given - // parser, "pathname", and 0. - if (parser.is_a_pathname_start()) { - parser.change_state(State::Pathname, 0); - } - // 2. Otherwise if the result of running is a search prefix given parser is true, then run change state - // given parser, "search", and 1. - else if (parser.is_a_search_prefix()) { - parser.change_state(State::Search, 1); - } - // 3. Otherwise if the result of running is a hash prefix given parser is true, then run change state given - // parser, "hash", and 1. - else if (parser.is_a_hash_prefix()) { - parser.change_state(State::Hash, 1); - } - break; - } - // -> "pathname", https://urlpattern.spec.whatwg.org/#ref-for-constructor-string-parser-state-pathname%E2%91%A3 - case State::Pathname: { - // 1. If the result of running is a search prefix given parser is true, then run change state given parser, - // "search", and 1. - if (parser.is_a_search_prefix()) { - parser.change_state(State::Search, 1); - } - // 2. Otherwise if the result of running is a hash prefix given parser is true, then run change state given - // parser, "hash", and 1. - else if (parser.is_a_hash_prefix()) { - parser.change_state(State::Hash, 1); - } - break; - } - // -> "search", https://urlpattern.spec.whatwg.org/#ref-for-constructor-string-parser-state-search%E2%91%A3 - case State::Search: { - // 1. If the result of running is a hash prefix given parser is true, then run change state given parser, - // "hash", and 1. - if (parser.is_a_hash_prefix()) - parser.change_state(State::Hash, 1); - break; - } - // -> "hash", https://urlpattern.spec.whatwg.org/#ref-for-constructor-string-parser-state-hash%E2%91%A4 - case State::Hash: { - // 1. Do nothing. - break; - } - // -> "done", https://urlpattern.spec.whatwg.org/#ref-for-constructor-string-parser-state-done%E2%91%A0 - case State::Done: { - // 1. Assert: This step is never reached. - VERIFY_NOT_REACHED(); - } - } - - // 6. Increment parser’s token index by parser’s token increment. - parser.m_token_index += parser.m_token_increment; - } - - // 3. If parser’s result contains "hostname" and not "port", then set parser’s result["port"] to the empty string. - if (parser.m_result.hostname.has_value() && !parser.m_result.port.has_value()) - parser.m_result.port = String {}; - - // NOTE: This is special-cased because when an author does not specify a port, they usually intend the default - // port. If any port is acceptable, the author can specify it as a wildcard explicitly. For example, - // "https://example.com/*" does not match URLs beginning with "https://example.com:8443/", which is a - // different origin. - - // 4. Return parser’s result. - return parser.m_result; -} - -// https://urlpattern.spec.whatwg.org/#make-a-component-string -String ConstructorStringParser::make_a_component_string() const -{ - // 1. Assert: parser’s token index is less than parser’s token list's size. - VERIFY(m_token_index < m_token_list.size()); - - // 2. Let token be parser’s token list[parser’s token index]. - auto const& token = m_token_list[m_token_index]; - - // 3. Let component start token be the result of running get a safe token given parser and parser’s component start. - auto const& component_start_token = get_a_safe_token(m_component_start); - - // 4. Let component start input index be component start token’s index. - auto component_start_input_index = component_start_token.index; - - // 5. Let end index be token’s index. - auto end_index = token.index; - - // 6. Return the code point substring from component start input index to end index within parser’s input. - auto sub_view = m_input.unicode_substring_view(component_start_input_index, end_index - component_start_input_index); - return MUST(String::from_utf8(sub_view.as_string())); -} - -// https://urlpattern.spec.whatwg.org/#compute-protocol-matches-a-special-scheme-flag -PatternErrorOr ConstructorStringParser::compute_protocol_matches_a_special_scheme_flag() -{ - // 1. Let protocol string be the result of running make a component string given parser. - auto protocol_string = make_a_component_string(); - - // 2. Let protocol component be the result of compiling a component given protocol string, canonicalize a protocol, and default options. - auto protocol_component = TRY(Component::compile(protocol_string.code_points(), canonicalize_a_protocol, Options::default_())); - - // 3. If the result of running protocol component matches a special scheme given protocol component is true, then set parser’s protocol matches a special scheme flag to true. - if (protocol_component_matches_a_special_scheme(protocol_component)) - m_protocol_matches_a_special_scheme = true; - - return {}; -} - -Optional const& ConstructorStringParser::result_for_active_state() const -{ - switch (m_state) { - case State::Protocol: - return m_result.protocol; - case State::Username: - return m_result.username; - case State::Password: - return m_result.password; - case State::Hostname: - return m_result.hostname; - case State::Port: - return m_result.port; - case State::Pathname: - return m_result.pathname; - case State::Search: - return m_result.search; - case State::Hash: - return m_result.hash; - case State::Initial: - case State::Authority: - case State::Done: - break; - } - VERIFY_NOT_REACHED(); -} - -void ConstructorStringParser::set_result_for_active_state(Optional value) -{ - switch (m_state) { - case State::Protocol: - m_result.protocol = move(value); - break; - case State::Username: - m_result.username = move(value); - break; - case State::Password: - m_result.password = move(value); - break; - case State::Hostname: - m_result.hostname = move(value); - break; - case State::Port: - m_result.port = move(value); - break; - case State::Pathname: - m_result.pathname = move(value); - break; - case State::Search: - m_result.search = move(value); - break; - case State::Hash: - m_result.hash = move(value); - break; - case State::Initial: - case State::Authority: - case State::Done: - VERIFY_NOT_REACHED(); - } -} - -// https://urlpattern.spec.whatwg.org/#change-state -void ConstructorStringParser::change_state(State new_state, u32 skip) -{ - // 1. If parser’s state is not "init", not "authority", and not "done", then set parser’s result[parser’s state] to - // the result of running make a component string given parser. - if (m_state != State::Initial && m_state != State::Authority && m_state != State::Done) - set_result_for_active_state(make_a_component_string()); - - // 2. If parser’s state is not "init" and new state is not "done", then: - if (m_state != State::Initial && new_state != State::Done) { - // 1. If parser’s state is "protocol", "authority", "username", or "password"; new state is "port", "pathname", - // "search", or "hash"; and parser’s result["hostname"] does not exist, then set parser’s result["hostname"] - // to the empty string. - if (first_is_one_of(m_state, State::Protocol, State::Authority, State::Username, State::Password) - && first_is_one_of(new_state, State::Port, State::Pathname, State::Search, State::Hash) - && !m_result.hostname.has_value()) { - m_result.hostname = String {}; - } - - // 2. If parser’s state is "protocol", "authority", "username", "password", "hostname", or "port"; new state is - // "search" or "hash"; and parser’s result["pathname"] does not exist, then: - if (first_is_one_of(m_state, State::Protocol, State::Authority, State::Username, State::Password, State::Hostname, State::Port) - && first_is_one_of(new_state, State::Search, State::Hash) - && !m_result.pathname.has_value()) { - // 1. If parser’s protocol matches a special scheme flag is true, then set parser’s result["pathname"] to "/". - if (m_protocol_matches_a_special_scheme) { - m_result.pathname = "/"_string; - } - // 2. Otherwise, set parser’s result["pathname"] to the empty string. - else { - m_result.pathname = String {}; - } - } - - // 3. If parser’s state is "protocol", "authority", "username", "password", "hostname", "port", or "pathname"; - // new state is "hash"; and parser’s result["search"] does not exist, then set parser’s result["search"] - // to the empty string. - if (first_is_one_of(m_state, State::Protocol, State::Authority, State::Username, State::Password, State::Hostname, State::Port, State::Pathname) - && new_state == State::Hash - && !m_result.search.has_value()) { - m_result.search = String {}; - } - } - - // 3. Set parser’s state to new state. - m_state = new_state; - - // 4. Increment parser’s token index by skip. - m_token_index += skip; - - // 5. Set parser’s component start to parser’s token index. - m_component_start = m_token_index; - - // 6. Set parser’s token increment to 0. - m_token_increment = 0; -} - -// https://urlpattern.spec.whatwg.org/#next-is-authority-slashes -bool ConstructorStringParser::next_is_authority_slashes() const -{ - // 1. If the result of running is a non-special pattern char given parser, parser’s token index + 1, and "/" is false, - // then return false. - if (!is_a_non_special_pattern_char(m_token_index + 1, '/')) - return false; - - // 2. If the result of running is a non-special pattern char given parser, parser’s token index + 2, and "/" is false, - // then return false. - if (!is_a_non_special_pattern_char(m_token_index + 2, '/')) - return false; - - // 3. Return true. - return true; -} - -// https://urlpattern.spec.whatwg.org/#is-an-identity-terminator -bool ConstructorStringParser::is_an_identity_terminator() const -{ - // 1. Return the result of running is a non-special pattern char given parser, parser’s token index, and "@". - return is_a_non_special_pattern_char(m_token_index, '@'); -} - -// https://urlpattern.spec.whatwg.org/#is-a-password-prefix -bool ConstructorStringParser::is_a_password_prefix() const -{ - // 1. Return the result of running is a non-special pattern char given parser, parser’s token index, and ":". - return is_a_non_special_pattern_char(m_token_index, ':'); -} - -// https://urlpattern.spec.whatwg.org/#is-a-port-prefix -bool ConstructorStringParser::is_a_port_prefix() const -{ - // 1. Return the result of running is a non-special pattern char given parser, parser’s token index, and ":". - return is_a_non_special_pattern_char(m_token_index, ':'); -} - -// https://urlpattern.spec.whatwg.org/#is-a-pathname-start -bool ConstructorStringParser::is_a_pathname_start() const -{ - // 1. Return the result of running is a non-special pattern char given parser, parser’s token index, and "/". - return is_a_non_special_pattern_char(m_token_index, '/'); -} - -// https://urlpattern.spec.whatwg.org/#is-a-search-prefix -bool ConstructorStringParser::is_a_search_prefix() const -{ - // 1. If result of running is a non-special pattern char given parser, parser’s token index and "?" is true, - // then return true. - if (is_a_non_special_pattern_char(m_token_index, '?')) - return true; - - // 2. If parser’s token list[parser’s token index]'s value is not "?", then return false. - if (m_token_list[m_token_index].value != "?"sv) - return false; - - // 3. Let previous index be parser’s token index − 1. - // 4. If previous index is less than 0, then return true. - if (m_token_index == 0) - return true; - auto previous_index = m_token_index - 1; - - // 5. Let previous token be the result of running get a safe token given parser and previous index. - auto const& previous_token = get_a_safe_token(previous_index); - - // 6. If any of the following are true, then return false: - // * previous token’s type is "name". - // * previous token’s type is "regexp". - // * previous token’s type is "close". - // * previous token’s type is "asterisk". - if (previous_token.type == Token::Type::Name - || previous_token.type == Token::Type::Regexp - || previous_token.type == Token::Type::Close - || previous_token.type == Token::Type::Asterisk) { - return false; - } - - // 7. Return true. - return true; -} - -// https://urlpattern.spec.whatwg.org/#is-a-protocol-suffix -bool ConstructorStringParser::is_a_protocol_suffix() const -{ - // 1. Return the result of running is a non-special pattern char given parser, parser’s token index, and ":". - return is_a_non_special_pattern_char(m_token_index, ':'); -} - -// https://urlpattern.spec.whatwg.org/#is-a-hash-prefix -bool ConstructorStringParser::is_a_hash_prefix() const -{ - // 1. Return the result of running is a non-special pattern char given parser, parser’s token index and "#". - return is_a_non_special_pattern_char(m_token_index, '#'); -} - -// https://urlpattern.spec.whatwg.org/#is-a-group-open -bool ConstructorStringParser::is_a_group_open() const -{ - // 1. If parser’s token list[parser’s token index]'s type is "open", then return true. - if (m_token_list[m_token_index].type == Token::Type::Open) - return true; - - // 2. Otherwise return false. - return false; -} - -// https://urlpattern.spec.whatwg.org/#is-a-group-close -bool ConstructorStringParser::is_a_group_close() const -{ - // 1. If parser’s token list[parser’s token index]'s type is "close", then return true. - if (m_token_list[m_token_index].type == Token::Type::Close) - return true; - - // 2. Otherwise return false. - return false; -} - -// https://urlpattern.spec.whatwg.org/#is-an-ipv6-open -bool ConstructorStringParser::is_an_ipv6_open() const -{ - // 1. Return the result of running is a non-special pattern char given parser, parser’s token index, and "[". - return is_a_non_special_pattern_char(m_token_index, '['); -} - -// https://urlpattern.spec.whatwg.org/#is-an-ipv6-close -bool ConstructorStringParser::is_an_ipv6_close() const -{ - // 1. Return the result of running is a non-special pattern char given parser, parser’s token index, and "]". - return is_a_non_special_pattern_char(m_token_index, ']'); -} - -// https://urlpattern.spec.whatwg.org/#get-a-safe-token -Token const& ConstructorStringParser::get_a_safe_token(u32 index) const -{ - // 1. If index is less than parser’s token list's size, then return parser’s token list[index]. - if (index < m_token_list.size()) - return m_token_list[index]; - - // 2. Assert: parser’s token list's size is greater than or equal to 1. - VERIFY(!m_token_list.is_empty()); - - // 3. Let last index be parser’s token list's size − 1. - // 4. Let token be parser’s token list[last index]. - auto const& token = m_token_list.last(); - - // 5. Assert: token’s type is "end". - VERIFY(token.type == Token::Type::End); - - // 6. Return token. - return token; -} - -// https://urlpattern.spec.whatwg.org/#is-a-non-special-pattern-char -bool ConstructorStringParser::is_a_non_special_pattern_char(u32 index, char value) const -{ - // 1. Let token be the result of running get a safe token given parser and index. - auto const& token = get_a_safe_token(index); - - // 2. If token’s value is not value, then return false. - if (token.value.is_empty() || token.value.bytes().first() != value) - return false; - - // 3. If any of the following are true: - // * token’s type is "char"; - // * token’s type is "escaped-char"; or - // * token’s type is "invalid-char", - // then return true. - if (token.type == Token::Type::Char - || token.type == Token::Type::EscapedChar - || token.type == Token::Type::InvalidChar) { - return true; - } - - // 4. Return false. - return false; -} - -// https://urlpattern.spec.whatwg.org/#rewind -void ConstructorStringParser::rewind() -{ - // 1. Set parser’s token index to parser’s component start. - m_token_index = m_component_start; - - // 2. Set parser’s token increment to 0. - m_token_increment = 0; -} - -// https://urlpattern.spec.whatwg.org/#rewind-and-set-state -void ConstructorStringParser::rewind_and_set_state(State state) -{ - // 1. Run rewind given parser. - rewind(); - - // 2. Set parser’s state to state. - m_state = state; -} - -} diff --git a/Libraries/LibURL/Pattern/ConstructorStringParser.h b/Libraries/LibURL/Pattern/ConstructorStringParser.h deleted file mode 100644 index baac22dd12..0000000000 --- a/Libraries/LibURL/Pattern/ConstructorStringParser.h +++ /dev/null @@ -1,106 +0,0 @@ -/* - * Copyright (c) 2025, Shannon Booth - * - * SPDX-License-Identifier: BSD-2-Clause - */ - -#pragma once - -#include -#include -#include - -namespace URL::Pattern { - -// https://urlpattern.spec.whatwg.org/#constructor-string-parser -class ConstructorStringParser { -public: - static PatternErrorOr parse(Utf8View const& input); - -private: - ConstructorStringParser(Utf8View const& input, Vector token_list); - - // https://urlpattern.spec.whatwg.org/#constructor-string-parser-state - enum class State { - Initial, - Protocol, - Authority, - Username, - Password, - Hostname, - Port, - Pathname, - Search, - Hash, - Done, - }; - StringView state_to_string() const; - - void rewind(); - void rewind_and_set_state(State); - - bool next_is_authority_slashes() const; - - bool is_an_identity_terminator() const; - bool is_a_port_prefix() const; - bool is_a_pathname_start() const; - bool is_a_password_prefix() const; - bool is_a_search_prefix() const; - bool is_a_hash_prefix() const; - bool is_a_protocol_suffix() const; - bool is_an_ipv6_open() const; - bool is_an_ipv6_close() const; - bool is_a_group_open() const; - bool is_a_group_close() const; - - Token const& get_a_safe_token(u32 index) const; - bool is_a_non_special_pattern_char(u32 index, char value) const; - void change_state(State, u32 skip); - String make_a_component_string() const; - PatternErrorOr compute_protocol_matches_a_special_scheme_flag(); - - Optional const& result_for_active_state() const; - void set_result_for_active_state(Optional value); - - // https://urlpattern.spec.whatwg.org/#constructor-string-parser-input - // A constructor string parser has an associated input, a string, which must be set upon creation. - Utf8View m_input; - - // https://urlpattern.spec.whatwg.org/#constructor-string-parser-token-list - // A constructor string parser has an associated token list, a token list, which must be set upon creation. - Vector m_token_list; - - // https://urlpattern.spec.whatwg.org/#constructor-string-parser-result - // A constructor string parser has an associated result, a URLPatternInit, initially set to a new URLPatternInit. - Init m_result; - - // https://urlpattern.spec.whatwg.org/#constructor-string-parser-component-start - // A constructor string parser has an associated component start, a number, initially set to 0. - u32 m_component_start { 0 }; - - // https://urlpattern.spec.whatwg.org/#constructor-string-parser-token-index - // A constructor string parser has an associated token index, a number, initially set to 0. - u32 m_token_index { 0 }; - - // https://urlpattern.spec.whatwg.org/#constructor-string-parser-token-increment - // A constructor string parser has an associated token increment, a number, initially set to 1. - u32 m_token_increment { 1 }; - - // https://urlpattern.spec.whatwg.org/#constructor-string-parser-group-depth - // A constructor string parser has an associated group depth, a number, initially set to 0. - u32 m_group_depth { 0 }; - - // https://urlpattern.spec.whatwg.org/#constructor-string-parser-hostname-ipv6-bracket-depth - // A constructor string parser has an associated hostname IPv6 bracket depth, a number, initially set to 0. - u32 m_hostname_ipv6_bracket_depth { 0 }; - - // https://urlpattern.spec.whatwg.org/#constructor-string-parser-protocol-matches-a-special-scheme-flag - // A constructor string parser has an associated protocol matches a special scheme flag, a boolean, initially set to false. - bool m_protocol_matches_a_special_scheme { false }; - - // https://urlpattern.spec.whatwg.org/#constructor-string-parser-state - // A constructor string parser has an associated state, a string, initially set to "init". - State m_state { State::Initial }; -}; - -} diff --git a/Libraries/LibURL/Pattern/Init.cpp b/Libraries/LibURL/Pattern/Init.cpp deleted file mode 100644 index 40c97202c4..0000000000 --- a/Libraries/LibURL/Pattern/Init.cpp +++ /dev/null @@ -1,367 +0,0 @@ -/* - * Copyright (c) 2025, Shannon Booth - * - * SPDX-License-Identifier: BSD-2-Clause - */ - -#include -#include -#include -#include - -namespace URL::Pattern { - -// https://urlpattern.spec.whatwg.org/#process-a-base-url-string -static String process_a_base_url_string(String const& input, PatternProcessType type) -{ - // 1. Assert: input is not null. - // 2. If type is not "pattern" return input. - if (type != PatternProcessType::Pattern) - return input; - - // 3. Return the result of escaping a pattern string given input. - return escape_a_pattern_string(input); -} - -// https://urlpattern.spec.whatwg.org/#is-an-absolute-pathname -static bool is_an_absolute_pathname(String const& input, PatternProcessType type) -{ - // 1. If input is the empty string, then return false. - if (input.is_empty()) - return false; - - // 2. If input[0] is U+002F (/), then return true. - if (input.bytes()[0] == '/') - return true; - - // 3. If type is "url", then return false. - if (type == PatternProcessType::URL) - return false; - - // 4. If input’s code point length is less than 2, then return false. - if (input.bytes().size() < 2) - return false; - - // 5. If input[0] is U+005C (\) and input[1] is U+002F (/), then return true. - if (input.bytes()[0] == '\\' && input.bytes()[1] == '/') - return true; - - // 6. If input[0] is U+007B ({) and input[1] is U+002F (/), then return true. - if (input.bytes()[0] == '{' && input.bytes()[1] == '/') - return true; - - // 7. Return false. - return false; -} - -// https://urlpattern.spec.whatwg.org/#process-protocol-for-init -static PatternErrorOr process_protocol_for_init(String const& value, PatternProcessType type) -{ - // 1. Let strippedValue be the given value with a single trailing U+003A (:) removed, if any. - auto stripped_value = value; - if (stripped_value.ends_with(':')) - stripped_value = MUST(stripped_value.substring_from_byte_offset(0, stripped_value.bytes().size() - 1)); - - // 2. If type is "pattern" then return strippedValue. - if (type == PatternProcessType::Pattern) - return stripped_value; - - // 3. Return the result of running canonicalize a protocol given strippedValue. - return canonicalize_a_protocol(stripped_value); -} - -// https://urlpattern.spec.whatwg.org/#process-username-for-init -static String process_username_for_init(String const& value, PatternProcessType type) -{ - // 1. If type is "pattern" then return value. - if (type == PatternProcessType::Pattern) - return value; - - // 2. Return the result of running canonicalize a username given value. - return canonicalize_a_username(value); -} - -// https://urlpattern.spec.whatwg.org/#process-password-for-init -static String process_password_for_init(String const& value, PatternProcessType type) -{ - // 1. If type is "pattern" then return value. - if (type == PatternProcessType::Pattern) - return value; - - // 2. Return the result of running canonicalize a password given value. - return canonicalize_a_password(value); -} - -// https://urlpattern.spec.whatwg.org/#process-hostname-for-init -static PatternErrorOr process_hostname_for_init(String const& value, PatternProcessType type) -{ - // 1. If type is "pattern" then return value. - if (type == PatternProcessType::Pattern) - return value; - - // 2. Return the result of running canonicalize a hostname given value. - return canonicalize_a_hostname(value); -} - -// https://urlpattern.spec.whatwg.org/#process-port-for-init -static PatternErrorOr process_port_for_init(String const& port_value, String const& protocol_value, PatternProcessType type) -{ - // 1. If type is "pattern" then return portValue. - if (type == PatternProcessType::Pattern) - return port_value; - - // 2. Return the result of running canonicalize a port given portValue and protocolValue. - return canonicalize_a_port(port_value, protocol_value); -} - -// https://urlpattern.spec.whatwg.org/#process-pathname-for-init -static PatternErrorOr process_pathname_for_init(String const& pathname_value, String const& protocol_value, PatternProcessType type) -{ - // 1. If type is "pattern" then return pathnameValue. - if (type == PatternProcessType::Pattern) - return pathname_value; - - // 2. If protocolValue is a special scheme or the empty string, then return the result of running canonicalize a - // pathname given pathnameValue. - // NOTE: If the protocolValue is the empty string then no value was provided for protocol in the constructor - // dictionary. Normally we do not special case empty string dictionary values, but in this case we treat - // it as a special scheme in order to default to the most common pathname canonicalization. - if (protocol_value.is_empty() || is_special_scheme(protocol_value)) - return canonicalize_a_pathname(pathname_value); - - // 3. Return the result of running canonicalize an opaque pathname given pathnameValue. - return canonicalize_an_opaque_pathname(pathname_value); -} - -// https://urlpattern.spec.whatwg.org/#process-search-for-init -static String process_search_for_init(String const& value, PatternProcessType type) -{ - // 1. Let strippedValue be the given value with a single leading U+003F (?) removed, if any. - auto stripped_value = value; - if (stripped_value.starts_with('?')) - stripped_value = MUST(stripped_value.substring_from_byte_offset(1)); - - // 2. If type is "pattern" then return strippedValue. - if (type == PatternProcessType::Pattern) - return stripped_value; - - // 3. Return the result of running canonicalize a search given strippedValue. - return canonicalize_a_search(stripped_value); -} - -// https://urlpattern.spec.whatwg.org/#process-hash-for-init -static String process_hash_for_init(String const& value, PatternProcessType type) -{ - // 1. Let strippedValue be the given value with a single leading U+0023 (#) removed, if any. - auto stripped_value = value; - if (stripped_value.starts_with('#')) - stripped_value = MUST(stripped_value.substring_from_byte_offset(1)); - - // 2. If type is "pattern" then return strippedValue. - if (type == PatternProcessType::Pattern) - return stripped_value; - - // 3. Return the result of running canonicalize a hash given strippedValue. - return canonicalize_a_hash(stripped_value); -} - -// https://urlpattern.spec.whatwg.org/#process-a-urlpatterninit -PatternErrorOr process_a_url_pattern_init( - Init const& init, - PatternProcessType type, - Optional const& protocol, - Optional const& username, - Optional const& password, - Optional const& hostname, - Optional const& port, - Optional const& pathname, - Optional const& search, - Optional const& hash) -{ - // 1. Let result be the result of creating a new URLPatternInit. - Init result; - - // 2. If protocol is not null, set result["protocol"] to protocol. - if (protocol.has_value()) - result.protocol = protocol; - - // 3. If username is not null, set result["username"] to username. - if (username.has_value()) - result.username = username; - - // 4. If password is not null, set result["password"] to password. - if (password.has_value()) - result.password = password; - - // 5. If hostname is not null, set result["hostname"] to hostname. - if (hostname.has_value()) - result.hostname = hostname; - - // 6. If port is not null, set result["port"] to port. - if (port.has_value()) - result.port = port; - - // 7. If pathname is not null, set result["pathname"] to pathname. - if (pathname.has_value()) - result.pathname = pathname; - - // 8. If search is not null, set result["search"] to search. - if (search.has_value()) - result.search = search; - - // 9. If hash is not null, set result["hash"] to hash. - if (hash.has_value()) - result.hash = hash; - - // 10. Let baseURL be null. - Optional base_url; - - // 11. If init["baseURL"] exists: - if (init.base_url.has_value()) { - // 1. Set baseURL to the result of running the basic URL parser on init["baseURL"]. - base_url = Parser::basic_parse(init.base_url.value()); - - // 2. If baseURL is failure, then throw a TypeError. - if (!base_url.has_value()) - return ErrorInfo { MUST(String::formatted("Invalid base URL '{}' provided for URLPattern", init.base_url.value())) }; - - // 3. If init["protocol"] does not exist, then set result["protocol"] to the result of processing a base URL - // string given baseURL’s scheme and type. - if (!init.protocol.has_value()) - result.protocol = process_a_base_url_string(base_url->scheme(), type); - - // 4. If type is not "pattern" and init contains none of "protocol", "hostname", "port" and "username", then - // set result["username"] to the result of processing a base URL string given baseURL’s username and type. - if (type != PatternProcessType::Pattern && !init.protocol.has_value() && !init.hostname.has_value() - && !init.port.has_value() && !init.username.has_value()) { - result.username = process_a_base_url_string(base_url->username(), type); - } - - // 5. If type is not "pattern" and init contains none of "protocol", "hostname", "port", "username" and - // "password", then set result["password"] to the result of processing a base URL string given baseURL’s - // password and type. - if (type != PatternProcessType::Pattern && !init.protocol.has_value() && !init.hostname.has_value() - && !init.port.has_value() && !init.username.has_value() && !init.password.has_value()) { - result.password = process_a_base_url_string(base_url->password(), type); - } - - // 6. If init contains neither "protocol" nor "hostname", then: - if (!init.protocol.has_value() && !init.hostname.has_value()) { - // 1. Let baseHost be the serialization of baseURL's host, if it is not null, and the empty string otherwise. - String base_host = base_url->host().has_value() ? base_url->host()->serialize() : String {}; - - // 2. Set result["hostname"] to the result of processing a base URL string given baseHost and type. - result.hostname = process_a_base_url_string(base_host, type); - } - - // 7. If init contains none of "protocol", "hostname", and "port", then: - if (!init.protocol.has_value() && !init.hostname.has_value() && !init.port.has_value()) { - // 1. If baseURL’s port is null, then set result["port"] to the empty string. - if (!base_url->port().has_value()) { - result.port = String {}; - } - // 2. Otherwise, set result["port"] to baseURL’s port, serialized. - else { - result.port = String::number(*base_url->port()); - } - } - - // 8. If init contains none of "protocol", "hostname", "port", and "pathname", then set result["pathname"] to - // the result of processing a base URL string given the result of URL path serializing baseURL and type. - if (!init.protocol.has_value() && !init.hostname.has_value() && !init.port.has_value() && !init.pathname.has_value()) - result.pathname = process_a_base_url_string(base_url->serialize_path(), type); - - // 9. If init contains none of "protocol", "hostname", "port", "pathname", and "search", then: - if (!init.protocol.has_value() && !init.hostname.has_value() && !init.port.has_value() && !init.pathname.has_value() && !init.search.has_value()) { - // 1. Let baseQuery be baseURL’s query. - auto const& base_query = base_url->query(); - - // 2. If baseQuery is null, then set baseQuery to the empty string. - // 3. Set result["search"] to the result of processing a base URL string given baseQuery and type. - result.search = process_a_base_url_string(base_query.value_or(String {}), type); - } - - // 10. If init contains none of "protocol", "hostname", "port", "pathname", "search", and "hash", then: - if (!init.protocol.has_value() && !init.hostname.has_value() && !init.port.has_value() && !init.pathname.has_value() - && !init.search.has_value() && !init.hash.has_value()) { - // 1. Let baseFragment be baseURL’s fragment. - auto const& base_fragment = base_url->fragment(); - - // 2. If baseFragment is null, then set baseFragment to the empty string. - // 3. Set result["hash"] to the result of processing a base URL string given baseFragment and type. - result.hash = process_a_base_url_string(base_fragment.value_or(String {}), type); - } - } - - // 12. If init["protocol"] exists, then set result["protocol"] to the result of process protocol for init given init["protocol"] and type. - if (init.protocol.has_value()) - result.protocol = TRY(process_protocol_for_init(init.protocol.value(), type)); - - // 13. If init["username"] exists, then set result["username"] to the result of process username for init given init["username"] and type. - if (init.username.has_value()) - result.username = process_username_for_init(init.username.value(), type); - - // 14. If init["password"] exists, then set result["password"] to the result of process password for init given init["password"] and type. - if (init.password.has_value()) - result.password = process_password_for_init(init.password.value(), type); - - // 15. If init["hostname"] exists, then set result["hostname"] to the result of process hostname for init given init["hostname"] and type. - if (init.hostname.has_value()) - result.hostname = TRY(process_hostname_for_init(init.hostname.value(), type)); - - // 16. Let resultProtocolString be result["protocol"] if it exists; otherwise the empty string. - auto result_protocol_string = result.protocol.value_or(String {}); - - // 17. If init["port"] exists, then set result["port"] to the result of process port for init given init["port"], resultProtocolString, and type. - if (init.port.has_value()) - result.port = TRY(process_port_for_init(init.port.value(), result_protocol_string, type)); - - // 18. If init["pathname"] exists: - if (init.pathname.has_value()) { - // 1. Set result["pathname"] to init["pathname"]. - result.pathname = init.pathname.value(); - - // 2. If the following are all true: - // * baseURL is not null; - // * baseURL does not have an opaque path; and - // * the result of running is an absolute pathname given result["pathname"] and type is false, - // then: - if (base_url.has_value() - && !base_url->has_an_opaque_path() - && !is_an_absolute_pathname(result.pathname.value(), type)) { - // 1. Let baseURLPath be the result of running process a base URL string given the result of URL path - // serializing baseURL and type. - auto base_url_path = process_a_base_url_string(base_url->serialize_path(), type); - - // 2. Let slash index be the index of the last U+002F (/) code point found in baseURLPath, interpreted as a - // sequence of code points, or null if there are no instances of the code point. - auto slash_index = base_url_path.bytes_as_string_view().find_last('/'); - - // 3. If slash index is not null: - if (slash_index.has_value()) { - // 1. Let new pathname be the code point substring from 0 to slash index + 1 within baseURLPath. - auto new_pathname = base_url_path.bytes_as_string_view().substring_view(0, *slash_index + 1); - - // 2. Append result["pathname"] to the end of new pathname. - // 3. Set result["pathname"] to new pathname. - result.pathname = MUST(String::formatted("{}{}", new_pathname, *result.pathname)); - } - } - - // 3. Set result["pathname"] to the result of process pathname for init given result["pathname"], resultProtocolString, and type. - result.pathname = TRY(process_pathname_for_init(result.pathname.value(), result_protocol_string, type)); - } - - // 19. If init["search"] exists then set result["search"] to the result of process search for init given init["search"] and type. - if (init.search.has_value()) - result.search = process_search_for_init(init.search.value(), type); - - // 20. If init["hash"] exists then set result["hash"] to the result of process hash for init given init["hash"] and type. - if (init.hash.has_value()) - result.hash = process_hash_for_init(init.hash.value(), type); - - // 21. Return result. - return result; -} - -} diff --git a/Libraries/LibURL/Pattern/Init.h b/Libraries/LibURL/Pattern/Init.h deleted file mode 100644 index aab1531ab3..0000000000 --- a/Libraries/LibURL/Pattern/Init.h +++ /dev/null @@ -1,45 +0,0 @@ -/* - * Copyright (c) 2025, Shannon Booth - * - * SPDX-License-Identifier: BSD-2-Clause - */ - -#pragma once - -#include -#include -#include - -namespace URL::Pattern { - -// https://urlpattern.spec.whatwg.org/#dictdef-urlpatterninit -struct Init { - Optional protocol; - Optional username; - Optional password; - Optional hostname; - Optional port; - Optional pathname; - Optional search; - Optional hash; - Optional base_url; -}; - -enum class PatternProcessType { - Pattern, - URL, -}; - -PatternErrorOr process_a_url_pattern_init( - Init const&, - PatternProcessType type, - Optional const& protocol, - Optional const& username, - Optional const& password, - Optional const& hostname, - Optional const& port, - Optional const& pathname, - Optional const& search, - Optional const& hash); - -} diff --git a/Libraries/LibURL/Pattern/Options.cpp b/Libraries/LibURL/Pattern/Options.cpp deleted file mode 100644 index dbeabc756e..0000000000 --- a/Libraries/LibURL/Pattern/Options.cpp +++ /dev/null @@ -1,41 +0,0 @@ -/* - * Copyright (c) 2025, Shannon Booth - * - * SPDX-License-Identifier: BSD-2-Clause - */ - -#include - -namespace URL::Pattern { - -// https://urlpattern.spec.whatwg.org/#default-options -Options Options::default_() -{ - // The default options is an options struct with delimiter code point set to the empty string and prefix code point set to the empty string. - return { - .delimiter_code_point = {}, - .prefix_code_point = {}, - }; -} - -// https://urlpattern.spec.whatwg.org/#hostname-options -Options Options::hostname() -{ - // The hostname options is an options struct with delimiter code point set "." and prefix code point set to the empty string. - return { - .delimiter_code_point = '.', - .prefix_code_point = {}, - }; -} - -// https://urlpattern.spec.whatwg.org/#pathname-options -Options Options::pathname() -{ - // The pathname options is an options struct with delimiter code point set "/" and prefix code point set to "/". - return { - .delimiter_code_point = '/', - .prefix_code_point = '/', - }; -} - -} diff --git a/Libraries/LibURL/Pattern/Options.h b/Libraries/LibURL/Pattern/Options.h deleted file mode 100644 index 47e2e4e695..0000000000 --- a/Libraries/LibURL/Pattern/Options.h +++ /dev/null @@ -1,30 +0,0 @@ -/* - * Copyright (c) 2025, Shannon Booth - * - * SPDX-License-Identifier: BSD-2-Clause - */ - -#pragma once - -#include - -namespace URL::Pattern { - -// https://urlpattern.spec.whatwg.org/#options -struct Options { - // https://urlpattern.spec.whatwg.org/#options-delimiter-code-point - Optional delimiter_code_point; - - // https://urlpattern.spec.whatwg.org/#options-prefix-code-point - Optional prefix_code_point; - - // https://urlpattern.spec.whatwg.org/#options-ignore-case - bool ignore_case { false }; - - static Options default_(); - static Options hostname(); - static Options pathname(); -}; -; - -} diff --git a/Libraries/LibURL/Pattern/Part.cpp b/Libraries/LibURL/Pattern/Part.cpp deleted file mode 100644 index 9c295b02aa..0000000000 --- a/Libraries/LibURL/Pattern/Part.cpp +++ /dev/null @@ -1,64 +0,0 @@ -/* - * Copyright (c) 2025, Shannon Booth - * - * SPDX-License-Identifier: BSD-2-Clause - */ - -#include -#include - -namespace URL::Pattern { - -Part::Part(Type type, String value, Modifier modifier) - : type(type) - , value(move(value)) - , modifier(modifier) -{ -} - -Part::Part(Type type, String value, Modifier modifier, String name, String prefix, String suffix) - : type(type) - , value(move(value)) - , modifier(modifier) - , name(move(name)) - , prefix(move(prefix)) - , suffix(move(suffix)) -{ -} - -StringView Part::type_to_string(Part::Type type) -{ - switch (type) { - case Type::FixedText: - return "FixedText"sv; - case Type::Regexp: - return "Regexp"sv; - case Type::SegmentWildcard: - return "SegmentWildcard"sv; - case Type::FullWildcard: - return "FullWildcard"sv; - } - - VERIFY_NOT_REACHED(); -} - -// https://urlpattern.spec.whatwg.org/#convert-a-modifier-to-a-string -StringView Part::convert_modifier_to_string(Part::Modifier modifier) -{ - // 1. If modifier is "zero-or-more", then return "*". - if (modifier == Modifier::ZeroOrMore) - return "*"sv; - - // 2. If modifier is "optional", then return "?". - if (modifier == Modifier::Optional) - return "?"sv; - - // 3. If modifier is "one-or-more", then return "+". - if (modifier == Modifier::OneOrMore) - return "+"sv; - - // 4. Return the empty string. - return ""sv; -} - -} diff --git a/Libraries/LibURL/Pattern/Part.h b/Libraries/LibURL/Pattern/Part.h deleted file mode 100644 index 0947bca913..0000000000 --- a/Libraries/LibURL/Pattern/Part.h +++ /dev/null @@ -1,79 +0,0 @@ -/* - * Copyright (c) 2025, Shannon Booth - * - * SPDX-License-Identifier: BSD-2-Clause - */ - -#pragma once - -#include -#include - -namespace URL::Pattern { - -// https://urlpattern.spec.whatwg.org/#part -struct Part { - // https://urlpattern.spec.whatwg.org/#part-type - enum class Type { - // The part represents a simple fixed text string. - FixedText, - - // The part represents a matching group with a custom regular expression. - Regexp, - - // The part represents a matching group that matches code points up to the next separator code point. This is - // typically used for a named group like ":foo" that does not have a custom regular expression. - SegmentWildcard, - - // The part represents a matching group that greedily matches all code points. This is typically used for - // the "*" wildcard matching group. - FullWildcard, - }; - - // https://urlpattern.spec.whatwg.org/#part-modifier - enum class Modifier { - // The part does not have a modifier. - None, - - // The part has an optional modifier indicated by the U+003F (?) code point. - Optional, - - // The part has a "zero or more" modifier indicated by the U+002A (*) code point. - ZeroOrMore, - - // The part has a "one or more" modifier indicated by the U+002B (+) code point. - OneOrMore, - }; - - static StringView convert_modifier_to_string(Modifier); - static StringView type_to_string(Type); - - Part(Type, String value, Modifier); - Part(Type, String value, Modifier, String name, String prefix, String suffix); - - // https://urlpattern.spec.whatwg.org/#part-type - // A part has an associated type, a string, which must be set upon creation. - Type type {}; - - // https://urlpattern.spec.whatwg.org/#part-value - // A part has an associated value, a string, which must be set upon creation. - String value; - - // https://urlpattern.spec.whatwg.org/#part-modifier - // A part has an associated modifier a string, which must be set upon creation. - Modifier modifier; - - // https://urlpattern.spec.whatwg.org/#part-name - // A part has an associated name, a string, initially the empty string. - String name; - - // https://urlpattern.spec.whatwg.org/#part-prefix - // A part has an associated prefix, a string, initially the empty string. - String prefix; - - // https://urlpattern.spec.whatwg.org/#part-suffix - // A part has an associated suffix, a string, initially the empty string. - String suffix; -}; - -} diff --git a/Libraries/LibURL/Pattern/Pattern.cpp b/Libraries/LibURL/Pattern/Pattern.cpp deleted file mode 100644 index f1dde1a3e4..0000000000 --- a/Libraries/LibURL/Pattern/Pattern.cpp +++ /dev/null @@ -1,434 +0,0 @@ -/* - * Copyright (c) 2025, Shannon Booth - * - * SPDX-License-Identifier: BSD-2-Clause - */ - -#include -#include -#include -#include -#include - -namespace URL::Pattern { - -// https://urlpattern.spec.whatwg.org/#hostname-pattern-is-an-ipv6-address -static bool hostname_pattern_is_an_ipv6_address(String const& input) -{ - // 1. If input’s code point length is less than 2, then return false. - if (input.bytes().size() < 2) - return false; - - // 2. Let input code points be input interpreted as a list of code points. - auto input_code_points = input.bytes(); - - // 3. If input code points[0] is U+005B ([), then return true. - if (input_code_points[0] == '[') - return true; - - // 4. If input code points[0] is U+007B ({) and input code points[1] is U+005B ([), then return true. - if (input_code_points[0] == '{' && input_code_points[1] == '[') - return true; - - // 5. If input code points[0] is U+005C (\) and input code points[1] is U+005B ([), then return true. - if (input_code_points[0] == '\\' && input_code_points[1] == '[') - return true; - - // 6. Return false. - return false; -} - -// https://urlpattern.spec.whatwg.org/#url-pattern-create -PatternErrorOr Pattern::create(Input const& input, Optional const& base_url, IgnoreCase ignore_case) -{ - // 1. Let init be null. - Init init; - - // 2. If input is a scalar value string then: - if (auto const* input_string = input.get_pointer()) { - // 1. Set init to the result of running parse a constructor string given input. - init = TRY(ConstructorStringParser::parse(input_string->code_points())); - - // 2. If baseURL is null and init["protocol"] does not exist, then throw a TypeError. - if (!base_url.has_value() && !init.protocol.has_value()) - return ErrorInfo { "Relative URLPattern constructor must provide one of baseURL or protocol"_string }; - - // 3. If baseURL is not null, set init["baseURL"] to baseURL. - if (base_url.has_value()) - init.base_url = base_url; - } - // 3. Otherwise: - else { - // 1. Assert: input is a URLPatternInit. - VERIFY(input.has()); - - // 2. If baseURL is not null, then throw a TypeError. - if (base_url.has_value()) - return ErrorInfo { "Constructor with URLPatternInit should provide no baseURL"_string }; - - // 3. Set init to input. - init = input.get(); - } - - // 4. Let processedInit be the result of process a URLPatternInit given init, "pattern", null, null, null, null, null, null, null, and null. - auto processed_init = TRY(process_a_url_pattern_init(init, PatternProcessType::Pattern, {}, {}, {}, {}, {}, {}, {}, {})); - - // 5. For each componentName of « "protocol", "username", "password", "hostname", "port", "pathname", "search", "hash" »: - // 1. If processedInit[componentName] does not exist, then set processedInit[componentName] to "*". - if (!processed_init.protocol.has_value()) - processed_init.protocol = "*"_string; - if (!processed_init.username.has_value()) - processed_init.username = "*"_string; - if (!processed_init.password.has_value()) - processed_init.password = "*"_string; - if (!processed_init.hostname.has_value()) - processed_init.hostname = "*"_string; - if (!processed_init.port.has_value()) - processed_init.port = "*"_string; - if (!processed_init.pathname.has_value()) - processed_init.pathname = "*"_string; - if (!processed_init.search.has_value()) - processed_init.search = "*"_string; - if (!processed_init.hash.has_value()) - processed_init.hash = "*"_string; - - // 6. If processedInit["protocol"] is a special scheme and processedInit["port"] is a string which represents its - // corresponding default port in radix-10 using ASCII digits then set processedInit["port"] to the empty string. - if (is_special_scheme(processed_init.protocol.value())) { - auto maybe_port = processed_init.port->to_number(TrimWhitespace::No); - if (maybe_port.has_value() && *maybe_port == default_port_for_scheme(*processed_init.protocol).value()) - processed_init.port = String {}; - } - - // 7. Let urlPattern be a new URL pattern. - Pattern url_pattern; - - // 8. Set urlPattern’s protocol component to the result of compiling a component given processedInit["protocol"], - // canonicalize a protocol, and default options. - url_pattern.m_protocol_component = TRY(Component::compile(processed_init.protocol->code_points(), canonicalize_a_protocol, Options::default_())); - - // 9. Set urlPattern’s username component to the result of compiling a component given processedInit["username"], - // canonicalize a username, and default options. - url_pattern.m_username_component = TRY(Component::compile(processed_init.username->code_points(), canonicalize_a_username, Options::default_())); - - // 10. Set urlPattern’s password component to the result of compiling a component given processedInit["password"], - // canonicalize a password, and default options. - url_pattern.m_password_component = TRY(Component::compile(processed_init.password->code_points(), canonicalize_a_password, Options::default_())); - - // 11. If the result running hostname pattern is an IPv6 address given processedInit["hostname"] is true, then set - // urlPattern’s hostname component to the result of compiling a component given processedInit["hostname"], - // canonicalize an IPv6 hostname, and hostname options. - if (hostname_pattern_is_an_ipv6_address(processed_init.hostname.value())) { - url_pattern.m_hostname_component = TRY(Component::compile(processed_init.hostname->code_points(), canonicalize_an_ipv6_hostname, Options::hostname())); - } - // 12. Otherwise, set urlPattern’s hostname component to the result of compiling a component given - // processedInit["hostname"], canonicalize a hostname, and hostname options. - else { - url_pattern.m_hostname_component = TRY(Component::compile(processed_init.hostname->code_points(), canonicalize_a_hostname, Options::hostname())); - } - - // 13. Set urlPattern’s port component to the result of compiling a component given processedInit["port"], - // canonicalize a port, and default options. - url_pattern.m_port_component = TRY(Component::compile(processed_init.port->code_points(), [](String const& value) { return canonicalize_a_port(value); }, Options::default_())); - - // 14. Let compileOptions be a copy of the default options with the ignore case property set to options["ignoreCase"]. - auto compile_options = Options::default_(); - compile_options.ignore_case = ignore_case == IgnoreCase::Yes; - - // 15. If the result of running protocol component matches a special scheme given urlPattern’s protocol component is true, then: - if (protocol_component_matches_a_special_scheme(url_pattern.m_protocol_component)) { - // 1. Let pathCompileOptions be copy of the pathname options with the ignore case property set to options["ignoreCase"]. - auto path_compile_options = Options::pathname(); - path_compile_options.ignore_case = ignore_case == IgnoreCase::Yes; - - // 2. Set urlPattern’s pathname component to the result of compiling a component given processedInit["pathname"], - // canonicalize a pathname, and pathCompileOptions. - url_pattern.m_pathname_component = TRY(Component::compile(processed_init.pathname->code_points(), canonicalize_a_pathname, path_compile_options)); - } - // 16. Otherwise set urlPattern’s pathname component to the result of compiling a component given - // processedInit["pathname"], canonicalize an opaque pathname, and compileOptions. - else { - url_pattern.m_pathname_component = TRY(Component::compile(processed_init.pathname->code_points(), canonicalize_an_opaque_pathname, compile_options)); - } - - // 17. Set urlPattern’s search component to the result of compiling a component given processedInit["search"], - // canonicalize a search, and compileOptions. - url_pattern.m_search_component = TRY(Component::compile(processed_init.search->code_points(), canonicalize_a_search, compile_options)); - - // 18. Set urlPattern’s hash component to the result of compiling a component given processedInit["hash"], - // canonicalize a hash, and compileOptions. - url_pattern.m_hash_component = TRY(Component::compile(processed_init.hash->code_points(), canonicalize_a_hash, compile_options)); - - // 19. Return urlPattern. - return url_pattern; -} - -// https://urlpattern.spec.whatwg.org/#url-pattern-match -PatternErrorOr> Pattern::match(Variant const& input, Optional const& base_url_string) const -{ - // 1. Let protocol be the empty string. - String protocol; - - // 2. Let username be the empty string. - String username; - - // 3. Let password be the empty string. - String password; - - // 4. Let hostname be the empty string. - String hostname; - - // 5. Let port be the empty string. - String port; - - // 6. Let pathname be the empty string. - String pathname; - - // 7. Let search be the empty string. - String search; - - // 8. Let hash be the empty string. - String hash; - - // 9. Let inputs be an empty list. - Vector inputs; - - // 10. If input is a URL, then append the serialization of input to inputs. - if (auto const* input_url = input.get_pointer()) { - inputs.append(input_url->serialize()); - } - // 11. Otherwise, append input to inputs. - else { - inputs.append(input.downcast()); - } - - // 12. If input is a URLPatternInit then: - if (auto const* input_init = input.get_pointer()) { - // 1. If baseURLString was given, throw a TypeError. - if (base_url_string.has_value()) - return ErrorInfo { "Base URL cannot be provided when URLPatternInput is provided"_string }; - - // 2. Let applyResult be the result of process a URLPatternInit given input, "url", protocol, username, password, - // hostname, port, pathname, search, and hash. If this throws an exception, catch it, and return null. - auto apply_result_or_error = process_a_url_pattern_init(*input_init, PatternProcessType::URL, - protocol, username, password, hostname, port, pathname, search, hash); - if (apply_result_or_error.is_error()) - return OptionalNone {}; - auto apply_result = apply_result_or_error.release_value(); - - // 3. Set protocol to applyResult["protocol"]. - protocol = apply_result.protocol.value(); - - // 4. Set username to applyResult["username"]. - username = apply_result.username.value(); - - // 5. Set password to applyResult["password"]. - password = apply_result.password.value(); - - // 6. Set hostname to applyResult["hostname"]. - hostname = apply_result.hostname.value(); - - // 7. Set port to applyResult["port"]. - port = apply_result.port.value(); - - // 8. Set pathname to applyResult["pathname"]. - pathname = apply_result.pathname.value(); - - // 9. Set search to applyResult["search"]. - search = apply_result.search.value(); - - // 10. Set hash to applyResult["hash"]. - hash = apply_result.hash.value(); - } - // 13. Otherwise: - else { - // 1. Let url be input. - auto url_or_string = input.downcast(); - - // 2. If input is a USVString: - if (auto const* input_string = input.get_pointer()) { - // 1. Let baseURL be null. - Optional base_url; - - // 2. If baseURLString was given, then: - if (base_url_string.has_value()) { - // 1. Set baseURL to the result of running the basic URL parser on baseURLString. - base_url = Parser::basic_parse(base_url_string.value()); - - // 2. If baseURL is failure, return null. - if (!base_url.has_value()) - return OptionalNone {}; - - // 3. Append baseURLString to inputs. - inputs.append(base_url_string.value()); - } - - // 3. Set url to the result of running the basic URL parser on input with baseURL. - // 4. If url is failure, return null. - auto maybe_url = Parser::basic_parse(*input_string, base_url); - if (!maybe_url.has_value()) - return OptionalNone {}; - url_or_string = maybe_url.release_value(); - } - - // 3. Assert: url is a URL. - VERIFY(url_or_string.has()); - auto& url = url_or_string.get(); - - // 4. Set protocol to url’s scheme. - protocol = url.scheme(); - - // 5. Set username to url’s username. - username = url.username(); - - // 6. Set password to url’s password. - password = url.password(); - - // 7. Set hostname to url’s host, serialized, or the empty string if the value is null. - if (url.host().has_value()) - hostname = url.host()->serialize(); - else - hostname = String {}; - - // 8. Set port to url’s port, serialized, or the empty string if the value is null. - if (url.port().has_value()) - port = String::number(url.port().value()); - else - port = String {}; - - // 9. Set pathname to the result of URL path serializing url. - pathname = url.serialize_path(); - - // 10. Set search to url’s query or the empty string if the value is null. - search = url.query().value_or(String {}); - - // 11. Set hash to url’s fragment or the empty string if the value is null. - hash = url.fragment().value_or(String {}); - } - - // 14. Let protocolExecResult be RegExpBuiltinExec(urlPattern’s protocol component's regular expression, protocol). - auto protocol_exec_result = m_protocol_component.execute(protocol); - if (!protocol_exec_result.success) - return OptionalNone {}; - - // 15. Let usernameExecResult be RegExpBuiltinExec(urlPattern’s username component's regular expression, username). - auto username_exec_result = m_username_component.execute(username); - if (!username_exec_result.success) - return OptionalNone {}; - - // 16. Let passwordExecResult be RegExpBuiltinExec(urlPattern’s password component's regular expression, password). - auto password_exec_result = m_password_component.execute(password); - if (!password_exec_result.success) - return OptionalNone {}; - - // 17. Let hostnameExecResult be RegExpBuiltinExec(urlPattern’s hostname component's regular expression, hostname). - auto hostname_exec_result = m_hostname_component.execute(hostname); - if (!hostname_exec_result.success) - return OptionalNone {}; - - // 18. Let portExecResult be RegExpBuiltinExec(urlPattern’s port component's regular expression, port). - auto port_exec_result = m_port_component.execute(port); - if (!port_exec_result.success) - return OptionalNone {}; - - // 19. Let pathnameExecResult be RegExpBuiltinExec(urlPattern’s pathname component's regular expression, pathname). - auto pathname_exec_result = m_pathname_component.execute(pathname); - if (!pathname_exec_result.success) - return OptionalNone {}; - - // 20. Let searchExecResult be RegExpBuiltinExec(urlPattern’s search component's regular expression, search). - auto search_exec_result = m_search_component.execute(search); - if (!search_exec_result.success) - return OptionalNone {}; - - // 21. Let hashExecResult be RegExpBuiltinExec(urlPattern’s hash component's regular expression, hash). - auto hash_exec_result = m_hash_component.execute(hash); - if (!hash_exec_result.success) - return OptionalNone {}; - - // 22. If protocolExecResult, usernameExecResult, passwordExecResult, hostnameExecResult, portExecResult, - // pathnameExecResult, searchExecResult, or hashExecResult are null then return null. - // NOTE: Done in steps above at point of exec. - - // 23. Let result be a new URLPatternResult. - Result result; - - // 24. Set result["inputs"] to inputs. - result.inputs = move(inputs); - - // 25. Set result["protocol"] to the result of creating a component match result given urlPattern’s protocol - // component, protocol, and protocolExecResult. - result.protocol = m_protocol_component.create_match_result(protocol, protocol_exec_result); - - // 26. Set result["username"] to the result of creating a component match result given urlPattern’s username - // component, username, and usernameExecResult. - result.username = m_username_component.create_match_result(username, username_exec_result); - - // 27. Set result["password"] to the result of creating a component match result given urlPattern’s password - // component, password, and passwordExecResult. - result.password = m_password_component.create_match_result(password, password_exec_result); - - // 28. Set result["hostname"] to the result of creating a component match result given urlPattern’s hostname - // component, hostname, and hostnameExecResult. - result.hostname = m_hostname_component.create_match_result(hostname, hostname_exec_result); - - // 29. Set result["port"] to the result of creating a component match result given urlPattern’s port component, - // port, and portExecResult. - result.port = m_port_component.create_match_result(port, port_exec_result); - - // 30. Set result["pathname"] to the result of creating a component match result given urlPattern’s pathname - // component, pathname, and pathnameExecResult. - result.pathname = m_pathname_component.create_match_result(pathname, pathname_exec_result); - - // 31. Set result["search"] to the result of creating a component match result given urlPattern’s search component, - // search, and searchExecResult. - result.search = m_search_component.create_match_result(search, search_exec_result); - - // 32. Set result["hash"] to the result of creating a component match result given urlPattern’s hash component, - // hash, and hashExecResult. - result.hash = m_hash_component.create_match_result(hash, hash_exec_result); - - // 33. Return result. - return result; -} - -// https://urlpattern.spec.whatwg.org/#url-pattern-has-regexp-groups -bool Pattern::has_regexp_groups() const -{ - // 1. If urlPattern’s protocol component has regexp groups is true, then return true. - if (m_protocol_component.has_regexp_groups) - return true; - - // 2. If urlPattern’s username component has regexp groups is true, then return true. - if (m_username_component.has_regexp_groups) - return true; - - // 3. If urlPattern’s password component has regexp groups is true, then return true. - if (m_password_component.has_regexp_groups) - return true; - - // 4. If urlPattern’s hostname component has regexp groups is true, then return true. - if (m_hostname_component.has_regexp_groups) - return true; - - // 5. If urlPattern’s port component has regexp groups is true, then return true. - if (m_port_component.has_regexp_groups) - return true; - - // 6. If urlPattern’s pathname component has regexp groups is true, then return true. - if (m_pathname_component.has_regexp_groups) - return true; - - // 7. If urlPattern’s search component has regexp groups is true, then return true. - if (m_search_component.has_regexp_groups) - return true; - - // 8. If urlPattern’s hash component has regexp groups is true, then return true. - if (m_hash_component.has_regexp_groups) - return true; - - // 9. Return false. - return false; -} - -} diff --git a/Libraries/LibURL/Pattern/Pattern.h b/Libraries/LibURL/Pattern/Pattern.h deleted file mode 100644 index 8a76716d89..0000000000 --- a/Libraries/LibURL/Pattern/Pattern.h +++ /dev/null @@ -1,95 +0,0 @@ -/* - * Copyright (c) 2025, Shannon Booth - * - * SPDX-License-Identifier: BSD-2-Clause - */ - -#pragma once - -#include -#include -#include -#include -#include -#include -#include -#include - -namespace URL::Pattern { - -// https://urlpattern.spec.whatwg.org/#typedefdef-urlpatterninput -using Input = Variant; - -// https://urlpattern.spec.whatwg.org/#dictdef-urlpatternresult -struct Result { - Vector inputs; - - Component::Result protocol; - Component::Result username; - Component::Result password; - Component::Result hostname; - Component::Result port; - Component::Result pathname; - Component::Result search; - Component::Result hash; -}; - -// https://urlpattern.spec.whatwg.org/#dictdef-urlpatternoptions -enum class IgnoreCase { - Yes, - No, -}; - -// https://urlpattern.spec.whatwg.org/#url-pattern -class Pattern { -public: - static PatternErrorOr create(Input const&, Optional const& base_url = {}, IgnoreCase = IgnoreCase::No); - - PatternErrorOr> match(Variant const&, Optional const& base_url_string) const; - - bool has_regexp_groups() const; - - Component const& protocol_component() const { return m_protocol_component; } - Component const& username_component() const { return m_username_component; } - Component const& password_component() const { return m_password_component; } - Component const& hostname_component() const { return m_hostname_component; } - Component const& port_component() const { return m_port_component; } - Component const& pathname_component() const { return m_pathname_component; } - Component const& search_component() const { return m_search_component; } - Component const& hash_component() const { return m_hash_component; } - -private: - // https://urlpattern.spec.whatwg.org/#url-pattern-protocol-component - // protocol component, a component - Component m_protocol_component; - - // https://urlpattern.spec.whatwg.org/#url-pattern-username-component - // username component, a component - Component m_username_component; - - // https://urlpattern.spec.whatwg.org/#url-pattern-password-component - // password component, a component - Component m_password_component; - - // https://urlpattern.spec.whatwg.org/#url-pattern-hostname-component - // hostname component, a component - Component m_hostname_component; - - // https://urlpattern.spec.whatwg.org/#url-pattern-port-component - // port component, a component - Component m_port_component; - - // https://urlpattern.spec.whatwg.org/#url-pattern-pathname-component - // pathname component, a component - Component m_pathname_component; - - // https://urlpattern.spec.whatwg.org/#url-pattern-search-component - // search component, a component - Component m_search_component; - - // https://urlpattern.spec.whatwg.org/#url-pattern-hash-component - // hash component, a component - Component m_hash_component; -}; - -} diff --git a/Libraries/LibURL/Pattern/PatternError.h b/Libraries/LibURL/Pattern/PatternError.h deleted file mode 100644 index 7d81b3cd16..0000000000 --- a/Libraries/LibURL/Pattern/PatternError.h +++ /dev/null @@ -1,23 +0,0 @@ -/* - * Copyright (c) 2025, Shannon Booth - * - * SPDX-License-Identifier: BSD-2-Clause - */ - -#pragma once - -#include -#include - -namespace URL::Pattern { - -// NOTE: All exceptions which are thrown by the URLPattern spec are TypeErrors which web-based callers are expected to assume. -// If this ever does not become the case, this should change to also include the error type. -struct ErrorInfo { - String message; -}; - -template -using PatternErrorOr = AK::ErrorOr; - -} diff --git a/Libraries/LibURL/Pattern/PatternParser.cpp b/Libraries/LibURL/Pattern/PatternParser.cpp deleted file mode 100644 index ebd7be9f64..0000000000 --- a/Libraries/LibURL/Pattern/PatternParser.cpp +++ /dev/null @@ -1,409 +0,0 @@ -/* - * Copyright (c) 2025, Shannon Booth - * - * SPDX-License-Identifier: BSD-2-Clause - */ - -#include -#include -#include - -namespace URL::Pattern { - -PatternParser::PatternParser(EncodingCallback encoding_callback, String segment_wildcard_regexp) - : m_encoding_callback(move(encoding_callback)) - , m_segment_wildcard_regexp(move(segment_wildcard_regexp)) -{ -} - -// https://urlpattern.spec.whatwg.org/#consume-a-required-token -PatternErrorOr PatternParser::consume_a_required_token(Token::Type type) -{ - // 1. Let result be the result of running try to consume a token given parser and type. - auto result = try_to_consume_a_token(type); - - // 2. If result is null, then throw a TypeError. - if (!result.has_value()) - return ErrorInfo { MUST(String::formatted("Missing required token '{}' in URL pattern", Token::type_to_string(type))) }; - - // 3. Return result. - // NOTE: No caller actually needs the result, so we just ignore it. - return {}; -} - -// https://urlpattern.spec.whatwg.org/#consume-text -String PatternParser::consume_text() -{ - // 1. Let result be the empty string. - StringBuilder result; - - // 1. While true: - while (true) { - // 1. Let token be the result of running try to consume a token given parser and "char". - auto token = try_to_consume_a_token(Token::Type::Char); - - // 2. If token is null, then set token to the result of running try to consume a token given parser and "escaped-char". - if (!token.has_value()) - token = try_to_consume_a_token(Token::Type::EscapedChar); - - // 3. If token is null, then break. - if (!token.has_value()) - break; - - // 4. Append token’s value to the end of result. - result.append(token->value); - } - - // 2. Return result. - return result.to_string_without_validation(); -} - -// https://urlpattern.spec.whatwg.org/#maybe-add-a-part-from-the-pending-fixed-value -PatternErrorOr PatternParser::maybe_add_a_part_from_the_pending_fixed_value() -{ - // 1. If parser’s pending fixed value is the empty string, then return. - if (m_pending_fixed_value.is_empty()) - return {}; - - // 2. Let encoded value be the result of running parser’s encoding callback given parser’s pending fixed value. - auto encoded_value = TRY(m_encoding_callback(m_pending_fixed_value.to_string_without_validation())); - - // 3. Set parser’s pending fixed value to the empty string. - m_pending_fixed_value.clear(); - - // 4. Let part be a new part whose type is "fixed-text", value is encoded value, and modifier is "none". - // 5. Append part to parser’s part list. - m_part_list.append({ Part::Type::FixedText, move(encoded_value), Part::Modifier::None }); - - return {}; -} - -// https://urlpattern.spec.whatwg.org/#is-a-duplicate-name -bool PatternParser::is_a_duplicate_name(String const& name) const -{ - // 1. For each part of parser’s part list: - for (auto const& part : m_part_list) { - // 1. If part’s name is name, then return true. - if (part.name == name) - return true; - } - - // 2. Return false. - return false; -} - -// https://urlpattern.spec.whatwg.org/#add-a-part -PatternErrorOr PatternParser::add_a_part( - String const& prefix, - Optional name_token, - Optional regexp_or_wildcard_token, - String const& suffix, - Optional modifier_token) -{ - // 1. Let modifier be "none". - auto modifier = Part::Modifier::None; - - // 2. If modifier token is not null: - if (modifier_token.has_value()) { - // 1. If modifier token’s value is "?" then set modifier to "optional". - if (modifier_token->value == "?"sv) { - modifier = Part::Modifier::Optional; - } - // 2. Otherwise if modifier token’s value is "*" then set modifier to "zero-or-more". - else if (modifier_token->value == "*"sv) { - modifier = Part::Modifier::ZeroOrMore; - } - // 3. Otherwise if modifier token’s value is "+" then set modifier to "one-or-more". - else if (modifier_token->value == "+"sv) { - modifier = Part::Modifier::OneOrMore; - } - } - - // 3. If name token is null and regexp or wildcard token is null and modifier is "none": - // NOTE: This was a "{foo}" grouping. We add this to the pending fixed value so that it will be combined with - // any previous or subsequent text. - if (!name_token.has_value() && !regexp_or_wildcard_token.has_value() && modifier == Part::Modifier::None) { - // 1. Append prefix to the end of parser’s pending fixed value. - m_pending_fixed_value.append(prefix); - - // 2. Return. - return {}; - } - - // 4. Run maybe add a part from the pending fixed value given parser. - TRY(maybe_add_a_part_from_the_pending_fixed_value()); - - // 5. If name token is null and regexp or wildcard token is null: - // NOTE: This was a "{foo}?" grouping. The modifier means we cannot combine it with other text. Therefore we - // add it as a part immediately. - if (!name_token.has_value() && !regexp_or_wildcard_token.has_value()) { - // 1. Assert: suffix is the empty string. - VERIFY(suffix.is_empty()); - - // 2. If prefix is the empty string, then return. - if (prefix.is_empty()) - return {}; - - // 3. Let encoded value be the result of running parser’s encoding callback given prefix. - auto encoded_value = TRY(m_encoding_callback(prefix)); - - // 4. Let part be a new part whose type is "fixed-text", value is encoded value, and modifier is modifier. - // 5. Append part to parser’s part list. - m_part_list.append({ Part::Type::FixedText, move(encoded_value), modifier }); - - // 6. Return. - return {}; - } - - // 6. Let regexp value be the empty string. - // NOTE: Next, we convert the regexp or wildcard token into a regular expression. - String regexp_value; - - // 7. If regexp or wildcard token is null, then set regexp value to parser’s segment wildcard regexp. - if (!regexp_or_wildcard_token.has_value()) { - regexp_value = m_segment_wildcard_regexp; - } - // 8. Otherwise if regexp or wildcard token’s type is "asterisk", then set regexp value to the full wildcard regexp value. - else if (regexp_or_wildcard_token->type == Token::Type::Asterisk) { - regexp_value = MUST(String::from_utf8(full_wildcard_regexp_value)); - } - // 9. Otherwise set regexp value to regexp or wildcard token’s value. - else { - regexp_value = regexp_or_wildcard_token->value; - } - - // 10. Let type be "regexp". - // NOTE: Next, we convert regexp value into a part type. We make sure to go to a regular expression first so - // that an equivalent "regexp" token will be treated the same as a "name" or "asterisk" token. - auto type = Part::Type::Regexp; - - // 11. If regexp value is parser’s segment wildcard regexp: - if (regexp_value == m_segment_wildcard_regexp) { - // 1. Set type to "segment-wildcard". - type = Part::Type::SegmentWildcard; - - // 2. Set regexp value to the empty string. - regexp_value = String {}; - } - // 12. Otherwise if regexp value is the full wildcard regexp value: - else if (regexp_value == full_wildcard_regexp_value) { - // 1. Set type to "full-wildcard". - type = Part::Type::FullWildcard; - - // 2. Set regexp value to the empty string. - regexp_value = String {}; - } - - // 13. Let name be the empty string. - // NOTE: Next, we determine the part name. This can be explicitly provided by a "name" token or be automatically assigned. - String name; - - // 14. If name token is not null, then set name to name token’s value. - if (name_token.has_value()) { - name = name_token->value; - } - // 15. Otherwise if regexp or wildcard token is not null: - else if (regexp_or_wildcard_token.has_value()) { - // 1. Set name to parser’s next numeric name, serialized. - name = String::number(m_next_numeric_name); - - // 2. Increment parser’s next numeric name by 1. - ++m_next_numeric_name; - } - - // 16. If the result of running is a duplicate name given parser and name is true, then throw a TypeError. - if (is_a_duplicate_name(name)) - return ErrorInfo { MUST(String::formatted("Duplicate name '{}' provided in URL pattern", name)) }; - - // 17. Let encoded prefix be the result of running parser’s encoding callback given prefix. - // NOTE: Finally, we encode the fixed text values and create the part. - auto encoded_prefix = TRY(m_encoding_callback(prefix)); - - // 18. Let encoded suffix be the result of running parser’s encoding callback given suffix. - auto encoded_suffix = TRY(m_encoding_callback(suffix)); - - // 19. Let part be a new part whose type is type, value is regexp value, modifier is modifier, name is name, prefix - // is encoded prefix, and suffix is encoded suffix. - // 20. Append part to parser’s part list. - m_part_list.append({ type, move(regexp_value), modifier, move(name), move(encoded_prefix), move(encoded_suffix) }); - - return {}; -} - -// https://urlpattern.spec.whatwg.org/#try-to-consume-a-modifier-token -Optional PatternParser::try_to_consume_a_modifier_token() -{ - // 1. Let token be the result of running try to consume a token given parser and "other-modifier". - auto token = try_to_consume_a_token(Token::Type::OtherModifier); - - // 2. If token is not null, then return token. - if (token.has_value()) - return token; - - // 3. Set token to the result of running try to consume a token given parser and "asterisk". - token = try_to_consume_a_token(Token::Type::Asterisk); - - // 4. Return token. - return token; -} - -// https://urlpattern.spec.whatwg.org/#try-to-consume-a-regexp-or-wildcard-token -Optional PatternParser::try_to_consume_a_regexp_or_wildcard_token(Optional name_token) -{ - // 1. Let token be the result of running try to consume a token given parser and "regexp". - auto token = try_to_consume_a_token(Token::Type::Regexp); - - // 2. If name token is null and token is null, then set token to the result of running try to consume a token given - // parser and "asterisk". - if (!name_token.has_value() && !token.has_value()) - token = try_to_consume_a_token(Token::Type::Asterisk); - - // 3. Return token. - return token; -} - -// https://urlpattern.spec.whatwg.org/#try-to-consume-a-token -Optional PatternParser::try_to_consume_a_token(Token::Type type) -{ - // 1. Assert: parser’s index is less than parser’s token list size. - VERIFY(m_index < m_token_list.size()); - - // 2. Let next token be parser’s token list[parser’s index]. - auto const& next_token = m_token_list[m_index]; - - // 3. If next token’s type is not type return null. - if (next_token.type != type) - return {}; - - // 4. Increment parser’s index by 1. - ++m_index; - - // 5. Return next token. - return next_token; -} - -// https://urlpattern.spec.whatwg.org/#parse-a-pattern-string -PatternErrorOr> PatternParser::parse(Utf8View const& input, Options const& options, EncodingCallback encoding_callback) -{ - // 1. Let parser be a new pattern parser whose encoding callback is encoding callback and segment wildcard regexp - // is the result of running generate a segment wildcard regexp given options. - PatternParser parser { move(encoding_callback), generate_a_segment_wildcard_regexp(options) }; - - // 2. Set parser’s token list to the result of running tokenize given input and "strict". - parser.m_token_list = TRY(Tokenizer::tokenize(input, Tokenizer::Policy::Strict)); - - // 3. While parser’s index is less than parser’s token list's size: - while (parser.m_index < parser.m_token_list.size()) { - // 1. Let char token be the result of running try to consume a token given parser and "char". - auto char_token = parser.try_to_consume_a_token(Token::Type::Char); - - // 2. Let name token be the result of running try to consume a token given parser and "name". - auto name_token = parser.try_to_consume_a_token(Token::Type::Name); - - // 3. Let regexp or wildcard token be the result of running try to consume a regexp or wildcard token given - // parser and name token. - auto regexp_or_wildcard_token = parser.try_to_consume_a_regexp_or_wildcard_token(name_token); - - // 4. If name token is not null or regexp or wildcard token is not null: - // NOTE: If there is a matching group, we need to add the part immediately. - if (name_token.has_value() || regexp_or_wildcard_token.has_value()) { - // 1. Let prefix be the empty string. - String prefix; - - // 2. If char token is not null then set prefix to char token’s value. - if (char_token.has_value()) - prefix = char_token->value; - - // 3. If prefix is not the empty string and not options’s prefix code point: - if (!prefix.is_empty() && (!options.prefix_code_point.has_value() || prefix != String::from_code_point(*options.prefix_code_point))) { - // 1. Append prefix to the end of parser’s pending fixed value. - parser.m_pending_fixed_value.append(prefix); - - // 2. Set prefix to the empty string. - prefix = String {}; - } - - // 4. Run maybe add a part from the pending fixed value given parser. - TRY(parser.maybe_add_a_part_from_the_pending_fixed_value()); - - // 5. Let modifier token be the result of running try to consume a modifier token given parser. - auto modifier_token = parser.try_to_consume_a_modifier_token(); - - // 6. Run add a part given parser, prefix, name token, regexp or wildcard token, the empty string, - // and modifier token. - TRY(parser.add_a_part(prefix, name_token, regexp_or_wildcard_token, String {}, modifier_token)); - - // 7. Continue. - continue; - } - - // 5. Let fixed token be char token. - // NOTE: If there was no matching group, then we need to buffer any fixed text. We want to collect as - // much text as possible before adding it as a "fixed-text" part. - auto fixed_token = char_token; - - // 6. If fixed token is null, then set fixed token to the result of running try to consume a token given - // parser and "escaped-char". - if (!fixed_token.has_value()) - fixed_token = parser.try_to_consume_a_token(Token::Type::EscapedChar); - - // 7. If fixed token is not null: - if (fixed_token.has_value()) { - // 1. Append fixed token’s value to parser’s pending fixed value. - parser.m_pending_fixed_value.append(fixed_token->value); - - // 2. Continue. - continue; - } - - // 8. Let open token be the result of running try to consume a token given parser and "open". - auto open_token = parser.try_to_consume_a_token(Token::Type::Open); - - // 9. If open token is not null: - if (open_token.has_value()) { - // 1. Let prefix be the result of running consume text given parser. - auto prefix = parser.consume_text(); - - // 2. Set name token to the result of running try to consume a token given parser and "name". - name_token = parser.try_to_consume_a_token(Token::Type::Name); - - // 3. Set regexp or wildcard token to the result of running try to consume a regexp or wildcard token - // given parser and name token. - regexp_or_wildcard_token = parser.try_to_consume_a_regexp_or_wildcard_token(name_token); - - // 4. Let suffix be the result of running consume text given parser. - auto suffix = parser.consume_text(); - - // 5. Run consume a required token given parser and "close". - TRY(parser.consume_a_required_token(Token::Type::Close)); - - // 6. Let modifier token to the result of running try to consume a modifier token given parser. - auto modifier_token = parser.try_to_consume_a_modifier_token(); - - // 7. Run add a part given parser, prefix, name token, regexp or wildcard token, suffix, and modifier token. - TRY(parser.add_a_part(prefix, name_token, regexp_or_wildcard_token, suffix, modifier_token)); - - // 8. Continue. - continue; - } - - // 10. Run maybe add a part from the pending fixed value given parser. - TRY(parser.maybe_add_a_part_from_the_pending_fixed_value()); - - // 11. Run consume a required token given parser and "end". - TRY(parser.consume_a_required_token(Token::Type::End)); - } - - if constexpr (URL_PATTERN_DEBUG) { - dbgln("Pattern parser produced the part list:"); - for (auto const& part : parser.m_part_list) { - dbgln("Type {}, Value '{}', Modifier {}, Name '{}', Prefix '{}', Suffix '{}'", - Part::type_to_string(part.type), part.value, Part::convert_modifier_to_string(part.modifier), - part.name, part.prefix, part.suffix); - } - } - // 4. Return parser’s part list. - return move(parser.m_part_list); -} - -} diff --git a/Libraries/LibURL/Pattern/PatternParser.h b/Libraries/LibURL/Pattern/PatternParser.h deleted file mode 100644 index cebdca828b..0000000000 --- a/Libraries/LibURL/Pattern/PatternParser.h +++ /dev/null @@ -1,74 +0,0 @@ -/* - * Copyright (c) 2025, Shannon Booth - * - * SPDX-License-Identifier: BSD-2-Clause - */ - -#pragma once - -#include -#include -#include -#include -#include - -namespace URL::Pattern { - -// https://urlpattern.spec.whatwg.org/#pattern-parser -class PatternParser { -public: - // https://urlpattern.spec.whatwg.org/#encoding-callback - // An encoding callback is an abstract algorithm that takes a given string input. The input will be a simple text - // piece of a pattern string. An implementing algorithm will validate and encode the input. It must return the - // encoded string or throw an exception. - using EncodingCallback = Function(String const&)>; - - static PatternErrorOr> parse(Utf8View const& input, Options const&, EncodingCallback); - -private: - PatternParser(EncodingCallback, String segment_wildcard_regexp); - - Optional try_to_consume_a_token(Token::Type); - Optional try_to_consume_a_modifier_token(); - Optional try_to_consume_a_regexp_or_wildcard_token(Optional name_token); - PatternErrorOr consume_a_required_token(Token::Type); - String consume_text(); - PatternErrorOr maybe_add_a_part_from_the_pending_fixed_value(); - PatternErrorOr add_a_part( - String const& prefix, - Optional name_token, - Optional regexp_or_wildcard_token, - String const& suffix, - Optional modifier_token); - bool is_a_duplicate_name(String const&) const; - - // https://urlpattern.spec.whatwg.org/#pattern-parser-token-list - // A pattern parser has an associated token list, a token list, initially an empty list. - Vector m_token_list; - - // https://urlpattern.spec.whatwg.org/#pattern-parser-encoding-callback - // A pattern parser has an associated encoding callback, a encoding callback, that must be set upon creation. - EncodingCallback m_encoding_callback; - - // https://urlpattern.spec.whatwg.org/#pattern-parser-segment-wildcard-regexp - // A pattern parser has an associated segment wildcard regexp, a string, that must be set upon creation. - String m_segment_wildcard_regexp; - - // https://urlpattern.spec.whatwg.org/#pattern-parser-part-list - // A pattern parser has an associated part list, a part list, initially an empty list. - Vector m_part_list; - - // https://urlpattern.spec.whatwg.org/#pattern-parser-pending-fixed-value - // A pattern parser has an associated pending fixed value, a string, initially the empty string. - StringBuilder m_pending_fixed_value; - - // https://urlpattern.spec.whatwg.org/#pattern-parser-index - // A pattern parser has an associated index, a number, initially 0. - size_t m_index { 0 }; - - // https://urlpattern.spec.whatwg.org/#pattern-parser-next-numeric-name - // A pattern parser has an associated next numeric name, a number, initially 0. - size_t m_next_numeric_name { 0 }; -}; - -} diff --git a/Libraries/LibURL/Pattern/String.cpp b/Libraries/LibURL/Pattern/String.cpp deleted file mode 100644 index d22c288ce5..0000000000 --- a/Libraries/LibURL/Pattern/String.cpp +++ /dev/null @@ -1,310 +0,0 @@ -/* - * Copyright (c) 2025, Shannon Booth - * - * SPDX-License-Identifier: BSD-2-Clause - */ - -#include -#include -#include - -namespace URL::Pattern { - -// https://urlpattern.spec.whatwg.org/#escape-a-pattern-string -String escape_a_pattern_string(String const& input) -{ - // 1. Assert: input is an ASCII string. - VERIFY(input.is_ascii()); - - // 2. Let result be the empty string. - StringBuilder result; - - // 3. Let index be 0. - // 4. While index is less than input’s length: - for (auto c : input.bytes_as_string_view()) { - // 1. Let c be input[index]. - // 2. Increment index by 1. - - // 3. If c is one of: - // * U+002B (+); - // * U+002A (*); - // * U+003F (?); - // * U+003A (:); - // * U+007B ({); - // * U+007D (}); - // * U+0028 ((); - // * U+0029 ()); or - // * U+005C (\), - // then append U+005C (\) to the end of result. - if ("+*?:{}()\\"sv.contains(c)) - result.append('\\'); - - // 4. Append c to the end of result. - result.append(c); - } - - // 5. Return result. - return result.to_string_without_validation(); -} - -// https://urlpattern.spec.whatwg.org/#escape-a-regexp-string -String escape_a_regexp_string(String const& input) -{ - // 1. Assert: input is an ASCII string. - VERIFY(input.is_ascii()); - - // 2. Let result be the empty string. - StringBuilder builder; - - // 3. Let index be 0. - // 4. While index is less than input’s length: - for (auto c : input.bytes_as_string_view()) { - // 1. Let c be input[index]. - // 2. Increment index by 1. - - // 3. If c is one of: - // * U+002E (.); - // * U+002B (+); - // * U+002A (*); - // * U+003F (?); - // * U+005E (^); - // * U+0024 ($); - // * U+007B ({); - // * U+007D (}); - // * U+0028 ((); - // * U+0029 ()); - // * U+005B ([); - // * U+005D (]); - // * U+007C (|); - // * U+002F (/); or - // * U+005C (\), - // then append "\" to the end of result. - if (".+*?^${}()[]|/\\"sv.contains(c)) - builder.append('\\'); - - // 4. Append c to the end of result. - builder.append(c); - } - - // 5. Return result. - return builder.to_string_without_validation(); -} - -// https://urlpattern.spec.whatwg.org/#generate-a-segment-wildcard-regexp -String generate_a_segment_wildcard_regexp(Options const& options) -{ - // 1. Let result be "[^". - StringBuilder result; - result.append("[^"sv); - - // 2. Append the result of running escape a regexp string given options’s delimiter code point to the end of result. - if (options.delimiter_code_point.has_value()) - result.append(escape_a_regexp_string(String::from_code_point(*options.delimiter_code_point))); - - // 3. Append "]+?" to the end of result. - result.append("]+?"sv); - - // 4. Return result. - return result.to_string_without_validation(); -} - -// https://urlpattern.spec.whatwg.org/#generate-a-pattern-string -String generate_a_pattern_string(ReadonlySpan part_list, Options const& options) -{ - // 1. Let result be the empty string. - StringBuilder result; - - // 2. Let index list be the result of getting the indices for part list. - // 3. For each index of index list: - for (size_t index = 0; index < part_list.size(); ++index) { - // 1. Let part be part list[index]. - auto const& part = part_list[index]; - - // 2. Let previous part be part list[index - 1] if index is greater than 0, otherwise let it be null. - Part const* previous_part = index > 0 ? &part_list[index - 1] : nullptr; - - // 3. Let next part be part list[index + 1] if index is less than index list’s size - 1, otherwise let it be null. - Part const* next_part = index + 1 < part_list.size() ? &part_list[index + 1] : nullptr; - - // 4. If part’s type is "fixed-text" then: - if (part.type == Part::Type::FixedText) { - // 1. If part’s modifier is "none" then: - if (part.modifier == Part::Modifier::None) { - // 1. Append the result of running escape a pattern string given part’s value to the end of result. - result.append(escape_a_pattern_string(part.value)); - - // 2. Continue. - continue; - } - - // 2. Append "{" to the end of result. - result.append('{'); - - // 3. Append the result of running escape a pattern string given part’s value to the end of result. - result.append(escape_a_pattern_string(part.value)); - - // 4. Append "}" to the end of result. - result.append('}'); - - // 5. Append the result of running convert a modifier to a string given part’s modifier to the end of result. - result.append(Part::convert_modifier_to_string(part.modifier)); - - // 6. Continue. - continue; - } - - // 5. Let custom name be true if part’s name[0] is not an ASCII digit; otherwise false. - bool custom_name = !is_ascii_digit(part.name.bytes()[0]); - - // 6. Let needs grouping be true if at least one of the following are true, otherwise let it be false: - // * part’s suffix is not the empty string. - // * part’s prefix is not the empty string and is not options’s prefix code point. - bool needs_grouping = !part.suffix.is_empty() - || (!part.prefix.is_empty() && (options.prefix_code_point.has_value() && part.prefix != String::from_code_point(*options.prefix_code_point))); - - // 7. If all of the following are true: - // * needs grouping is false; and - // * custom name is true; and - // * part’s type is "segment-wildcard"; and - // * part’s modifier is "none"; and - // * next part is not null; and - // * next part’s prefix is the empty string; and - // * next part’s suffix is the empty string - // then: - if (!needs_grouping - && custom_name - && part.type == Part::Type::SegmentWildcard - && part.modifier == Part::Modifier::None - && next_part != nullptr - && next_part->prefix.is_empty() - && next_part->suffix.is_empty()) { - // 1. If next part’s type is "fixed-text": - if (next_part->type == Part::Type::FixedText) { - // 1. Set needs grouping to true if the result of running is a valid name code point given next part’s - // value's first code point and the boolean false is true. - // FIXME: Raise spec bug, the language here is weird. - needs_grouping = Tokenizer::is_a_valid_name_code_point(*next_part->value.code_points().begin(), false); - } - // 2. Otherwise: - else { - // 1. Set needs grouping to true if next part’s name[0] is an ASCII digit. - needs_grouping = is_ascii_digit(*next_part->name.code_points().begin()); - } - } - - // 8. If all of the following are true: - // * needs grouping is false; and - // * part’s prefix is the empty string; and - // * previous part is not null; and - // * previous part’s type is "fixed-text"; and - // * previous part’s value's last code point is options’s prefix code point. - // then set needs grouping to true. - if (!needs_grouping - && part.prefix.is_empty() - && previous_part != nullptr - && previous_part->type == Part::Type::FixedText - && ((previous_part->value.is_empty() && !options.prefix_code_point.has_value()) - || (options.prefix_code_point.has_value() && previous_part->value == String::from_code_point(*options.prefix_code_point)))) { - needs_grouping = true; - } - - // 9. Assert: part’s name is not the empty string or null. - VERIFY(!part.name.is_empty()); - - // 10. If needs grouping is true, then append "{" to the end of result. - if (needs_grouping) - result.append('{'); - - // 11. Append the result of running escape a pattern string given part’s prefix to the end of result. - result.append(escape_a_pattern_string(part.prefix)); - - // 12. If custom name is true: - if (custom_name) { - // 1. Append ":" to the end of result. - result.append(':'); - - // 2. Append part’s name to the end of result. - result.append(part.name); - } - - // 13. If part’s type is "regexp" then: - if (part.type == Part::Type::Regexp) { - // 1. Append "(" to the end of result. - result.append('('); - - // 2. Append part’s value to the end of result. - result.append(part.value); - - // 3. Append ")" to the end of result. - result.append(')'); - } - // 14. Otherwise if part’s type is "segment-wildcard" and custom name is false: - else if (part.type == Part::Type::SegmentWildcard && !custom_name) { - // 1. Append "(" to the end of result. - result.append('('); - - // 2. Append the result of running generate a segment wildcard regexp given options to the end of result. - result.append(generate_a_segment_wildcard_regexp(options)); - - // 3. Append ")" to the end of result. - result.append(')'); - } - // 15. Otherwise if part’s type is "full-wildcard": - else if (part.type == Part::Type::FullWildcard) { - // 1. If custom name is false and one of the following is true: - // * previous part is null; or - // * previous part’s type is "fixed-text"; or - // * previous part’s modifier is not "none"; or - // * needs grouping is true; or - // * part’s prefix is not the empty string - // then append "*" to the end of result. - if (!custom_name - && (previous_part == nullptr - || previous_part->type == Part::Type::FixedText - || previous_part->modifier != Part::Modifier::None - || needs_grouping - || !part.prefix.is_empty())) { - result.append('*'); - } - // 2. Otherwise: - else { - // 1. Append "(" to the end of result. - result.append('('); - - // 2. Append full wildcard regexp value to the end of result. - result.append(full_wildcard_regexp_value); - - // 3. Append ")" to the end of result. - result.append(')'); - } - } - - // 16. If all of the following are true: - // * part’s type is "segment-wildcard"; and - // * custom name is true; and - // * part’s suffix is not the empty string; and - // * The result of running is a valid name code point given part’s suffix's first code point and the boolean false is true - // then append U+005C (\) to the end of result. - if (part.type == Part::Type::SegmentWildcard - && custom_name - && !part.suffix.is_empty() - && Tokenizer::is_a_valid_name_code_point(*part.suffix.code_points().begin(), false)) { - result.append('\\'); - } - - // 17. Append the result of running escape a pattern string given part’s suffix to the end of result. - result.append(escape_a_pattern_string(part.suffix)); - - // 18. If needs grouping is true, then append "}" to the end of result. - if (needs_grouping) - result.append('}'); - - // 19. Append the result of running convert a modifier to a string given part’s modifier to the end of result. - result.append(Part::convert_modifier_to_string(part.modifier)); - } - - // 4. Return result. - return result.to_string_without_validation(); -} - -} diff --git a/Libraries/LibURL/Pattern/String.h b/Libraries/LibURL/Pattern/String.h deleted file mode 100644 index ea6ae80779..0000000000 --- a/Libraries/LibURL/Pattern/String.h +++ /dev/null @@ -1,23 +0,0 @@ -/* - * Copyright (c) 2025, Shannon Booth - * - * SPDX-License-Identifier: BSD-2-Clause - */ - -#pragma once - -#include -#include -#include - -namespace URL::Pattern { - -// https://urlpattern.spec.whatwg.org/#full-wildcard-regexp-value -static inline constexpr auto full_wildcard_regexp_value = ".*"sv; - -String escape_a_pattern_string(String const&); -String escape_a_regexp_string(String const&); -String generate_a_segment_wildcard_regexp(Options const&); -String generate_a_pattern_string(ReadonlySpan, Options const&); - -} diff --git a/Libraries/LibURL/Pattern/Tokenizer.cpp b/Libraries/LibURL/Pattern/Tokenizer.cpp deleted file mode 100644 index 2f3964de27..0000000000 --- a/Libraries/LibURL/Pattern/Tokenizer.cpp +++ /dev/null @@ -1,440 +0,0 @@ -/* - * Copyright (c) 2025, Shannon Booth - * - * SPDX-License-Identifier: BSD-2-Clause - */ - -#include -#include -#include - -namespace URL::Pattern { - -StringView Token::type_to_string(Token::Type type) -{ - switch (type) { - case Token::Type::Open: - return "Open"sv; - case Token::Type::Close: - return "Close"sv; - case Token::Type::Regexp: - return "Regexp"sv; - case Token::Type::Name: - return "Name"sv; - case Token::Type::Char: - return "Char"sv; - case Token::Type::EscapedChar: - return "EscapedChar"sv; - case Token::Type::OtherModifier: - return "OtherModifier"sv; - case Token::Type::Asterisk: - return "Asterisk"sv; - case Token::Type::End: - return "End"sv; - case Token::Type::InvalidChar: - return "InvalidChar"sv; - } - VERIFY_NOT_REACHED(); -} - -String Token::to_string() const -{ - return MUST(String::formatted("{}, index: {}, value: '{}'", type_to_string(type), index, value)); -} - -Tokenizer::Tokenizer(Utf8View const& input, Policy policy) - : m_input(input) - , m_policy(policy) -{ -} - -// https://urlpattern.spec.whatwg.org/#tokenize -PatternErrorOr> Tokenizer::tokenize(Utf8View const& input, Tokenizer::Policy policy) -{ - dbgln_if(URL_PATTERN_DEBUG, "URLPattern tokenizing input: '{}'", input.as_string()); - VERIFY(input.validate()); - - // 1. Let tokenizer be a new tokenizer. - // 2. Set tokenizer’s input to input. - // 3. Set tokenizer’s policy to policy. - Tokenizer tokenizer { input, policy }; - - // 4. While tokenizer’s index is less than tokenizer’s input's code point length: - while (tokenizer.m_index < tokenizer.m_input.length()) { - // 1. Run seek and get the next code point given tokenizer and tokenizer’s index. - tokenizer.seek_and_get_the_next_code_point(tokenizer.m_index); - - // 2. If tokenizer’s code point is U+002A (*): - if (tokenizer.m_code_point == '*') { - // 1. Run add a token with default position and length given tokenizer and "asterisk". - tokenizer.add_a_token_with_default_position_and_length(Token::Type::Asterisk); - - // 2. Continue. - continue; - } - - // 3. If tokenizer’s code point is U+002B (+) or U+003F (?): - if (tokenizer.m_code_point == '+' || tokenizer.m_code_point == '?') { - // 1. Run add a token with default position and length given tokenizer and "other-modifier". - tokenizer.add_a_token_with_default_position_and_length(Token::Type::OtherModifier); - - // 2. Continue. - continue; - } - - // 4. If tokenizer’s code point is U+005C (\): - if (tokenizer.m_code_point == '\\') { - // 1. If tokenizer’s index is equal to tokenizer’s input's code point length − 1: - if (tokenizer.m_index == tokenizer.m_input.length() - 1) { - // 1. Run process a tokenizing error given tokenizer, tokenizer’s next index, and tokenizer’s index. - TRY(tokenizer.process_a_tokenizing_error(tokenizer.m_next_index, tokenizer.m_index)); - - // 2. Continue. - continue; - } - - // 2. Let escaped index be tokenizer’s next index. - auto escaped_index = tokenizer.m_next_index; - - // 3. Run get the next code point given tokenizer. - tokenizer.get_the_next_code_point(); - - // 4. Run add a token with default length given tokenizer, "escaped-char", tokenizer’s next index, and escaped index. - tokenizer.add_a_token_with_default_length(Token::Type::EscapedChar, tokenizer.m_next_index, escaped_index); - - // 5. Continue. - continue; - } - - // 5. If tokenizer’s code point is U+007B ({): - if (tokenizer.m_code_point == '{') { - // 1. Run add a token with default position and length given tokenizer and "open". - tokenizer.add_a_token_with_default_position_and_length(Token::Type::Open); - - // 2. Continue. - continue; - } - - // 6. If tokenizer’s code point is U+007D (}): - if (tokenizer.m_code_point == '}') { - // 1. Run add a token with default position and length given tokenizer and "close". - tokenizer.add_a_token_with_default_position_and_length(Token::Type::Close); - - // 2. Continue. - continue; - } - - // 1. If tokenizer’s code point is U+003A (:): - if (tokenizer.m_code_point == ':') { - // 1. Let name position be tokenizer’s next index. - auto name_position = tokenizer.m_next_index; - - // 2. Let name start be name position. - auto name_start = name_position; - - // 3. While name position is less than tokenizer’s input's code point length: - while (name_position < tokenizer.m_input.length()) { - // 1. Run seek and get the next code point given tokenizer and name position. - tokenizer.seek_and_get_the_next_code_point(name_position); - - // 2. Let first code point be true if name position equals name start and false otherwise. - bool first_code_point = name_position == name_start; - - // 3. Let valid code point be the result of running is a valid name code point given tokenizer’s code point and first code point. - bool valid_code_point = is_a_valid_name_code_point(tokenizer.m_code_point, first_code_point); - - // 4. If valid code point is false break. - if (!valid_code_point) - break; - - // 5. Set name position to tokenizer’s next index. - name_position = tokenizer.m_next_index; - } - - // 4. If name position is less than or equal to name start: - if (name_position <= name_start) { - // 1. Run process a tokenizing error given tokenizer, name start, and tokenizer’s index. - TRY(tokenizer.process_a_tokenizing_error(name_start, tokenizer.m_index)); - - // 2. Continue. - continue; - } - - // 5. Run add a token with default length given tokenizer, "name", name position, and name start. - tokenizer.add_a_token_with_default_length(Token::Type::Name, name_position, name_start); - - // 6. Continue. - continue; - } - - // 8. If tokenizer’s code point is U+0028 ((): - if (tokenizer.m_code_point == '(') { - // 1. Let depth be 1. - u32 depth = 1; - - // 2. Let regexp position be tokenizer’s next index. - auto regexp_position = tokenizer.m_next_index; - - // 3. Let regexp start be regexp position. - auto regexp_start = regexp_position; - - // 4. Let error be false. - bool error = false; - - // 5. While regexp position is less than tokenizer’s input's code point length: - while (regexp_position < tokenizer.m_input.length()) { - // 1. Run seek and get the next code point given tokenizer and regexp position. - tokenizer.seek_and_get_the_next_code_point(regexp_position); - - // 2. If the result of running is ASCII given tokenizer’s code point is false: - if (!is_ascii(tokenizer.m_code_point)) { - // 1. Run process a tokenizing error given tokenizer, regexp start, and tokenizer’s index. - TRY(tokenizer.process_a_tokenizing_error(regexp_start, tokenizer.m_index)); - - // 2. Set error to true. - error = true; - - // 3. Break. - break; - } - - // 3. If regexp position equals regexp start and tokenizer’s code point is U+003F (?): - if (regexp_position == regexp_start && tokenizer.m_code_point == '?') { - // 1. Run process a tokenizing error given tokenizer, regexp start, and tokenizer’s index. - TRY(tokenizer.process_a_tokenizing_error(regexp_start, tokenizer.m_index)); - - // 2. Set error to true. - error = true; - - // 3. Break. - break; - } - - // 4. If tokenizer’s code point is U+005C (\): - if (tokenizer.m_code_point == '\\') { - // 1. If regexp position equals tokenizer’s input's code point length − 1: - if (regexp_position == tokenizer.m_input.length() - 1) { - // 1. Run process a tokenizing error given tokenizer, regexp start, and tokenizer’s index. - TRY(tokenizer.process_a_tokenizing_error(regexp_start, tokenizer.m_index)); - - // 2. Set error to true. - error = true; - - // 3. Break - break; - } - - // 2. Run get the next code point given tokenizer. - tokenizer.get_the_next_code_point(); - - // 3. If the result of running is ASCII given tokenizer’s code point is false: - if (!is_ascii(tokenizer.m_code_point)) { - // 1. Run process a tokenizing error given tokenizer, regexp start, and tokenizer’s index. - TRY(tokenizer.process_a_tokenizing_error(regexp_start, tokenizer.m_index)); - - // 2. Set error to true. - error = true; - - // 3. Break. - break; - } - - // 4. Set regexp position to tokenizer’s next index. - regexp_position = tokenizer.m_next_index; - - // 5. Continue. - continue; - } - - // 5. If tokenizer’s code point is U+0029 ()): - if (tokenizer.m_code_point == ')') { - // 1. Decrement depth by 1. - --depth; - - // 1. If depth is 0: - if (depth == 0) { - // 1. Set regexp position to tokenizer’s next index. - regexp_position = tokenizer.m_next_index; - - // 2. Break. - break; - } - } - // 6. Otherwise if tokenizer’s code point is U+0028 ((): - else if (tokenizer.m_code_point == '(') { - // 1. Increment depth by 1. - ++depth; - - // 2. If regexp position equals tokenizer’s input's code point length − 1: - if (regexp_position == tokenizer.m_input.length() - 1) { - // 1. Run process a tokenizing error given tokenizer, regexp start, and tokenizer’s index. - TRY(tokenizer.process_a_tokenizing_error(regexp_start, tokenizer.m_index)); - - // 2. Set error to true. - error = true; - - // 3. Break - break; - } - - // 3. Let temporary position be tokenizer’s next index. - auto temporary_position = tokenizer.m_next_index; - - // 4. Run get the next code point given tokenizer. - tokenizer.get_the_next_code_point(); - - // 5. If tokenizer’s code point is not U+003F (?): - if (tokenizer.m_code_point != '?') { - // 1. Run process a tokenizing error given tokenizer, regexp start, and tokenizer’s index. - TRY(tokenizer.process_a_tokenizing_error(regexp_start, tokenizer.m_index)); - - // 2. Set error to true. - error = true; - - // 3. Break. - break; - } - - // 6. Set tokenizer’s next index to temporary position. - tokenizer.m_next_index = temporary_position; - } - - // 7. Set regexp position to tokenizer’s next index. - regexp_position = tokenizer.m_next_index; - } - - // 6. If error is true continue. - if (error) - continue; - - // 7. If depth is not zero: - if (depth != 0) { - // 1. Run process a tokenizing error given tokenizer, regexp start, and tokenizer’s index. - TRY(tokenizer.process_a_tokenizing_error(regexp_start, tokenizer.m_index)); - - // 2. Continue. - continue; - } - - // 8. Let regexp length be regexp position − regexp start − 1. - auto regexp_length = regexp_position - regexp_start - 1; - - // 9. If regexp length is zero: - if (regexp_length == 0) { - // 1. Run process a tokenizing error given tokenizer, regexp start, and tokenizer’s index. - TRY(tokenizer.process_a_tokenizing_error(regexp_start, tokenizer.m_index)); - - // 2. Continue. - continue; - } - - // 10. Run add a token given tokenizer, "regexp", regexp position, regexp start, and regexp length. - tokenizer.add_a_token(Token::Type::Regexp, regexp_position, regexp_start, regexp_length); - - // 11. Continue. - continue; - } - - // 9. Run add a token with default position and length given tokenizer and "char". - tokenizer.add_a_token_with_default_position_and_length(Token::Type::Char); - } - - // 5. Run add a token with default length given tokenizer, "end", tokenizer’s index, and tokenizer’s index. - tokenizer.add_a_token_with_default_length(Token::Type::End, tokenizer.m_index, tokenizer.m_index); - - // 6. Return tokenizer’s token list. - if constexpr (URL_PATTERN_DEBUG) { - for (auto const& token : tokenizer.m_token_list) - dbgln("{}", token.to_string()); - } - - return tokenizer.m_token_list; -} - -// https://urlpattern.spec.whatwg.org/#get-the-next-code-point -void Tokenizer::get_the_next_code_point() -{ - // 1. Set tokenizer’s code point to the Unicode code point in tokenizer’s input at the position indicated by tokenizer’s next index. - m_code_point = *m_input.unicode_substring_view(m_next_index, 1).begin(); - - // 2. Increment tokenizer’s next index by 1. - ++m_next_index; -} - -// https://urlpattern.spec.whatwg.org/#seek-and-get-the-next-code-point -void Tokenizer::seek_and_get_the_next_code_point(u32 index) -{ - // 1. Set tokenizer’s next index to index. - m_next_index = index; - - // 2. Run get the next code point given tokenizer. - get_the_next_code_point(); -} - -// https://urlpattern.spec.whatwg.org/#add-a-token -void Tokenizer::add_a_token(Token::Type type, u32 next_position, u32 value_position, u32 value_length) -{ - // 1. Let token be a new token. - Token token; - - // 2. Set token’s type to type. - token.type = type; - - // 3. Set token’s index to tokenizer’s index. - token.index = m_index; - - // 4. Set token’s value to the code point substring from value position with length value length within tokenizer’s input. - token.value = MUST(String::from_utf8(m_input.unicode_substring_view(value_position, value_length).as_string())); - - // 5. Append token to the back of tokenizer’s token list. - m_token_list.append(move(token)); - - // 5. Set tokenizer’s index to next position. - m_index = next_position; -} - -// https://urlpattern.spec.whatwg.org/#add-a-token-with-default-length -void Tokenizer::add_a_token_with_default_length(Token::Type type, u32 next_position, u32 value_position) -{ - // 1. Let computed length be next position − value position. - auto computed_length = next_position - value_position; - - // 2. Run add a token given tokenizer, type, next position, value position, and computed length. - add_a_token(type, next_position, value_position, computed_length); -} - -// https://urlpattern.spec.whatwg.org/#add-a-token-with-default-position-and-length -void Tokenizer::add_a_token_with_default_position_and_length(Token::Type type) -{ - // 1. Run add a token with default length given tokenizer, type, tokenizer’s next index, and tokenizer’s index. - add_a_token_with_default_length(type, m_next_index, m_index); -} - -// https://urlpattern.spec.whatwg.org/#process-a-tokenizing-error -PatternErrorOr Tokenizer::process_a_tokenizing_error(u32 next_position, u32 value_position) -{ - // 1. If tokenizer’s policy is "strict", then throw a TypeError. - if (m_policy == Policy::Strict) - return ErrorInfo { "Error processing a token"_string }; // FIXME: Improve this error! - - // 2. Assert: tokenizer’s policy is "lenient". - VERIFY(m_policy == Policy::Lenient); - - // 3. Run add a token with default length given tokenizer, "invalid-char", next position, and value position. - add_a_token_with_default_length(Token::Type::InvalidChar, next_position, value_position); - - return {}; -} - -// https://urlpattern.spec.whatwg.org/#is-a-valid-name-code-point -bool Tokenizer::is_a_valid_name_code_point(u32 code_point, bool first) -{ - // 1. If first is true return the result of checking if code point is contained in the IdentifierStart set of code points. - if (first) - return code_point == '$' || code_point == '_' || Unicode::code_point_has_identifier_start_property(code_point); - - // 2. Otherwise return the result of checking if code point is contained in the IdentifierPart set of code points. - return code_point == '$' || Unicode::code_point_has_identifier_continue_property(code_point); -} - -} diff --git a/Libraries/LibURL/Pattern/Tokenizer.h b/Libraries/LibURL/Pattern/Tokenizer.h deleted file mode 100644 index 31500694bb..0000000000 --- a/Libraries/LibURL/Pattern/Tokenizer.h +++ /dev/null @@ -1,117 +0,0 @@ -/* - * Copyright (c) 2025, Shannon Booth - * - * SPDX-License-Identifier: BSD-2-Clause - */ - -#pragma once - -#include -#include - -namespace URL::Pattern { - -// https://urlpattern.spec.whatwg.org/#token -// A token is a struct representing a single lexical token within a pattern string. -struct Token { - // https://urlpattern.spec.whatwg.org/#token-type - enum class Type { - // The token represents a U+007B ({) code point. - Open, - - // The token represents a U+007D (}) code point. - Close, - - // The token represents a string of the form "()". The regular expression is required to consist of only ASCII code points. - Regexp, - - // The token represents a string of the form ":". The name value is restricted to code points that are consistent with JavaScript identifiers. - Name, - - // The token represents a valid pattern code point without any special syntactical meaning. - Char, - - // The token represents a code point escaped using a backslash like "\". - EscapedChar, - - // The token represents a matching group modifier that is either the U+003F (?) or U+002B (+) code points. - OtherModifier, - - // The token represents a U+002A (*) code point that can be either a wildcard matching group or a matching group modifier. - Asterisk, - - // The token represents the end of the pattern string. - End, - - // The token represents a code point that is invalid in the pattern. This could be because of the code point value - // itself or due to its location within the pattern relative to other syntactic elements. - InvalidChar, - }; - - // https://urlpattern.spec.whatwg.org/#token-type - // A token has an associated type, a string, initially "invalid-char". - Type type { Type::InvalidChar }; - - // https://urlpattern.spec.whatwg.org/#token-index - // A token has an associated index, a number, initially 0. It is the position of the first code point in the pattern string represented by the token. - u32 index { 0 }; - - // https://urlpattern.spec.whatwg.org/#token-value - // A token has an associated value, a string, initially the empty string. It contains the code points from the pattern string represented by the token. - String value; - - String to_string() const; - static StringView type_to_string(Token::Type); -}; - -// https://urlpattern.spec.whatwg.org/#tokenizer -// A tokenizer is a struct. -class Tokenizer { -public: - // https://urlpattern.spec.whatwg.org/#tokenize-policy - // A tokenize policy is a string that must be either "strict" or "lenient". - enum class Policy { - Strict, - Lenient, - }; - - static PatternErrorOr> tokenize(Utf8View const&, Policy); - - static bool is_a_valid_name_code_point(u32 code_point, bool first); - -private: - Tokenizer(Utf8View const& input, Policy); - - void get_the_next_code_point(); - void seek_and_get_the_next_code_point(u32 index); - void add_a_token(Token::Type, u32 next_position, u32 value_position, u32 value_length); - void add_a_token_with_default_length(Token::Type, u32 next_position, u32 value_position); - void add_a_token_with_default_position_and_length(Token::Type); - PatternErrorOr process_a_tokenizing_error(u32 next_position, u32 value_position); - - // https://urlpattern.spec.whatwg.org/#tokenizer-input - // A tokenizer has an associated input, a pattern string, initially the empty string. - Utf8View m_input; - - // https://urlpattern.spec.whatwg.org/#tokenizer-policy - // A tokenizer has an associated policy, a tokenize policy, initially "strict". - Policy m_policy { Policy::Strict }; - - // https://urlpattern.spec.whatwg.org/#tokenizer-token-list - // A tokenizer has an associated token list, a token list, initially an empty list. - Vector m_token_list; - - // https://urlpattern.spec.whatwg.org/#tokenizer-index - // A tokenizer has an associated index, a number, initially 0. - size_t m_index { 0 }; - - // https://urlpattern.spec.whatwg.org/#tokenizer-next-index - // A tokenizer has an associated next index, a number, initially 0. - size_t m_next_index { 0 }; - - // https://urlpattern.spec.whatwg.org/#tokenizer-code-point - // A tokenizer has an associated code point, a Unicode code point, initially null. - u32 m_code_point {}; -}; - -} diff --git a/Tests/LibURL/CMakeLists.txt b/Tests/LibURL/CMakeLists.txt index f3960ee5c8..b7711936e0 100644 --- a/Tests/LibURL/CMakeLists.txt +++ b/Tests/LibURL/CMakeLists.txt @@ -1,7 +1,6 @@ set(URL_TEST_SOURCES TestURL.cpp TestURLPattern.cpp - TestURLPatternConstructorStringParser.cpp TestPublicSuffix.cpp ) diff --git a/Tests/LibURL/TestURLPatternConstructorStringParser.cpp b/Tests/LibURL/TestURLPatternConstructorStringParser.cpp deleted file mode 100644 index 60e0b1e14d..0000000000 --- a/Tests/LibURL/TestURLPatternConstructorStringParser.cpp +++ /dev/null @@ -1,174 +0,0 @@ -/* - * Copyright (c) 2025, Shannon Booth - * - * SPDX-License-Identifier: BSD-2-Clause - */ - -#include - -#include - -TEST_CASE(basic_http_url_no_pattern_or_path) -{ - auto input = "http://www.serenityos.org"_string; - auto result = MUST(URL::Pattern::ConstructorStringParser::parse(input.code_points())); - EXPECT_EQ(result.protocol, "http"sv); - EXPECT_EQ(result.username, OptionalNone {}); - EXPECT_EQ(result.password, OptionalNone {}); - EXPECT_EQ(result.hostname, "www.serenityos.org"sv); - EXPECT_EQ(result.port, ""sv); - EXPECT_EQ(result.pathname, OptionalNone {}); - EXPECT_EQ(result.search, OptionalNone {}); - EXPECT_EQ(result.hash, OptionalNone {}); - EXPECT_EQ(result.base_url, OptionalNone {}); -} - -TEST_CASE(pathname_with_regexp) -{ - auto input = "/books/(\\d+)"_string; - auto result = MUST(URL::Pattern::ConstructorStringParser::parse(input.code_points())); - EXPECT_EQ(result.protocol, OptionalNone {}); - EXPECT_EQ(result.username, OptionalNone {}); - EXPECT_EQ(result.password, OptionalNone {}); - EXPECT_EQ(result.hostname, OptionalNone {}); - EXPECT_EQ(result.port, OptionalNone {}); - EXPECT_EQ(result.pathname, "/books/(\\d+)"sv); - EXPECT_EQ(result.search, OptionalNone {}); - EXPECT_EQ(result.hash, OptionalNone {}); - EXPECT_EQ(result.base_url, OptionalNone {}); -} - -TEST_CASE(url_with_pathname_and_regexp) -{ - auto input = "https://example.com/2022/feb/*"_string; - auto result = MUST(URL::Pattern::ConstructorStringParser::parse(input.code_points())); - EXPECT_EQ(result.protocol, "https"sv); - EXPECT_EQ(result.username, OptionalNone {}); - EXPECT_EQ(result.password, OptionalNone {}); - EXPECT_EQ(result.hostname, "example.com"sv); - EXPECT_EQ(result.port, ""sv); - EXPECT_EQ(result.pathname, "/2022/feb/*"sv); - EXPECT_EQ(result.search, OptionalNone {}); - EXPECT_EQ(result.hash, OptionalNone {}); - EXPECT_EQ(result.base_url, OptionalNone {}); -} - -TEST_CASE(http_url_regexp_in_pathname_and_hostname) -{ - auto input = "https://cdn-*.example.com/*.jpg"_string; - auto result = MUST(URL::Pattern::ConstructorStringParser::parse(input.code_points())); - EXPECT_EQ(result.protocol, "https"sv); - EXPECT_EQ(result.username, OptionalNone {}); - EXPECT_EQ(result.password, OptionalNone {}); - EXPECT_EQ(result.hostname, "cdn-*.example.com"sv); - EXPECT_EQ(result.port, ""sv); - EXPECT_EQ(result.pathname, "/*.jpg"sv); - EXPECT_EQ(result.search, OptionalNone {}); - EXPECT_EQ(result.hash, OptionalNone {}); - EXPECT_EQ(result.base_url, OptionalNone {}); -} - -TEST_CASE(https_url_with_fragment) -{ - auto input = "https://example.com/#foo"_string; - auto result = MUST(URL::Pattern::ConstructorStringParser::parse(input.code_points())); - EXPECT_EQ(result.protocol, "https"sv); - EXPECT_EQ(result.username, OptionalNone {}); - EXPECT_EQ(result.password, OptionalNone {}); - EXPECT_EQ(result.hostname, "example.com"sv); - EXPECT_EQ(result.port, ""sv); - EXPECT_EQ(result.pathname, "/"sv); - EXPECT_EQ(result.search, ""sv); - EXPECT_EQ(result.hash, "foo"sv); - EXPECT_EQ(result.base_url, OptionalNone {}); -} - -TEST_CASE(http_url_with_query) -{ - auto input = "https://example.com/?q=*&v=?&hmm={}&umm=()"_string; - auto result = MUST(URL::Pattern::ConstructorStringParser::parse(input.code_points())); - EXPECT_EQ(result.protocol, "https"sv); - EXPECT_EQ(result.username, OptionalNone {}); - EXPECT_EQ(result.password, OptionalNone {}); - EXPECT_EQ(result.hostname, "example.com"sv); - EXPECT_EQ(result.port, ""sv); - EXPECT_EQ(result.pathname, "/"sv); - EXPECT_EQ(result.search, "q=*&v=?&hmm={}&umm=()"sv); - EXPECT_EQ(result.hash, OptionalNone {}); - EXPECT_EQ(result.base_url, OptionalNone {}); -} - -TEST_CASE(matches_on_sub_url) -{ - auto input = "https://{sub.}?example.com/foo"_string; - auto result = MUST(URL::Pattern::ConstructorStringParser::parse(input.code_points())); - EXPECT_EQ(result.protocol, "https"sv); - EXPECT_EQ(result.username, OptionalNone {}); - EXPECT_EQ(result.password, OptionalNone {}); - EXPECT_EQ(result.hostname, "{sub.}?example.com"sv); - EXPECT_EQ(result.port, ""sv); - EXPECT_EQ(result.pathname, "/foo"sv); - EXPECT_EQ(result.search, OptionalNone {}); - EXPECT_EQ(result.hash, OptionalNone {}); - EXPECT_EQ(result.base_url, OptionalNone {}); -} - -TEST_CASE(ipv6_with_port_number) -{ - auto input = "http://[\\:\\:1]:8080"_string; - auto result = MUST(URL::Pattern::ConstructorStringParser::parse(input.code_points())); - EXPECT_EQ(result.protocol, "http"sv); - EXPECT_EQ(result.username, OptionalNone {}); - EXPECT_EQ(result.password, OptionalNone {}); - EXPECT_EQ(result.hostname, "[\\:\\:1]"sv); - EXPECT_EQ(result.port, "8080"sv); - EXPECT_EQ(result.pathname, OptionalNone {}); - EXPECT_EQ(result.search, OptionalNone {}); - EXPECT_EQ(result.hash, OptionalNone {}); - EXPECT_EQ(result.base_url, OptionalNone {}); -} - -TEST_CASE(data_url) -{ - auto input = "data\\:foobar"_string; - auto result = MUST(URL::Pattern::ConstructorStringParser::parse(input.code_points())); - EXPECT_EQ(result.protocol, "data"sv); - EXPECT_EQ(result.username, OptionalNone {}); - EXPECT_EQ(result.password, OptionalNone {}); - EXPECT_EQ(result.hostname, ""sv); - EXPECT_EQ(result.port, ""sv); - EXPECT_EQ(result.pathname, "foobar"sv); - EXPECT_EQ(result.search, OptionalNone {}); - EXPECT_EQ(result.hash, OptionalNone {}); - EXPECT_EQ(result.base_url, OptionalNone {}); -} - -TEST_CASE(non_special_scheme_and_arbitrary_hostname) -{ - auto input = "foo://bar"_string; - auto result = MUST(URL::Pattern::ConstructorStringParser::parse(input.code_points())); - EXPECT_EQ(result.protocol, "foo"sv); - EXPECT_EQ(result.username, OptionalNone {}); - EXPECT_EQ(result.password, OptionalNone {}); - EXPECT_EQ(result.hostname, "bar"sv); - EXPECT_EQ(result.port, ""sv); - EXPECT_EQ(result.pathname, OptionalNone {}); - EXPECT_EQ(result.search, OptionalNone {}); - EXPECT_EQ(result.hash, OptionalNone {}); - EXPECT_EQ(result.base_url, OptionalNone {}); -} - -TEST_CASE(ipv6_with_named_group) -{ - auto input = "http://[:address]/"_string; - auto result = MUST(URL::Pattern::ConstructorStringParser::parse(input.code_points())); - EXPECT_EQ(result.protocol, "http"sv); - EXPECT_EQ(result.username, OptionalNone {}); - EXPECT_EQ(result.password, OptionalNone {}); - EXPECT_EQ(result.hostname, "[:address]"sv); - EXPECT_EQ(result.port, ""sv); - EXPECT_EQ(result.pathname, "/"sv); - EXPECT_EQ(result.search, OptionalNone {}); - EXPECT_EQ(result.hash, OptionalNone {}); - EXPECT_EQ(result.base_url, OptionalNone {}); -}