Commit 12b8c58be41 for php
commit 12b8c58be412278680a228e65a7732d00075a947
Merge: fa42fde968c beee5444d01
Author: Alexandre Daubois <alex.daubois@gmail.com>
Date: Tue Oct 6 09:01:23 2026 +0200
Merge branch 'PHP-8.5' into PHP-8.6
* PHP-8.5:
lexbor: Merge upstream WHATWG URL and IDNA fixes
diff --cc NEWS
index ddc73c19be3,a42642d601e..8b30e7f4d3e
--- a/NEWS
+++ b/NEWS
@@@ -62,6 -83,30 +62,17 @@@ PH
treating the offset as UTF-16 code units instead of bytes.
(Ilia Alshanetsky)
+ - Lexbor:
- . Merge patches lexbor/lexbor@8a14bc0 and lexbor/lexbor@f67ce4b, fixing a
- heap buffer overflow in :lexbor-contains() parsing and buffer overflows
- in malformed decode replay. (alexandre-daubois)
+ . Merge patches lexbor/lexbor@859f100, lexbor/lexbor@a36e09a,
+ lexbor/lexbor@b0f7412, lexbor/lexbor@1b215a8, lexbor/lexbor@385afff,
+ lexbor/lexbor@e6c068f, lexbor/lexbor@327a8b6, lexbor/lexbor@917742f and
+ lexbor/lexbor@e89c258, fixing dropped usernames containing an at sign,
+ uninitialized memory in IDNA buffer growth, the encoding of a space
+ before a query or fragment in an opaque path, a query or fragment lost
+ after a dot segment in a path, replacement file drive paths, the output
+ encoding used for percent-encoding, the URLSearchParams tail pointer and
+ fragment serialization without a query. (alexandre-daubois)
+
-- MBString:
- . Fixed bug GH-23106 (mb_strpos() reads past the end of a haystack ending in
- a truncated UTF-8 sequence). (Lazizbek Ergashev)
- . Fixed mbstring functions emitting surrogates in UTF-8 output and flagging
- it as valid UTF-8. (Nicolas Grekas)
-
-- MySQLi:
- . Fix GH-22854: Fixed failed assertion when accessing mysqli property after
- failed reconnection. (Kamil Tekiela)
-
- MySQLnd:
. Fixed field_count not resetting on OK packet. (Kamil Tekiela)
. Fixed memory leak when closing a prepared statement after its connection
diff --cc ext/lexbor/lexbor/url/url.c
index 69d91969a6a,146f0bda292..c88f8e7a7ad
--- a/ext/lexbor/lexbor/url/url.c
+++ b/ext/lexbor/lexbor/url/url.c
@@@ -1739,16 -1778,13 +1764,14 @@@ again
break;
}
- if (pswd == NULL || !at_sign) {
- tmp = (pswd != NULL) ? pswd - 1 : p;
-
- if (tmp > begin) {
- status = lxb_url_percent_encode_after_utf_8(begin,
- tmp, &url->username, url->mraw, lxb_url_map,
- LXB_URL_MAP_USERINFO, false);
- if (status != LXB_STATUS_OK) {
- lxb_url_parse_return(orig_data, buf, status);
- }
+ tmp = (pswd != NULL) ? pswd - 1 : p;
+ if (tmp > begin) {
+ status = lxb_url_percent_encode_after_utf_8(begin, tmp,
+ &url->username, url->mraw,
- LXB_URL_MAP_USERINFO, false);
++ lxb_url_map, LXB_URL_MAP_USERINFO,
++ false);
+ if (status != LXB_STATUS_OK) {
+ lxb_url_parse_return(orig_data, buf, status);
}
}
@@@ -3180,10 -3236,9 +3237,10 @@@ lxb_url_percent_encode_after_encoding(c
const lxb_char_t *buf_end = buf + sizeof(buffer);
static const lexbor_str_t esc_str = lexbor_str("%26%23");
- if (encoding->encoding == LXB_ENCODING_UTF_8) {
+ if (lxb_url_output_encoding(encoding->encoding) == LXB_ENCODING_UTF_8) {
return lxb_url_percent_encode_after_utf_8(data, end, str, mraw,
- enmap, space_as_plus);
+ url_map, enmap,
+ space_as_plus);
}
lxb_url_encoding_init(encoding, &encode);
diff --cc ext/lexbor/lexbor/url/url.h
index d2c93080c92,b9e4973a674..5ecc1da5dfb
--- a/ext/lexbor/lexbor/url/url.h
+++ b/ext/lexbor/lexbor/url/url.h
@@@ -335,114 -321,6 +335,118 @@@ lxb_url_parse_basic(lxb_url_parser_t *p
const lxb_char_t *data, size_t length,
lxb_url_state_t override_state, lxb_encoding_t encoding);
+/*
+ * IPv6 parser.
+ *
+ * This function is an implementation of IPv6 parsing according to the WHATWG
+ * specification.
+ * https://url.spec.whatwg.org/#concept-ipv6-parser
+ *
+ * The address can be passed both with and without the surrounding square
+ * brackets: "::1" and "[::1]" give the same result. If the opening bracket is
+ * present, the closing one is required.
+ *
+ * The output buffer is zeroed by the function, there is no need to prepare it.
+ * Use the lxb_url_serialize_host_ipv6() function to serialize the result.
+ *
+ * @param[in] lxb_url_parser_t *. Can be NULL.
+ * @param[in] Pointer to the beginning of the data. Not NULL.
+ * @param[in] Length of the data. Can be 0.
+ * @param[out] Buffer for eight (uint16_t[8]) IPv6 pieces. Not NULL. The value
+ * is meaningful only if LXB_STATUS_OK is returned.
+ *
+ * @return LXB_STATUS_OK if successful, otherwise an error status value.
+ */
+LXB_API lxb_status_t
+lxb_url_parse_host_ipv6(lxb_url_parser_t *parser, const lxb_char_t *data,
+ size_t length, uint16_t *ipv6);
+
+/*
+ * UTF-8 percent-encoder.
+ *
+ * Percent-encodes bytes from data according to url_map and appends the result
+ * to str. A byte is encoded as "%HH" when the result of
+ * (url_map[byte] & enmap) is non-zero. Uppercase hexadecimal digits are used.
+ * If space_as_plus is true, U+0020 SPACE is encoded as '+' before the map is
+ * checked.
+ *
+ * The input is expected to be valid UTF-8; the function does not validate it.
+ *
+ * @param[in] Pointer to UTF-8 data. Not NULL.
+ * @param[in] Length of data. Can be 0.
+ * @param[in, out] Output string. Can be uninitialized (data = NULL). Encoded
+ * data is appended to any existing content. Not NULL.
+ * @param[in] Memory object used to allocate or resize the output string. Not
+ * NULL.
+ * @param[in] Table of 256 entries indexed by input byte, each entry is a bit
+ * mask of lxb_url_map_type_t values. Not NULL.
+ * @param[in] Mask selecting the bytes to percent-encode.
+ * @param[in] Replace U+0020 SPACE with '+' if true.
+ *
+ * @return LXB_STATUS_OK if successful, otherwise an error status value.
+ */
+LXB_API lxb_status_t
+lxb_url_percent_encode_utf_8(const lxb_char_t *data, size_t length,
+ lexbor_str_t *str, lexbor_mraw_t *mraw,
+ const uint8_t *url_map, lxb_url_map_type_t enmap,
+ bool space_as_plus);
+
+/*
+ * Percent-encode after encoding.
+ *
+ * Converts valid UTF-8 data to the specified encoding and appends the
+ * percent-encoded result to str. Each encoded byte for which
+ * (url_map[byte] & enmap) is non-zero is written as "%HH" using uppercase
+ * hexadecimal digits. If a code point cannot be represented in the target
+ * encoding, its percent-encoded numeric character reference is appended.
+ *
++ * The output encoding of the target encoding is used: UTF-16BE, UTF-16LE and
++ * replacement are replaced with UTF-8, see
++ * https://encoding.spec.whatwg.org/#get-an-output-encoding
++ *
+ * If encoding is UTF-8, no conversion is performed. If space_as_plus is true,
+ * an encoded U+0020 SPACE is replaced with '+' before the map is checked.
+ * The input is expected to be valid UTF-8; the function does not validate it.
+ *
+ * @param[in] Pointer to UTF-8 data. Not NULL.
+ * @param[in] Length of data. Can be 0.
+ * @param[in, out] Output string. Can be uninitialized (data = NULL). Encoded
+ * data is appended to any existing content. Not NULL.
+ * @param[in] Memory object used to allocate or resize the output string. Not
+ * NULL.
+ * @param[in] Table of 256 entries indexed by encoded byte, each entry is a bit
+ * mask of lxb_url_map_type_t values. Not NULL.
+ * @param[in] Target encoding. Not NULL.
+ * @param[in] Mask selecting the bytes to percent-encode.
+ * @param[in] Replace an encoded U+0020 SPACE with '+' if true.
+ *
+ * @return LXB_STATUS_OK if successful, otherwise an error status value.
+ */
+LXB_API lxb_status_t
+lxb_url_percent_encode_encoding(const lxb_char_t *data, size_t length,
+ lexbor_str_t *str, lexbor_mraw_t *mraw,
+ const uint8_t *url_map,
+ const lxb_encoding_data_t *encoding,
+ lxb_url_map_type_t enmap,
+ bool space_as_plus);
+
+/*
+ * Get the URL percent-encoding map.
+ *
+ * Returns the built-in lookup table for the percent-encode sets defined by the
+ * URL specification. The table contains 256 entries indexed by byte value.
+ * Each entry is a bit mask of the lxb_url_map_type_t sets in which the byte
+ * must be percent-encoded.
+ *
+ * The returned map can be passed to lxb_url_percent_encode_utf_8() or
+ * lxb_url_percent_encode_encoding(). It has static storage duration and must
+ * not be modified or freed.
+ *
+ * @return Pointer to a read-only table of 256 entries. Never NULL.
+ */
+LXB_API const uint8_t *
+lxb_url_get_percent_encoding_map(void);
+
/*
* Erase URL.
*
@@@ -885,15 -763,51 +889,60 @@@ LXB_API lxb_status_
lxb_url_search_params_serialize(lxb_url_search_params_t *search_params,
lexbor_callback_f cb, void *ctx);
+/**
+ * Returns whether the URL is special.
+ *
+ * @param[in] lxb_url_t *. Cannot be NULL.
+ * @return true if URL is special, false otherwise.
+ */
+LXB_API bool
+lxb_url_is_special(const lxb_url_t *url);
+
+ /*
+ * Reset the URL path to an empty list.
+ *
+ * Frees the path buffer using url->mraw, resets the segment count and clears
+ * the opaque flag. Does nothing if the path buffer is already NULL.
+ *
+ * @param[in, out] URL object. Not NULL.
+ */
+ LXB_API void
+ lxb_url_path_set_null(lxb_url_t *url);
+
+ /*
+ * Set the host to the empty host.
+ *
+ * Frees any domain or opaque host buffer using mraw and sets the host type
+ * to LXB_URL_HOST_TYPE_EMPTY.
+ *
+ * @param[in, out] Host object. Not NULL.
+ * @param[in] Memory object associated with the host. Not NULL.
+ */
+ LXB_API void
+ lxb_url_host_set_empty(lxb_url_host_t *host, lexbor_mraw_t *mraw);
+
+ /*
+ * Set the URL query to null.
+ *
+ * Frees the query buffer using url->mraw. Does nothing if the query
+ * is already null.
+ *
+ * @param[in, out] URL object. Not NULL.
+ */
+ LXB_API void
+ lxb_url_query_set_null(lxb_url_t *url);
+
+ /*
+ * Set the URL fragment to null.
+ *
+ * Frees the fragment buffer using url->mraw. Does nothing if the
+ * fragment is already null.
+ *
+ * @param[in, out] URL object. Not NULL.
+ */
+ LXB_API void
+ lxb_url_fragment_set_null(lxb_url_t *url);
+
/*
* Inline functions.
*/
diff --cc ext/lexbor/patches/0001-Expose-line-and-column-information-for-use-in-PHP.patch
index 6bc4929e9b0,56079940029..2cd28f9eaf1
--- a/ext/lexbor/patches/0001-Expose-line-and-column-information-for-use-in-PHP.patch
+++ b/ext/lexbor/patches/0001-Expose-line-and-column-information-for-use-in-PHP.patch
@@@ -1,7 -1,7 +1,7 @@@
From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
From: Niels Dossche <7771979+nielsdos@users.noreply.github.com>
Date: Sat, 26 Aug 2023 15:08:59 +0200
- Subject: [PATCH 01/15] Expose line and column information for use in PHP
-Subject: [PATCH 01/21] Expose line and column information for use in PHP
++Subject: [PATCH 01/24] Expose line and column information for use in PHP
---
source/lexbor/dom/interfaces/node.h | 2 ++
diff --cc ext/lexbor/patches/0002-Track-implied-added-nodes-for-options-use-in-PHP.patch
index 29bc4b12adc,8dc8cb984d2..4cff1b906f8
--- a/ext/lexbor/patches/0002-Track-implied-added-nodes-for-options-use-in-PHP.patch
+++ b/ext/lexbor/patches/0002-Track-implied-added-nodes-for-options-use-in-PHP.patch
@@@ -1,7 -1,7 +1,7 @@@
From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
From: Niels Dossche <7771979+nielsdos@users.noreply.github.com>
Date: Mon, 14 Aug 2023 20:18:51 +0200
- Subject: [PATCH 02/15] Track implied added nodes for options use in PHP
-Subject: [PATCH 02/21] Track implied added nodes for options use in PHP
++Subject: [PATCH 02/24] Track implied added nodes for options use in PHP
---
source/lexbor/html/tree.h | 3 +++
diff --cc ext/lexbor/patches/0003-Patch-utilities-and-data-structure-to-be-able-to-gen.patch
index 286fd2e16fd,f93d9fe8f86..19b411b2ffc
--- a/ext/lexbor/patches/0003-Patch-utilities-and-data-structure-to-be-able-to-gen.patch
+++ b/ext/lexbor/patches/0003-Patch-utilities-and-data-structure-to-be-able-to-gen.patch
@@@ -1,7 -1,7 +1,7 @@@
From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
From: Niels Dossche <7771979+nielsdos@users.noreply.github.com>
Date: Thu, 24 Aug 2023 22:57:48 +0200
- Subject: [PATCH 03/15] Patch utilities and data structure to be able to
-Subject: [PATCH 03/21] Patch utilities and data structure to be able to
++Subject: [PATCH 03/24] Patch utilities and data structure to be able to
generate smaller lookup tables
Changed the generation script to check if everything fits in 32-bits.
diff --cc ext/lexbor/patches/0004-Remove-unused-upper-case-tag-static-data.patch
index 34b58217aa6,35bec95e6b9..67f241a96cf
--- a/ext/lexbor/patches/0004-Remove-unused-upper-case-tag-static-data.patch
+++ b/ext/lexbor/patches/0004-Remove-unused-upper-case-tag-static-data.patch
@@@ -1,7 -1,7 +1,7 @@@
From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
From: Niels Dossche <7771979+nielsdos@users.noreply.github.com>
Date: Wed, 29 Nov 2023 21:26:47 +0100
- Subject: [PATCH 04/15] Remove unused upper case tag static data
-Subject: [PATCH 04/21] Remove unused upper case tag static data
++Subject: [PATCH 04/24] Remove unused upper case tag static data
---
source/lexbor/tag/res.h | 2 ++
diff --cc ext/lexbor/patches/0005-Shrink-size-of-static-binary-search-tree.patch
index 0c88f603172,68f2d4d379e..a36832333e1
--- a/ext/lexbor/patches/0005-Shrink-size-of-static-binary-search-tree.patch
+++ b/ext/lexbor/patches/0005-Shrink-size-of-static-binary-search-tree.patch
@@@ -1,7 -1,7 +1,7 @@@
From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
From: Niels Dossche <7771979+nielsdos@users.noreply.github.com>
Date: Wed, 29 Nov 2023 21:29:31 +0100
- Subject: [PATCH 05/15] Shrink size of static binary search tree
-Subject: [PATCH 05/21] Shrink size of static binary search tree
++Subject: [PATCH 05/24] Shrink size of static binary search tree
This also makes it more efficient on the data cache.
---
diff --cc ext/lexbor/patches/0006-Patch-out-unused-CSS-style-code.patch
index 3e31f792588,5d63d17116f..fa25a94ef3f
--- a/ext/lexbor/patches/0006-Patch-out-unused-CSS-style-code.patch
+++ b/ext/lexbor/patches/0006-Patch-out-unused-CSS-style-code.patch
@@@ -1,7 -1,7 +1,7 @@@
From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
From: Niels Dossche <7771979+nielsdos@users.noreply.github.com>
Date: Sun, 7 Jan 2024 21:59:28 +0100
- Subject: [PATCH 06/15] Patch out unused CSS style code
-Subject: [PATCH 06/21] Patch out unused CSS style code
++Subject: [PATCH 06/24] Patch out unused CSS style code
---
source/lexbor/css/rule.h | 2 ++
diff --cc ext/lexbor/patches/0007-Add-lxb_url_is_special-to-the-public-API-362.patch
index 505cb66844c,00000000000..2f939cb69f2
mode 100644,000000..100644
--- a/ext/lexbor/patches/0007-Add-lxb_url_is_special-to-the-public-API-362.patch
+++ b/ext/lexbor/patches/0007-Add-lxb_url_is_special-to-the-public-API-362.patch
@@@ -1,44 -1,0 +1,44 @@@
+From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
+From: =?UTF-8?q?M=C3=A1t=C3=A9=20Kocsis?= <kocsismate@woohoolabs.com>
+Date: Sun, 17 May 2026 22:17:14 +0200
- Subject: [PATCH 07/15] Add lxb_url_is_special() to the public API (#362)
++Subject: [PATCH 07/24] Add lxb_url_is_special() to the public API (#362)
+
+As https://wiki.php.net/rfc/uri_followup#uri_type_detection relies on this information.
+---
+ source/lexbor/url/url.c | 2 +-
+ source/lexbor/url/url.h | 9 +++++++++
+ 2 files changed, 10 insertions(+), 1 deletion(-)
+
+diff --git a/source/lexbor/url/url.c b/source/lexbor/url/url.c
+index 5a11434..a5b323f 100644
+--- a/source/lexbor/url/url.c
++++ b/source/lexbor/url/url.c
+@@ -860,7 +860,7 @@ lxb_url_is_url_codepoint(lxb_codepoint_t cp)
+ return lxb_url_codepoint_alphanumeric[(lxb_char_t) cp] != 0xFF;
+ }
+
+-lxb_inline bool
++bool
+ lxb_url_is_special(const lxb_url_t *url)
+ {
+ return url->scheme.type != LXB_URL_SCHEMEL_TYPE__UNKNOWN;
+diff --git a/source/lexbor/url/url.h b/source/lexbor/url/url.h
+index 4ed3f32..6cc6f10 100644
+--- a/source/lexbor/url/url.h
++++ b/source/lexbor/url/url.h
+@@ -763,6 +763,15 @@ LXB_API lxb_status_t
+ lxb_url_search_params_serialize(lxb_url_search_params_t *search_params,
+ lexbor_callback_f cb, void *ctx);
+
++/**
++ * Returns whether the URL is special.
++ *
++ * @param[in] lxb_url_t *. Cannot be NULL.
++ * @return true if URL is special, false otherwise.
++ */
++LXB_API bool
++lxb_url_is_special(const lxb_url_t *url);
++
+ /*
+ * Inline functions.
+ */
diff --cc ext/lexbor/patches/0008-URL-fixed-setters-for-empty-hosts.patch
index 1a6b8278a11,3e69d07f95c..d72c82fbdae
--- a/ext/lexbor/patches/0008-URL-fixed-setters-for-empty-hosts.patch
+++ b/ext/lexbor/patches/0008-URL-fixed-setters-for-empty-hosts.patch
@@@ -1,7 -1,7 +1,7 @@@
From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
From: Alexander Borisov <lex.borisov@gmail.com>
Date: Fri, 26 Jun 2026 18:55:56 +0300
- Subject: [PATCH 08/15] URL: fixed setters for empty hosts.
-Subject: [PATCH 07/21] URL: fixed setters for empty hosts.
++Subject: [PATCH 08/24] URL: fixed setters for empty hosts.
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
diff --cc ext/lexbor/patches/0009-URL-fixed-uninitialized-memory-in-the-path-buffer-gr.patch
index 0b942644bf8,3f7d7e1eab3..b2ac318e2ec
--- a/ext/lexbor/patches/0009-URL-fixed-uninitialized-memory-in-the-path-buffer-gr.patch
+++ b/ext/lexbor/patches/0009-URL-fixed-uninitialized-memory-in-the-path-buffer-gr.patch
@@@ -1,7 -1,7 +1,7 @@@
From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
From: Alexander Borisov <lex.borisov@gmail.com>
Date: Fri, 5 Jun 2026 22:13:32 +0300
- Subject: [PATCH 09/15] URL: fixed uninitialized memory in the path buffer
-Subject: [PATCH 08/21] URL: fixed uninitialized memory in the path buffer
++Subject: [PATCH 09/24] URL: fixed uninitialized memory in the path buffer
growth.
When a path was long enough to outgrow the on-stack buffer, the first
diff --cc ext/lexbor/patches/0010-Fix-parsing-for-URL-containing-empty-host-and-userin.patch
index 4ab6177f9a2,c94e76622a4..937722a246b
--- a/ext/lexbor/patches/0010-Fix-parsing-for-URL-containing-empty-host-and-userin.patch
+++ b/ext/lexbor/patches/0010-Fix-parsing-for-URL-containing-empty-host-and-userin.patch
@@@ -1,7 -1,7 +1,7 @@@
From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?M=C3=A1t=C3=A9=20Kocsis?= <kocsismate@woohoolabs.com>
Date: Thu, 9 Jul 2026 21:51:05 +0200
- Subject: [PATCH 10/15] Fix parsing for URL containing empty host and userinfo
-Subject: [PATCH 09/21] Fix parsing for URL containing empty host and userinfo
++Subject: [PATCH 10/24] Fix parsing for URL containing empty host and userinfo
The returned error code (LXB_URL_ERROR_TYPE_INVALID_CREDENTIALS) apparently contradicts the specification:
diff --cc ext/lexbor/patches/0011-Percent-encode-the-caret-in-the-path.patch
index 0a07b7095fd,69156c42159..9c1a0708a75
--- a/ext/lexbor/patches/0011-Percent-encode-the-caret-in-the-path.patch
+++ b/ext/lexbor/patches/0011-Percent-encode-the-caret-in-the-path.patch
@@@ -1,7 -1,7 +1,7 @@@
From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?M=C3=A1t=C3=A9=20Kocsis?= <kocsismate@woohoolabs.com>
Date: Fri, 10 Jul 2026 22:31:16 +0200
- Subject: [PATCH 11/15] Percent-encode the caret in the path
-Subject: [PATCH 10/21] Percent-encode the caret in the path
++Subject: [PATCH 11/24] Percent-encode the caret in the path
The caret (^) is part of the path percent-encode set:
diff --cc ext/lexbor/patches/0012-URL-added-public-IPv6-parser.patch
index 42cd11a6729,00000000000..a4007d2a68f
mode 100644,000000..100644
--- a/ext/lexbor/patches/0012-URL-added-public-IPv6-parser.patch
+++ b/ext/lexbor/patches/0012-URL-added-public-IPv6-parser.patch
@@@ -1,320 -1,0 +1,320 @@@
+From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
+From: Alexander Borisov <lex.borisov@gmail.com>
+Date: Wed, 12 Aug 2026 23:29:20 +0300
- Subject: [PATCH 12/15] URL: added public IPv6 parser.
++Subject: [PATCH 12/24] URL: added public IPv6 parser.
+MIME-Version: 1.0
+Content-Type: text/plain; charset=UTF-8
+Content-Transfer-Encoding: 8bit
+
+Added lxb_url_parse_host_ipv6() — a public entry point to the IPv6
+parser from the WHATWG specification:
+https://url.spec.whatwg.org/#concept-ipv6-parser
+
+The address is accepted both with and without the surrounding square
+brackets: "::1" and "[::1]" give the same result.
+
+https://github.com/lexbor/lexbor/pull/402
+
+The API was requested in #402 for use by php/php-src#22268.
+
+Suggested-by: Máté Kocsis (@kocsismate)
+---
+ source/lexbor/url/url.c | 42 +++++++
+ source/lexbor/url/url.h | 26 ++++
+ test/lexbor/url/parse_host_ipv6.c | 190 ++++++++++++++++++++++++++++++
+ 3 files changed, 258 insertions(+)
+ create mode 100644 test/lexbor/url/parse_host_ipv6.c
+
+diff --git a/source/lexbor/url/url.c b/source/lexbor/url/url.c
+index 8099c12..7487762 100644
+--- a/source/lexbor/url/url.c
++++ b/source/lexbor/url/url.c
+@@ -3752,6 +3752,46 @@ lxb_url_is_ipv4(lxb_url_parser_t *parser, const lxb_char_t *data,
+ return status != LXB_STATUS_ERROR;
+ }
+
++lxb_status_t
++lxb_url_parse_host_ipv6(lxb_url_parser_t *parser, const lxb_char_t *data,
++ size_t length, uint16_t *ipv6)
++{
++ lxb_status_t status;
++ lxb_url_parser_t self_parser;
++
++ if (parser == NULL) {
++ parser = &self_parser;
++
++ parser->log = NULL;
++ parser->idna = NULL;
++ parser->buffer = NULL;
++ }
++
++ if (data < data + length && *data == '[') {
++ if (data[length - 1] != ']') {
++ (void) lxb_url_log_append(parser, &data[length - 1],
++ LXB_URL_ERROR_TYPE_IPV6_UNCLOSED);
++
++ status = LXB_STATUS_ERROR_UNEXPECTED_DATA;
++
++ goto done;
++ }
++
++ data += 1;
++ length -= 2;
++ }
++
++ status = lxb_url_ipv6_parse(parser, data, data + length, ipv6);
++
++done:
++
++ if (parser == &self_parser) {
++ lxb_url_parser_destroy(parser, false);
++ }
++
++ return status;
++}
++
+ static lxb_status_t
+ lxb_url_ipv6_parse(lxb_url_parser_t *parser, const lxb_char_t *data,
+ const lxb_char_t *end, uint16_t *ipv6)
+@@ -3763,6 +3803,8 @@ lxb_url_ipv6_parse(lxb_url_parser_t *parser, const lxb_char_t *data,
+ const lxb_char_t *p;
+ lxb_url_error_type_t err_type;
+
++ memset(ipv6, 0x00, sizeof(uint16_t) * 8);
++
+ piece = ipv6;
+ compress = NULL;
+ p = data;
+diff --git a/source/lexbor/url/url.h b/source/lexbor/url/url.h
+index 6cc6f10..aa50485 100644
+--- a/source/lexbor/url/url.h
++++ b/source/lexbor/url/url.h
+@@ -321,6 +321,32 @@ lxb_url_parse_basic(lxb_url_parser_t *parser, lxb_url_t *url,
+ const lxb_char_t *data, size_t length,
+ lxb_url_state_t override_state, lxb_encoding_t encoding);
+
++/*
++ * IPv6 parser.
++ *
++ * This function is an implementation of IPv6 parsing according to the WHATWG
++ * specification.
++ * https://url.spec.whatwg.org/#concept-ipv6-parser
++ *
++ * The address can be passed both with and without the surrounding square
++ * brackets: "::1" and "[::1]" give the same result. If the opening bracket is
++ * present, the closing one is required.
++ *
++ * The output buffer is zeroed by the function, there is no need to prepare it.
++ * Use the lxb_url_serialize_host_ipv6() function to serialize the result.
++ *
++ * @param[in] lxb_url_parser_t *. Can be NULL.
++ * @param[in] Pointer to the beginning of the data. Not NULL.
++ * @param[in] Length of the data. Can be 0.
++ * @param[out] Buffer for eight (uint16_t[8]) IPv6 pieces. Not NULL. The value
++ * is meaningful only if LXB_STATUS_OK is returned.
++ *
++ * @return LXB_STATUS_OK if successful, otherwise an error status value.
++ */
++LXB_API lxb_status_t
++lxb_url_parse_host_ipv6(lxb_url_parser_t *parser, const lxb_char_t *data,
++ size_t length, uint16_t *ipv6);
++
+ /*
+ * Erase URL.
+ *
+diff --git a/test/lexbor/url/parse_host_ipv6.c b/test/lexbor/url/parse_host_ipv6.c
+new file mode 100644
+index 0000000..bbf5293
+--- /dev/null
++++ b/test/lexbor/url/parse_host_ipv6.c
+@@ -0,0 +1,190 @@
++/*
++ * Copyright (C) 2026 Alexander Borisov
++ *
++ * Author: Alexander Borisov <borisov@lexbor.com>
++ */
++
++#include <unit/test.h>
++#include <lexbor/url/url.h>
++
++
++typedef struct {
++ const lxb_char_t *input;
++ size_t length;
++ uint16_t ipv6[8];
++}
++ipv6_success_t;
++
++typedef struct {
++ const lxb_char_t *input;
++ lxb_url_error_type_t error;
++}
++ipv6_failure_t;
++
++
++static const ipv6_success_t success_entries[] = {
++ {
++ (const lxb_char_t *) "::",
++ sizeof("::") - 1,
++ {0, 0, 0, 0, 0, 0, 0, 0}
++ },
++ {
++ (const lxb_char_t *) "::1",
++ sizeof("::1") - 1,
++ {0, 0, 0, 0, 0, 0, 0, 1}
++ },
++ {
++ (const lxb_char_t *) "[::1]",
++ sizeof("[::1]") - 1,
++ {0, 0, 0, 0, 0, 0, 0, 1}
++ },
++ {
++ (const lxb_char_t *) "1:2:3:4:5:6:7:8",
++ sizeof("1:2:3:4:5:6:7:8") - 1,
++ {1, 2, 3, 4, 5, 6, 7, 8}
++ },
++ {
++ (const lxb_char_t *) "2001:db8::ff00:42:8329",
++ sizeof("2001:db8::ff00:42:8329") - 1,
++ {0x2001, 0x0db8, 0, 0, 0, 0xff00, 0x0042, 0x8329}
++ },
++ {
++ (const lxb_char_t *) "::ffff:192.0.2.1",
++ sizeof("::ffff:192.0.2.1") - 1,
++ {0, 0, 0, 0, 0, 0xffff, 0xc000, 0x0201}
++ },
++ {
++ (const lxb_char_t *) "[::1]ignored",
++ 5,
++ {0, 0, 0, 0, 0, 0, 0, 1}
++ }
++};
++
++static const ipv6_failure_t failure_entries[] = {
++ {
++ (const lxb_char_t *) "",
++ LXB_URL_ERROR_TYPE_IPV6_TOO_FEW_PIECES
++ },
++ {
++ (const lxb_char_t *) "[::1",
++ LXB_URL_ERROR_TYPE_IPV6_UNCLOSED
++ },
++ {
++ (const lxb_char_t *) ":",
++ LXB_URL_ERROR_TYPE_IPV6_INVALID_COMPRESSION
++ },
++ {
++ (const lxb_char_t *) "1::2::3",
++ LXB_URL_ERROR_TYPE_IPV6_MULTIPLE_COMPRESSION
++ },
++ {
++ (const lxb_char_t *) "1:2:3:4:5:6:7:8:9",
++ LXB_URL_ERROR_TYPE_IPV6_TOO_MANY_PIECES
++ },
++ {
++ (const lxb_char_t *) "1:2:3:4:5:6:7",
++ LXB_URL_ERROR_TYPE_IPV6_TOO_FEW_PIECES
++ },
++ {
++ (const lxb_char_t *) "1:2:3:4:5:6:7:g",
++ LXB_URL_ERROR_TYPE_IPV6_INVALID_CODE_POINT
++ },
++ {
++ (const lxb_char_t *) "1:2:3:4:5:6:7:1.2.3.4",
++ LXB_URL_ERROR_TYPE_IPV4_IN_IPV6_TOO_MANY_PIECES
++ },
++ {
++ (const lxb_char_t *) "::ffff:.1.2.3",
++ LXB_URL_ERROR_TYPE_IPV4_IN_IPV6_INVALID_CODE_POINT
++ },
++ {
++ (const lxb_char_t *) "::ffff:192.0.2.256",
++ LXB_URL_ERROR_TYPE_IPV4_IN_IPV6_OUT_OF_RANGE_PART
++ },
++ {
++ (const lxb_char_t *) "::ffff:192.0.2",
++ LXB_URL_ERROR_TYPE_IPV4_IN_IPV6_TOO_FEW_PARTS
++ }
++};
++
++
++TEST_BEGIN(parse_success)
++{
++ size_t length;
++ lxb_status_t status;
++ uint16_t ipv6[8];
++
++ length = sizeof(success_entries) / sizeof(ipv6_success_t);
++
++ for (size_t i = 0; i < length; i++) {
++ memset(ipv6, 0xff, sizeof(ipv6));
++
++ status = lxb_url_parse_host_ipv6(NULL, success_entries[i].input,
++ success_entries[i].length, ipv6);
++ test_eq(status, LXB_STATUS_OK);
++
++ for (size_t j = 0; j < 8; j++) {
++ test_eq_u_short(ipv6[j], success_entries[i].ipv6[j]);
++ }
++ }
++}
++TEST_END
++
++TEST_BEGIN(parse_failure)
++{
++ size_t length;
++ lxb_status_t status;
++ lxb_url_parser_t parser;
++ lexbor_plog_entry_t *error;
++
++ status = lxb_url_parser_init(&parser, NULL);
++ test_eq(status, LXB_STATUS_OK);
++
++ length = sizeof(failure_entries) / sizeof(ipv6_failure_t);
++
++ for (size_t i = 0; i < length; i++) {
++ status = lxb_url_parse_host_ipv6(
++ &parser, failure_entries[i].input,
++ strlen((const char *) failure_entries[i].input),
++ (uint16_t[8]) {0});
++
++ test_eq(status, LXB_STATUS_ERROR_UNEXPECTED_DATA);
++ test_ne(parser.log, NULL);
++ test_eq_size(lexbor_plog_length(parser.log), 1UL);
++
++ error = lexbor_array_obj_get(&parser.log->list, 0);
++ test_ne(error, NULL);
++ test_eq(error->id, failure_entries[i].error);
++
++ lxb_url_parser_clean(&parser);
++ }
++
++ lxb_url_parser_memory_destroy(&parser);
++ lxb_url_parser_destroy(&parser, false);
++}
++TEST_END
++
++TEST_BEGIN(parse_failure_without_parser)
++{
++ lxb_status_t status;
++ uint16_t ipv6[8];
++
++ static const lexbor_str_t input = lexbor_str("::ffff:192.0.2.256");
++
++ status = lxb_url_parse_host_ipv6(NULL, input.data, input.length, ipv6);
++ test_eq(status, LXB_STATUS_ERROR_UNEXPECTED_DATA);
++}
++TEST_END
++
++int
++main(int argc, const char *argv[])
++{
++ TEST_INIT();
++
++ TEST_ADD(parse_success);
++ TEST_ADD(parse_failure);
++ TEST_ADD(parse_failure_without_parser);
++
++ TEST_RUN("lexbor/url/parse_host_ipv6");
++ TEST_RELEASE();
++}
diff --cc ext/lexbor/patches/0013-URL-added-public-percent-encoder-API.patch
index edefe14ebc0,00000000000..729969019b4
mode 100644,000000..100644
--- a/ext/lexbor/patches/0013-URL-added-public-percent-encoder-API.patch
+++ b/ext/lexbor/patches/0013-URL-added-public-percent-encoder-API.patch
@@@ -1,626 -1,0 +1,626 @@@
+From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
+From: Alexander Borisov <lex.borisov@gmail.com>
+Date: Thu, 13 Aug 2026 23:29:16 +0300
- Subject: [PATCH 13/15] URL: added public percent-encoder API.
++Subject: [PATCH 13/24] URL: added public percent-encoder API.
+MIME-Version: 1.0
+Content-Type: text/plain; charset=UTF-8
+Content-Transfer-Encoding: 8bit
+
+Added public entry points to the percent-encoder from the WHATWG
+specification:
+https://url.spec.whatwg.org/#percent-encoded-bytes
+
+ lxb_url_percent_encode_utf_8()
+ lxb_url_percent_encode_encoding()
+ lxb_url_get_percent_encoding_map()
+
+Both encoders take a caller-supplied table of 256 entries indexed by byte
+value, where each entry is a bit mask of lxb_url_map_type_t values, so the
+percent-encode sets can be adjusted without patching the library.
+lxb_url_get_percent_encoding_map() returns the built-in table for callers
+that need only the sets defined by the specification.
+
+https://github.com/lexbor/lexbor/pull/404
+
+The API was requested in #404 for use by PHP:
+https://wiki.php.net/rfc/uri_followup#percent-encoding_support
+
+Based-on-patch-by: Máté Kocsis (@kocsismate)
+---
+ source/lexbor/url/url.c | 93 ++++++++-----
+ source/lexbor/url/url.h | 96 +++++++++++++
+ test/lexbor/url/percent_encode.c | 228 +++++++++++++++++++++++++++++++
+ 3 files changed, 380 insertions(+), 37 deletions(-)
+ create mode 100644 test/lexbor/url/percent_encode.c
+
+diff --git a/source/lexbor/url/url.c b/source/lexbor/url/url.c
+index 7487762..69d9196 100644
+--- a/source/lexbor/url/url.c
++++ b/source/lexbor/url/url.c
+@@ -27,20 +27,6 @@
+ #define LXB_URL_BUFFER_NUM_SIZE 128
+
+
+-typedef enum {
+- LXB_URL_MAP_UNDEF = 0x00,
+- LXB_URL_MAP_C0 = 0x01,
+- LXB_URL_MAP_FRAGMENT = 0x02,
+- LXB_URL_MAP_QUERY = 0x04,
+- LXB_URL_MAP_SPECIAL_QUERY = 0x08,
+- LXB_URL_MAP_PATH = 0x10,
+- LXB_URL_MAP_USERINFO = 0x20,
+- LXB_URL_MAP_COMPONENT = 0x40,
+- LXB_URL_MAP_X_WWW_FORM = 0x80,
+- LXB_URL_MAP_ALL = 0xff
+-}
+-lxb_url_map_type_t;
+-
+ typedef enum {
+ LXB_URL_HOST_OPT_UNDEF = 0 << 0,
+ LXB_URL_HOST_OPT_NOT_SPECIAL = 1 << 0,
+@@ -563,7 +549,7 @@ lxb_url_path_fix_windows_drive(lxb_url_t *url, lxb_char_t *sbuf,
+ static lxb_status_t
+ lxb_url_percent_encode_after_encoding(const lxb_char_t *data,
+ const lxb_char_t *end, lexbor_str_t *str,
+- lexbor_mraw_t *mraw,
++ lexbor_mraw_t *mraw, const uint8_t *url_map,
+ const lxb_encoding_data_t *encoding,
+ lxb_url_map_type_t enmap,
+ bool space_as_plus);
+@@ -571,7 +557,7 @@ lxb_url_percent_encode_after_encoding(const lxb_char_t *data,
+ static lxb_status_t
+ lxb_url_percent_encode_after_utf_8(const lxb_char_t *data,
+ const lxb_char_t *end, lexbor_str_t *str,
+- lexbor_mraw_t *mraw,
++ lexbor_mraw_t *mraw, const uint8_t *url_map,
+ lxb_url_map_type_t enmap,
+ bool space_as_plus);
+
+@@ -1757,9 +1743,9 @@ again:
+ tmp = (pswd != NULL) ? pswd - 1 : p;
+
+ if (tmp > begin) {
+- status = lxb_url_percent_encode_after_utf_8(begin, tmp,
+- &url->username, url->mraw,
+- LXB_URL_MAP_USERINFO, false);
++ status = lxb_url_percent_encode_after_utf_8(begin,
++ tmp, &url->username, url->mraw, lxb_url_map,
++ LXB_URL_MAP_USERINFO, false);
+ if (status != LXB_STATUS_OK) {
+ lxb_url_parse_return(orig_data, buf, status);
+ }
+@@ -1768,8 +1754,8 @@ again:
+
+ if (pswd != NULL && p > pswd) {
+ status = lxb_url_percent_encode_after_utf_8(pswd, p,
+- &url->password, url->mraw,
+- LXB_URL_MAP_USERINFO, false);
++ &url->password, url->mraw, lxb_url_map,
++ LXB_URL_MAP_USERINFO, false);
+ if (status != LXB_STATUS_OK) {
+ lxb_url_parse_return(orig_data, buf, status);
+ }
+@@ -2319,8 +2305,8 @@ again:
+ if (p >= end) {
+ tmp_str.data = NULL;
+
+- status = lxb_url_percent_encode_after_utf_8(begin, p,
+- &tmp_str, url->mraw,
++ status = lxb_url_percent_encode_after_utf_8(begin, p, &tmp_str,
++ url->mraw, lxb_url_map,
+ LXB_URL_MAP_C0, false);
+ if (status != LXB_STATUS_OK) {
+ lxb_url_parse_return(orig_data, buf, status);
+@@ -2336,8 +2322,8 @@ again:
+ if (c == '#' || c == '?') {
+ tmp_str.data = NULL;
+
+- status = lxb_url_percent_encode_after_utf_8(begin, p,
+- &tmp_str, url->mraw,
++ status = lxb_url_percent_encode_after_utf_8(begin, p, &tmp_str,
++ url->mraw, lxb_url_map,
+ LXB_URL_MAP_C0, false);
+ if (status != LXB_STATUS_OK) {
+ lxb_url_parse_return(orig_data, buf, status);
+@@ -2407,7 +2393,8 @@ again:
+
+ status = lxb_url_percent_encode_after_encoding(begin, p,
+ &url->query,
+- url->mraw, enc,
++ url->mraw,
++ lxb_url_map, enc,
+ map_type, false);
+ if (status != LXB_STATUS_OK) {
+ lxb_url_parse_return(orig_data, buf, status);
+@@ -2461,7 +2448,7 @@ again:
+ }
+
+ status = lxb_url_percent_encode_after_utf_8(begin, p, &url->fragment,
+- url->mraw,
++ url->mraw, lxb_url_map,
+ LXB_URL_MAP_FRAGMENT, false);
+ lxb_url_parse_return(orig_data, buf, status);
+
+@@ -3161,10 +3148,23 @@ lxb_url_scheme_find(const lxb_char_t *data, size_t length)
+ return &lxb_url_scheme_res[LXB_URL_SCHEMEL_TYPE__UNKNOWN];
+ }
+
++lxb_status_t
++lxb_url_percent_encode_encoding(const lxb_char_t *data, size_t length,
++ lexbor_str_t *str, lexbor_mraw_t *mraw,
++ const uint8_t *url_map,
++ const lxb_encoding_data_t *encoding,
++ lxb_url_map_type_t enmap,
++ bool space_as_plus)
++{
++ return lxb_url_percent_encode_after_encoding(data, data + length, str, mraw,
++ url_map, encoding, enmap,
++ space_as_plus);
++}
++
+ static lxb_status_t
+ lxb_url_percent_encode_after_encoding(const lxb_char_t *data,
+ const lxb_char_t *end, lexbor_str_t *str,
+- lexbor_mraw_t *mraw,
++ lexbor_mraw_t *mraw, const uint8_t *url_map,
+ const lxb_encoding_data_t *encoding,
+ lxb_url_map_type_t enmap,
+ bool space_as_plus)
+@@ -3182,7 +3182,8 @@ lxb_url_percent_encode_after_encoding(const lxb_char_t *data,
+
+ if (encoding->encoding == LXB_ENCODING_UTF_8) {
+ return lxb_url_percent_encode_after_utf_8(data, end, str, mraw,
+- enmap, space_as_plus);
++ url_map, enmap,
++ space_as_plus);
+ }
+
+ lxb_url_encoding_init(encoding, &encode);
+@@ -3193,7 +3194,7 @@ lxb_url_percent_encode_after_encoding(const lxb_char_t *data,
+ /* Only valid for UTF-8. */
+
+ while (p < end) {
+- if (lxb_url_map[*p++] & enmap) {
++ if (url_map[*p++] & enmap) {
+ length += 2;
+ }
+ }
+@@ -3249,7 +3250,7 @@ lxb_url_percent_encode_after_encoding(const lxb_char_t *data,
+ return LXB_STATUS_ERROR_MEMORY_ALLOCATION;
+ }
+ }
+- else if (lxb_url_map[c] & enmap) {
++ else if (url_map[c] & enmap) {
+ percent[1] = lexbor_str_res_char_to_two_hex_value[c][0];
+ percent[2] = lexbor_str_res_char_to_two_hex_value[c][1];
+
+@@ -3280,10 +3281,20 @@ lxb_url_percent_encode_after_encoding(const lxb_char_t *data,
+ return LXB_STATUS_OK;
+ }
+
++lxb_status_t
++lxb_url_percent_encode_utf_8(const lxb_char_t *data, size_t length,
++ lexbor_str_t *str, lexbor_mraw_t *mraw,
++ const uint8_t *url_map, lxb_url_map_type_t enmap,
++ bool space_as_plus)
++{
++ return lxb_url_percent_encode_after_utf_8(data, data + length, str, mraw,
++ url_map, enmap, space_as_plus);
++}
++
+ static lxb_status_t
+ lxb_url_percent_encode_after_utf_8(const lxb_char_t *data,
+ const lxb_char_t *end, lexbor_str_t *str,
+- lexbor_mraw_t *mraw,
++ lexbor_mraw_t *mraw, const uint8_t *url_map,
+ lxb_url_map_type_t enmap,
+ bool space_as_plus)
+ {
+@@ -3298,7 +3309,7 @@ lxb_url_percent_encode_after_utf_8(const lxb_char_t *data,
+ /* Only valid for UTF-8. */
+
+ while (p < end) {
+- if (lxb_url_map[*p++] & enmap) {
++ if (url_map[*p++] & enmap) {
+ length += 2;
+ }
+ }
+@@ -3317,7 +3328,7 @@ lxb_url_percent_encode_after_utf_8(const lxb_char_t *data,
+ if (space_as_plus && c == ' ') {
+ *pd++ = '+';
+ }
+- else if (lxb_url_map[c] & enmap) {
++ else if (url_map[c] & enmap) {
+ *pd++ = '%';
+ *pd++ = lexbor_str_res_char_to_two_hex_value[c][0];
+ *pd++ = lexbor_str_res_char_to_two_hex_value[c][1];
+@@ -3335,6 +3346,12 @@ lxb_url_percent_encode_after_utf_8(const lxb_char_t *data,
+ return LXB_STATUS_OK;
+ }
+
++const uint8_t *
++lxb_url_get_percent_encoding_map(void)
++{
++ return lxb_url_map;
++}
++
+ static lxb_status_t
+ lxb_url_host_parse(lxb_url_parser_t *parser, const lxb_char_t *data,
+ const lxb_char_t *end, lxb_url_host_t *host,
+@@ -4065,7 +4082,7 @@ lxb_url_opaque_host_parse(lxb_url_parser_t *parser, const lxb_char_t *data,
+ host->type = LXB_URL_HOST_TYPE_OPAQUE;
+
+ return lxb_url_percent_encode_after_utf_8(data, end, &host->u.opaque, mraw,
+- LXB_URL_MAP_C0, false);
++ lxb_url_map, LXB_URL_MAP_C0, false);
+ }
+
+ static lxb_status_t
+@@ -4344,7 +4361,8 @@ lxb_url_api_username_set(lxb_url_t *url,
+
+ return lxb_url_percent_encode_after_utf_8(username, username + length,
+ &url->username, url->mraw,
+- LXB_URL_MAP_USERINFO, false);
++ lxb_url_map, LXB_URL_MAP_USERINFO,
++ false);
+ }
+
+ lxb_status_t
+@@ -4364,7 +4382,8 @@ lxb_url_api_password_set(lxb_url_t *url,
+
+ return lxb_url_percent_encode_after_utf_8(password, password + length,
+ &url->password, url->mraw,
+- LXB_URL_MAP_USERINFO, false);
++ lxb_url_map, LXB_URL_MAP_USERINFO,
++ false);
+ }
+
+ lxb_status_t
+diff --git a/source/lexbor/url/url.h b/source/lexbor/url/url.h
+index aa50485..d2c9308 100644
+--- a/source/lexbor/url/url.h
++++ b/source/lexbor/url/url.h
+@@ -81,6 +81,20 @@ typedef enum {
+ }
+ lxb_url_state_t;
+
++typedef enum {
++ LXB_URL_MAP_UNDEF = 0x00,
++ LXB_URL_MAP_C0 = 0x01,
++ LXB_URL_MAP_FRAGMENT = 0x02,
++ LXB_URL_MAP_QUERY = 0x04,
++ LXB_URL_MAP_SPECIAL_QUERY = 0x08,
++ LXB_URL_MAP_PATH = 0x10,
++ LXB_URL_MAP_USERINFO = 0x20,
++ LXB_URL_MAP_COMPONENT = 0x40,
++ LXB_URL_MAP_X_WWW_FORM = 0x80,
++ LXB_URL_MAP_ALL = 0xff
++}
++lxb_url_map_type_t;
++
+ /*
+ * New values can only be added downwards.
+ * Before LXB_URL_SCHEMEL_TYPE__LAST_ENTRY.
+@@ -347,6 +361,88 @@ LXB_API lxb_status_t
+ lxb_url_parse_host_ipv6(lxb_url_parser_t *parser, const lxb_char_t *data,
+ size_t length, uint16_t *ipv6);
+
++/*
++ * UTF-8 percent-encoder.
++ *
++ * Percent-encodes bytes from data according to url_map and appends the result
++ * to str. A byte is encoded as "%HH" when the result of
++ * (url_map[byte] & enmap) is non-zero. Uppercase hexadecimal digits are used.
++ * If space_as_plus is true, U+0020 SPACE is encoded as '+' before the map is
++ * checked.
++ *
++ * The input is expected to be valid UTF-8; the function does not validate it.
++ *
++ * @param[in] Pointer to UTF-8 data. Not NULL.
++ * @param[in] Length of data. Can be 0.
++ * @param[in, out] Output string. Can be uninitialized (data = NULL). Encoded
++ * data is appended to any existing content. Not NULL.
++ * @param[in] Memory object used to allocate or resize the output string. Not
++ * NULL.
++ * @param[in] Table of 256 entries indexed by input byte, each entry is a bit
++ * mask of lxb_url_map_type_t values. Not NULL.
++ * @param[in] Mask selecting the bytes to percent-encode.
++ * @param[in] Replace U+0020 SPACE with '+' if true.
++ *
++ * @return LXB_STATUS_OK if successful, otherwise an error status value.
++ */
++LXB_API lxb_status_t
++lxb_url_percent_encode_utf_8(const lxb_char_t *data, size_t length,
++ lexbor_str_t *str, lexbor_mraw_t *mraw,
++ const uint8_t *url_map, lxb_url_map_type_t enmap,
++ bool space_as_plus);
++
++/*
++ * Percent-encode after encoding.
++ *
++ * Converts valid UTF-8 data to the specified encoding and appends the
++ * percent-encoded result to str. Each encoded byte for which
++ * (url_map[byte] & enmap) is non-zero is written as "%HH" using uppercase
++ * hexadecimal digits. If a code point cannot be represented in the target
++ * encoding, its percent-encoded numeric character reference is appended.
++ *
++ * If encoding is UTF-8, no conversion is performed. If space_as_plus is true,
++ * an encoded U+0020 SPACE is replaced with '+' before the map is checked.
++ * The input is expected to be valid UTF-8; the function does not validate it.
++ *
++ * @param[in] Pointer to UTF-8 data. Not NULL.
++ * @param[in] Length of data. Can be 0.
++ * @param[in, out] Output string. Can be uninitialized (data = NULL). Encoded
++ * data is appended to any existing content. Not NULL.
++ * @param[in] Memory object used to allocate or resize the output string. Not
++ * NULL.
++ * @param[in] Table of 256 entries indexed by encoded byte, each entry is a bit
++ * mask of lxb_url_map_type_t values. Not NULL.
++ * @param[in] Target encoding. Not NULL.
++ * @param[in] Mask selecting the bytes to percent-encode.
++ * @param[in] Replace an encoded U+0020 SPACE with '+' if true.
++ *
++ * @return LXB_STATUS_OK if successful, otherwise an error status value.
++ */
++LXB_API lxb_status_t
++lxb_url_percent_encode_encoding(const lxb_char_t *data, size_t length,
++ lexbor_str_t *str, lexbor_mraw_t *mraw,
++ const uint8_t *url_map,
++ const lxb_encoding_data_t *encoding,
++ lxb_url_map_type_t enmap,
++ bool space_as_plus);
++
++/*
++ * Get the URL percent-encoding map.
++ *
++ * Returns the built-in lookup table for the percent-encode sets defined by the
++ * URL specification. The table contains 256 entries indexed by byte value.
++ * Each entry is a bit mask of the lxb_url_map_type_t sets in which the byte
++ * must be percent-encoded.
++ *
++ * The returned map can be passed to lxb_url_percent_encode_utf_8() or
++ * lxb_url_percent_encode_encoding(). It has static storage duration and must
++ * not be modified or freed.
++ *
++ * @return Pointer to a read-only table of 256 entries. Never NULL.
++ */
++LXB_API const uint8_t *
++lxb_url_get_percent_encoding_map(void);
++
+ /*
+ * Erase URL.
+ *
+diff --git a/test/lexbor/url/percent_encode.c b/test/lexbor/url/percent_encode.c
+new file mode 100644
+index 0000000..361e22e
+--- /dev/null
++++ b/test/lexbor/url/percent_encode.c
+@@ -0,0 +1,228 @@
++/*
++ * Copyright (C) 2026 Alexander Borisov
++ *
++ * Author: Alexander Borisov <borisov@lexbor.com>
++ */
++
++#include <unit/test.h>
++#include <lexbor/url/url.h>
++
++
++typedef struct {
++ const lexbor_str_t input;
++ const lexbor_str_t output;
++ lxb_encoding_t encoding;
++ lxb_url_map_type_t enmap;
++ bool space_as_plus;
++ const uint8_t *url_map;
++ const lexbor_str_t initial;
++}
++percent_encode_entry_t;
++
++
++static const uint8_t custom_url_map[256] = {
++ ['a'] = LXB_URL_MAP_QUERY,
++ ['b'] = LXB_URL_MAP_PATH
++};
++
++static const percent_encode_entry_t entries[] = {
++ {
++ lexbor_str(""),
++ lexbor_str(""),
++ LXB_ENCODING_UTF_8,
++ LXB_URL_MAP_COMPONENT,
++ false,
++ NULL,
++ lexbor_str("")
++ },
++ {
++ lexbor_str("AZaz09-._~!*'()"),
++ lexbor_str("AZaz09-._~!*'()"),
++ LXB_ENCODING_UTF_8,
++ LXB_URL_MAP_COMPONENT,
++ false,
++ NULL,
++ lexbor_str("")
++ },
++ {
++ lexbor_str("\x00" "A /?\xC3\xA9"),
++ lexbor_str("%00A%20%2F%3F%C3%A9"),
++ LXB_ENCODING_UTF_8,
++ LXB_URL_MAP_COMPONENT,
++ false,
++ NULL,
++ lexbor_str("")
++ },
++ {
++ lexbor_str("A b"),
++ lexbor_str("prefix:A+b"),
++ LXB_ENCODING_UTF_8,
++ LXB_URL_MAP_COMPONENT,
++ true,
++ NULL,
++ lexbor_str("prefix:")
++ },
++ {
++ lexbor_str("\xE2\x89\xA1\xE2\x80\xBD"),
++ lexbor_str("%E2%89%A1%E2%80%BD"),
++ LXB_ENCODING_UTF_8,
++ LXB_URL_MAP_SPECIAL_QUERY,
++ false,
++ NULL,
++ lexbor_str("")
++ },
++ {
++ lexbor_str("\xE2\x89\xA1"),
++ lexbor_str("%81%DF"),
++ LXB_ENCODING_SHIFT_JIS,
++ LXB_URL_MAP_SPECIAL_QUERY,
++ false,
++ NULL,
++ lexbor_str("")
++ },
++ {
++ lexbor_str("\xE2\x80\xBD"),
++ lexbor_str("%26%238253%3B"),
++ LXB_ENCODING_SHIFT_JIS,
++ LXB_URL_MAP_SPECIAL_QUERY,
++ false,
++ NULL,
++ lexbor_str("")
++ },
++ {
++ lexbor_str("1+1 \xE2\x89\xA1 2%20\xE2\x80\xBD"),
++ lexbor_str("1+1%20%81%DF%202%20%26%238253%3B"),
++ LXB_ENCODING_SHIFT_JIS,
++ LXB_URL_MAP_SPECIAL_QUERY,
++ false,
++ NULL,
++ lexbor_str("")
++ },
++ {
++ lexbor_str("\xC2\xA5"),
++ lexbor_str("%1B(J\\%1B(B"),
++ LXB_ENCODING_ISO_2022_JP,
++ LXB_URL_MAP_SPECIAL_QUERY,
++ false,
++ NULL,
++ lexbor_str("")
++ },
++ {
++ lexbor_str("caf\xC3\xA9"),
++ lexbor_str("caf%E9"),
++ LXB_ENCODING_WINDOWS_1252,
++ LXB_URL_MAP_COMPONENT,
++ false,
++ NULL,
++ lexbor_str("")
++ },
++ {
++ lexbor_str("\xD0\xAF"),
++ lexbor_str("%DF"),
++ LXB_ENCODING_WINDOWS_1251,
++ LXB_URL_MAP_COMPONENT,
++ false,
++ NULL,
++ lexbor_str("")
++ },
++ {
++ lexbor_str("\xE4\xB8\xAD\xE6\x96\x87"),
++ lexbor_str("%A4%A4%A4%E5"),
++ LXB_ENCODING_BIG5,
++ LXB_URL_MAP_COMPONENT,
++ false,
++ NULL,
++ lexbor_str("")
++ },
++ {
++ lexbor_str("a b+c~"),
++ lexbor_str("a+b%2Bc%7E"),
++ LXB_ENCODING_UTF_8,
++ LXB_URL_MAP_X_WWW_FORM,
++ true,
++ NULL,
++ lexbor_str("")
++ },
++ {
++ lexbor_str("abc"),
++ lexbor_str("%61bc"),
++ LXB_ENCODING_UTF_8,
++ LXB_URL_MAP_QUERY,
++ false,
++ custom_url_map,
++ lexbor_str("")
++ }
++};
++
++
++TEST_BEGIN(percent_encode)
++{
++ size_t length;
++ lxb_char_t *data;
++ lxb_status_t status;
++ lexbor_mraw_t mraw;
++ lexbor_str_t str;
++ const uint8_t *url_map, *default_url_map;
++ const lxb_encoding_data_t *encoding;
++ const percent_encode_entry_t *entry;
++
++ status = lexbor_mraw_init(&mraw, 1024);
++ test_eq(status, LXB_STATUS_OK);
++
++ default_url_map = lxb_url_get_percent_encoding_map();
++ test_ne(default_url_map, NULL);
++
++ length = sizeof(entries) / sizeof(percent_encode_entry_t);
++
++ for (size_t i = 0; i < length; i++) {
++ entry = &entries[i];
++ encoding = lxb_encoding_data(entry->encoding);
++ test_ne(encoding, NULL);
++
++ str = (lexbor_str_t) {0};
++
++ if (entry->initial.length != 0) {
++ data = lexbor_str_init_append(&str, &mraw, entry->initial.data,
++ entry->initial.length);
++ test_ne(data, NULL);
++ }
++
++ url_map = entry->url_map;
++ if (url_map == NULL) {
++ url_map = default_url_map;
++ }
++
++ status = lxb_url_percent_encode_encoding(entry->input.data,
++ entry->input.length,
++ &str, &mraw, url_map, encoding,
++ entry->enmap,
++ entry->space_as_plus);
++ test_eq(status, LXB_STATUS_OK);
++
++ if (str.length != entry->output.length
++ || memcmp(str.data, entry->output.data, str.length) != 0)
++ {
++ TEST_PRINTLN("Percent-encode entry %zu (%s)", i + 1,
++ encoding->name);
++ }
++
++ test_eq_str_n(str.data, str.length, entry->output.data,
++ entry->output.length);
++
++ lexbor_str_destroy(&str, &mraw, false);
++ }
++
++ lexbor_mraw_destroy(&mraw, false);
++}
++TEST_END
++
++int
++main(int argc, const char *argv[])
++{
++ TEST_INIT();
++
++ TEST_ADD(percent_encode);
++
++ TEST_RUN("lexbor/url/percent_encode");
++ TEST_RELEASE();
++}
diff --cc ext/lexbor/patches/0014-CSS-fixed-heap-buffer-overflow-in-lexbor-contains-pa.patch
index 05e049118c9,32d52517acb..373a579e084
--- a/ext/lexbor/patches/0014-CSS-fixed-heap-buffer-overflow-in-lexbor-contains-pa.patch
+++ b/ext/lexbor/patches/0014-CSS-fixed-heap-buffer-overflow-in-lexbor-contains-pa.patch
@@@ -1,7 -1,7 +1,7 @@@
From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
From: Alexander Borisov <lex.borisov@gmail.com>
Date: Fri, 5 Jun 2026 22:34:23 +0300
- Subject: [PATCH 14/15] CSS: fixed heap buffer overflow in :lexbor-contains()
-Subject: [PATCH 11/21] CSS: fixed heap buffer overflow in :lexbor-contains()
++Subject: [PATCH 14/24] CSS: fixed heap buffer overflow in :lexbor-contains()
parsing.
The contains string buffer was allocated by the size of the string
diff --cc ext/lexbor/patches/0015-Encoding-fixed-buffer-overflows-in-malformed-decode-.patch
index 31c3f5f95f9,6214cfbda10..68756c46547
--- a/ext/lexbor/patches/0015-Encoding-fixed-buffer-overflows-in-malformed-decode-.patch
+++ b/ext/lexbor/patches/0015-Encoding-fixed-buffer-overflows-in-malformed-decode-.patch
@@@ -1,7 -1,7 +1,7 @@@
From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
From: Alexander Borisov <lex.borisov@gmail.com>
Date: Wed, 10 Jun 2026 19:50:10 +0300
- Subject: [PATCH 15/15] Encoding: fixed buffer overflows in malformed decode
-Subject: [PATCH 12/21] Encoding: fixed buffer overflows in malformed decode
++Subject: [PATCH 15/24] Encoding: fixed buffer overflows in malformed decode
replay.
Fixed out-of-bounds writes in buffering decoders when replacement output
diff --cc ext/lexbor/patches/0016-URL-fixed-tail-pointer-in-URLSearchParams-for-delimi.patch
index 00000000000,89adb8132fc..bc44d2ec9b4
mode 000000,100644..100644
--- a/ext/lexbor/patches/0016-URL-fixed-tail-pointer-in-URLSearchParams-for-delimi.patch
+++ b/ext/lexbor/patches/0016-URL-fixed-tail-pointer-in-URLSearchParams-for-delimi.patch
@@@ -1,0 -1,29 +1,29 @@@
+ From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
+ From: Alexander Borisov <lex.borisov@gmail.com>
+ Date: Fri, 5 Jun 2026 21:46:26 +0300
-Subject: [PATCH 13/21] URL: fixed tail pointer in URLSearchParams for
++Subject: [PATCH 16/24] URL: fixed tail pointer in URLSearchParams for
+ delimiter-free query.
+
+ When a query had a single token without '=' or '&' (e.g. "?abc"), the
+ internal tail pointer wasn't updated, so a later append() could lose the
+ added parameter (and write through a stale pointer). Fixed by keeping the
+ tail pointer in sync.
+
+ Per report from Xiansheng Cao (@HMF2021)
+ ---
+ source/lexbor/url/url.c | 2 ++
+ 1 file changed, 2 insertions(+)
+
+ diff --git a/source/lexbor/url/url.c b/source/lexbor/url/url.c
-index de19239..fcae2d6 100644
++index 69d9196..337c172 100644
+ --- a/source/lexbor/url/url.c
+ +++ b/source/lexbor/url/url.c
-@@ -5106,6 +5106,8 @@ lxb_url_search_params_parse(lxb_url_search_params_t *search_params,
++@@ -5167,6 +5167,8 @@ lxb_url_search_params_parse(lxb_url_search_params_t *search_params,
+ return status;
+ }
+
+ + last = entry;
+ +
+ lexbor_str_init(&entry->value, mraw, 0);
+ if (entry->value.data == NULL) {
+ return LXB_STATUS_ERROR_MEMORY_ALLOCATION;
diff --cc ext/lexbor/patches/0017-URL-Fix-parsing-of-usernames-containing.patch
index 00000000000,9aa3b403994..f6009b9709f
mode 000000,100644..100644
--- a/ext/lexbor/patches/0017-URL-Fix-parsing-of-usernames-containing.patch
+++ b/ext/lexbor/patches/0017-URL-Fix-parsing-of-usernames-containing.patch
@@@ -1,0 -1,38 +1,39 @@@
+ From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
+ From: =?UTF-8?q?Tim=20D=C3=BCsterhus?= <tim@bastelstu.be>
+ Date: Fri, 31 Jul 2026 20:56:18 +0200
-Subject: [PATCH 14/21] URL: Fix parsing of usernames containing `@`
++Subject: [PATCH 17/24] URL: Fix parsing of usernames containing `@`
+
+ Fixes lexbor/lexbor#399.
+ ---
- source/lexbor/url/url.c | 17 +++++++----------
- 1 file changed, 7 insertions(+), 10 deletions(-)
++ source/lexbor/url/url.c | 18 ++++++++----------
++ 1 file changed, 8 insertions(+), 10 deletions(-)
+
+ diff --git a/source/lexbor/url/url.c b/source/lexbor/url/url.c
-index fcae2d6..654e2e6 100644
++index 337c172..a5bc3f0 100644
+ --- a/source/lexbor/url/url.c
+ +++ b/source/lexbor/url/url.c
-@@ -1753,16 +1753,13 @@ again:
++@@ -1739,16 +1739,14 @@ again:
+ break;
+ }
+
+ - if (pswd == NULL || !at_sign) {
+ - tmp = (pswd != NULL) ? pswd - 1 : p;
+ -
+ - if (tmp > begin) {
-- status = lxb_url_percent_encode_after_utf_8(begin, tmp,
-- &url->username, url->mraw,
-- LXB_URL_MAP_USERINFO, false);
++- status = lxb_url_percent_encode_after_utf_8(begin,
++- tmp, &url->username, url->mraw, lxb_url_map,
++- LXB_URL_MAP_USERINFO, false);
+ - if (status != LXB_STATUS_OK) {
+ - lxb_url_parse_return(orig_data, buf, status);
+ - }
+ + tmp = (pswd != NULL) ? pswd - 1 : p;
+ + if (tmp > begin) {
+ + status = lxb_url_percent_encode_after_utf_8(begin, tmp,
+ + &url->username, url->mraw,
-+ LXB_URL_MAP_USERINFO, false);
+++ lxb_url_map, LXB_URL_MAP_USERINFO,
+++ false);
+ + if (status != LXB_STATUS_OK) {
+ + lxb_url_parse_return(orig_data, buf, status);
+ }
+ }
+
diff --cc ext/lexbor/patches/0018-Unicode-fixed-uninitialized-memory-in-IDNA-buffer-gr.patch
index 00000000000,177022aaa53..23c4cf1acc0
mode 000000,100644..100644
--- a/ext/lexbor/patches/0018-Unicode-fixed-uninitialized-memory-in-IDNA-buffer-gr.patch
+++ b/ext/lexbor/patches/0018-Unicode-fixed-uninitialized-memory-in-IDNA-buffer-gr.patch
@@@ -1,0 -1,81 +1,81 @@@
+ From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
+ From: Alexander Borisov <lex.borisov@gmail.com>
+ Date: Mon, 7 Sep 2026 22:11:39 +0300
-Subject: [PATCH 15/21] Unicode: fixed uninitialized memory in IDNA buffer
++Subject: [PATCH 18/24] Unicode: fixed uninitialized memory in IDNA buffer
+ growth.
+
+ When an IDNA buffer outgrew the stack allocation, the move to the heap
+ did not copy the existing contents. The converted domain could therefore
+ contain uninitialized heap data.
+
+ Fixed copying of the codepoint, ASCII and UTF-8 buffers.
+
+ Per report from Muhammad Daffa (@daffainfo).
+ ---
+ source/lexbor/unicode/idna.c | 28 +++++++++++++++++++---------
+ 1 file changed, 19 insertions(+), 9 deletions(-)
+
+ diff --git a/source/lexbor/unicode/idna.c b/source/lexbor/unicode/idna.c
+ index 754f6b2..b31ae33 100644
+ --- a/source/lexbor/unicode/idna.c
+ +++ b/source/lexbor/unicode/idna.c
+ @@ -117,12 +117,14 @@ lxb_unicode_idna_realloc(lxb_codepoint_t *buf, const lxb_codepoint_t *buffer,
+ lxb_codepoint_t *tmp;
+
+ nlen = ((*buf_end - buf) * 4) + len;
+ -
+ +
+ if (buf == buffer) {
+ tmp = lexbor_malloc(nlen * sizeof(lxb_codepoint_t));
+ if (tmp == NULL) {
+ return NULL;
+ }
+ +
+ + memcpy(tmp, buf, (*buf_p - buf) * sizeof(lxb_codepoint_t));
+ }
+ else {
+ tmp = lexbor_realloc(buf, nlen * sizeof(lxb_codepoint_t));
+ @@ -458,13 +460,17 @@ lxb_unicode_idna_ascii_puny_cb(const lxb_char_t *data, size_t length, void *ctx,
+
+ if (asc->buf == asc->buffer) {
+ tmp = lexbor_malloc(nlen);
+ + if (tmp == NULL) {
+ + return LXB_STATUS_ERROR_MEMORY_ALLOCATION;
+ + }
+ +
+ + memcpy(tmp, asc->buf, asc->p - asc->buf);
+ }
+ else {
+ tmp = lexbor_realloc(asc->buf, nlen);
+ - }
+ -
+ - if (tmp == NULL) {
+ - return LXB_STATUS_ERROR_MEMORY_ALLOCATION;
+ + if (tmp == NULL) {
+ + return LXB_STATUS_ERROR_MEMORY_ALLOCATION;
+ + }
+ }
+
+ asc->p = tmp + (asc->p - asc->buf);
+ @@ -711,13 +717,17 @@ lxb_unicode_idna_to_unicode_cb(const lxb_codepoint_t *part, size_t len,
+
+ if (asc->buf == asc->buffer) {
+ tmp = lexbor_malloc(nlen);
+ + if (tmp == NULL) {
+ + return LXB_STATUS_ERROR_MEMORY_ALLOCATION;
+ + }
+ +
+ + memcpy(tmp, asc->buf, asc->p - asc->buf);
+ }
+ else {
+ tmp = lexbor_realloc(asc->buf, nlen);
+ - }
+ -
+ - if (tmp == NULL) {
+ - return LXB_STATUS_ERROR_MEMORY_ALLOCATION;
+ + if (tmp == NULL) {
+ + return LXB_STATUS_ERROR_MEMORY_ALLOCATION;
+ + }
+ }
+
+ asc->p = tmp + (asc->p - asc->buf);
diff --cc ext/lexbor/patches/0019-URL-encode-opaque-path-spaces-before-query-and-fragm.patch
index 00000000000,4484e113231..4d833852199
mode 000000,100644..100644
--- a/ext/lexbor/patches/0019-URL-encode-opaque-path-spaces-before-query-and-fragm.patch
+++ b/ext/lexbor/patches/0019-URL-encode-opaque-path-spaces-before-query-and-fragm.patch
@@@ -1,0 -1,38 +1,38 @@@
+ From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
+ From: =?UTF-8?q?M=C3=A1t=C3=A9=20Kocsis?= <kocsismate@woohoolabs.com>
+ Date: Fri, 11 Sep 2026 21:58:45 +0200
-Subject: [PATCH 16/21] URL: encode opaque path spaces before query and
++Subject: [PATCH 19/24] URL: encode opaque path spaces before query and
+ fragment. (#405)
+ MIME-Version: 1.0
+ Content-Type: text/plain; charset=UTF-8
+ Content-Transfer-Encoding: 8bit
+
+ Percent-encode only the space immediately preceding a query or fragment delimiter, while preserving validation errors for every parsed space.
+
+ This way, Lexbor will correctly follow "If remaining starts with U+003F (?) or U+0023 (#), then append "%20" to url’s path." in the "opaque path state".
+ ---
+ source/lexbor/url/url.c | 11 +++++++++++
+ 1 file changed, 11 insertions(+)
+
+ diff --git a/source/lexbor/url/url.c b/source/lexbor/url/url.c
-index 654e2e6..98ee304 100644
++index a5bc3f0..1354f45 100644
+ --- a/source/lexbor/url/url.c
+ +++ b/source/lexbor/url/url.c
-@@ -2340,6 +2340,17 @@ again:
++@@ -2327,6 +2327,17 @@ again:
+ lxb_url_parse_return(orig_data, buf, status);
+ }
+
+ + /* Encode only the space immediately before a query or fragment. */
+ + if (p > begin && p[-1] == ' ') {
+ + tmp_str.length--;
+ + if (lexbor_str_append(&tmp_str, url->mraw,
+ + (const lxb_char_t *) "%20", 3) == NULL)
+ + {
+ + lxb_url_parse_return(orig_data, buf,
+ + LXB_STATUS_ERROR_MEMORY_ALLOCATION);
+ + }
+ + }
+ +
+ status = lxb_url_path_list_push(url, &tmp_str);
+ if (status != LXB_STATUS_OK) {
+ lxb_url_parse_return(orig_data, buf, status);
diff --cc ext/lexbor/patches/0020-URL-Fix-lxb_url_serialize_fragment-without-a-query-4.patch
index 00000000000,98139927410..a0eecf7234b
mode 000000,100644..100644
--- a/ext/lexbor/patches/0020-URL-Fix-lxb_url_serialize_fragment-without-a-query-4.patch
+++ b/ext/lexbor/patches/0020-URL-Fix-lxb_url_serialize_fragment-without-a-query-4.patch
@@@ -1,0 -1,25 +1,25 @@@
+ From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
+ From: =?UTF-8?q?Tim=20D=C3=BCsterhus?= <tim@tideways-gmbh.com>
+ Date: Fri, 11 Sep 2026 22:45:33 +0200
-Subject: [PATCH 17/21] URL: Fix `lxb_url_serialize_fragment()` without a query
++Subject: [PATCH 20/24] URL: Fix `lxb_url_serialize_fragment()` without a query
+ (#410)
+
+ The `lxb_url_serialize_fragment()` function checked if the query contained data
+ before correctly serializing the fragment. Check for the fragment instead.
+ ---
+ source/lexbor/url/url.c | 2 +-
+ 1 file changed, 1 insertion(+), 1 deletion(-)
+
+ diff --git a/source/lexbor/url/url.c b/source/lexbor/url/url.c
-index 98ee304..faf553b 100644
++index 1354f45..06bdd2e 100644
+ --- a/source/lexbor/url/url.c
+ +++ b/source/lexbor/url/url.c
-@@ -4915,7 +4915,7 @@ lxb_status_t
++@@ -4977,7 +4977,7 @@ lxb_status_t
+ lxb_url_serialize_fragment(const lxb_url_t *url,
+ lexbor_serialize_cb_f cb, void *ctx)
+ {
+ - if (url->query.data != NULL) {
+ + if (url->fragment.data != NULL) {
+ return cb(url->fragment.data, url->fragment.length, ctx);
+ }
+
diff --cc ext/lexbor/patches/0021-URL-expose-component-reset-functions.-415.patch
index 00000000000,c259de69dff..deaac125e82
mode 000000,100644..100644
--- a/ext/lexbor/patches/0021-URL-expose-component-reset-functions.-415.patch
+++ b/ext/lexbor/patches/0021-URL-expose-component-reset-functions.-415.patch
@@@ -1,0 -1,110 +1,110 @@@
+ From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
+ From: =?UTF-8?q?M=C3=A1t=C3=A9=20Kocsis?= <kocsismate@woohoolabs.com>
+ Date: Tue, 22 Sep 2026 22:20:28 +0200
-Subject: [PATCH 18/21] URL: expose component reset functions. (#415)
++Subject: [PATCH 21/24] URL: expose component reset functions. (#415)
+
+ * URL: expose component reset functions.
+
+ Make the path, host and fragment reset functions public and add lxb_url_query_set_null().
+
+ * Add documentation for the newly exposed functions
+ ---
+ source/lexbor/url/url.c | 14 ++++++++++---
+ source/lexbor/url/url.h | 45 +++++++++++++++++++++++++++++++++++++++++
+ 2 files changed, 56 insertions(+), 3 deletions(-)
+
+ diff --git a/source/lexbor/url/url.c b/source/lexbor/url/url.c
-index faf553b..9bfc2aa 100644
++index 06bdd2e..06325cd 100644
+ --- a/source/lexbor/url/url.c
+ +++ b/source/lexbor/url/url.c
-@@ -923,7 +923,7 @@ lxb_url_scheme_copy_special(const lxb_url_scheme_data_t *src,
++@@ -909,7 +909,7 @@ lxb_url_scheme_copy_special(const lxb_url_scheme_data_t *src,
+ return lxb_url_str_copy(&src->name, &dst->name, dst_mraw);
+ }
+
+ -static void
+ +void
+ lxb_url_path_set_null(lxb_url_t *url)
+ {
+ if (url->path.str.data == NULL) {
-@@ -1147,7 +1147,7 @@ lxb_url_host_destroy(lxb_url_host_t *host, lexbor_mraw_t *mraw)
++@@ -1133,7 +1133,7 @@ lxb_url_host_destroy(lxb_url_host_t *host, lexbor_mraw_t *mraw)
+ }
+ }
+
+ -static void
+ +void
+ lxb_url_host_set_empty(lxb_url_host_t *host, lexbor_mraw_t *mraw)
+ {
+ lxb_url_host_destroy(host, mraw);
-@@ -1197,7 +1197,15 @@ lxb_url_port_set(lxb_url_t *url, uint16_t port)
++@@ -1183,7 +1183,15 @@ lxb_url_port_set(lxb_url_t *url, uint16_t port)
+ url->has_port = true;
+ }
+
+ -static void
+ +void
+ +lxb_url_query_set_null(lxb_url_t *url)
+ +{
+ + if (url->query.data != NULL) {
+ + (void) lexbor_str_destroy(&url->query, url->mraw, false);
+ + }
+ +}
+ +
+ +void
+ lxb_url_fragment_set_null(lxb_url_t *url)
+ {
+ if (url->fragment.data != NULL) {
+ diff --git a/source/lexbor/url/url.h b/source/lexbor/url/url.h
-index 4ed3f32..b9e4973 100644
++index d2c9308..a68cf65 100644
+ --- a/source/lexbor/url/url.h
+ +++ b/source/lexbor/url/url.h
-@@ -763,6 +763,51 @@ LXB_API lxb_status_t
- lxb_url_search_params_serialize(lxb_url_search_params_t *search_params,
- lexbor_callback_f cb, void *ctx);
++@@ -894,6 +894,51 @@ lxb_url_search_params_serialize(lxb_url_search_params_t *search_params,
++ LXB_API bool
++ lxb_url_is_special(const lxb_url_t *url);
+
+ +/*
+ + * Reset the URL path to an empty list.
+ + *
+ + * Frees the path buffer using url->mraw, resets the segment count and clears
+ + * the opaque flag. Does nothing if the path buffer is already NULL.
+ + *
+ + * @param[in, out] URL object. Not NULL.
+ + */
+ +LXB_API void
+ +lxb_url_path_set_null(lxb_url_t *url);
+ +
+ +/*
+ + * Set the host to the empty host.
+ + *
+ + * Frees any domain or opaque host buffer using mraw and sets the host type
+ + * to LXB_URL_HOST_TYPE_EMPTY.
+ + *
+ + * @param[in, out] Host object. Not NULL.
+ + * @param[in] Memory object associated with the host. Not NULL.
+ + */
+ +LXB_API void
+ +lxb_url_host_set_empty(lxb_url_host_t *host, lexbor_mraw_t *mraw);
+ +
+ +/*
+ + * Set the URL query to null.
+ + *
+ + * Frees the query buffer using url->mraw. Does nothing if the query
+ + * is already null.
+ + *
+ + * @param[in, out] URL object. Not NULL.
+ + */
+ +LXB_API void
+ +lxb_url_query_set_null(lxb_url_t *url);
+ +
+ +/*
+ + * Set the URL fragment to null.
+ + *
+ + * Frees the fragment buffer using url->mraw. Does nothing if the
+ + * fragment is already null.
+ + *
+ + * @param[in, out] URL object. Not NULL.
+ + */
+ +LXB_API void
+ +lxb_url_fragment_set_null(lxb_url_t *url);
+ +
+ /*
+ * Inline functions.
+ */
diff --cc ext/lexbor/patches/0022-URL-Fix-parsing-of-query-and-fragment-after-a-dot-co.patch
index 00000000000,c2dd006b4b8..4f9607068b9
mode 000000,100644..100644
--- a/ext/lexbor/patches/0022-URL-Fix-parsing-of-query-and-fragment-after-a-dot-co.patch
+++ b/ext/lexbor/patches/0022-URL-Fix-parsing-of-query-and-fragment-after-a-dot-co.patch
@@@ -1,0 -1,416 +1,416 @@@
+ From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
+ From: =?UTF-8?q?Tim=20D=C3=BCsterhus?= <tim@tideways-gmbh.com>
+ Date: Sat, 26 Sep 2026 18:41:05 +0200
-Subject: [PATCH 19/21] URL: Fix parsing of query and fragment after a
++Subject: [PATCH 22/24] URL: Fix parsing of query and fragment after a
+ dot-component in path (#411)
+ MIME-Version: 1.0
+ Content-Type: text/plain; charset=UTF-8
+ Content-Transfer-Encoding: 8bit
+
+ * URL: fixed dot segment terminators in path parsing.
+
+ After a "." or ".." path segment the parser skipped the next code point
+ unconditionally. For "?" and "#" this lost the delimiter, so the query
+ or the fragment became part of the path: "https://example.com/..#frag"
+ was parsed as "https://example.com/frag". The fast path fix from #411
+ did not cover the slow path, which is taken after a code point that must
+ be percent-encoded, for example "https://example.com/café/..#frag".
+
+ Also fixed in the same code:
+ - "\" after a dot segment in a special URL did not report
+ invalid-reverse-solidus.
+ - The empty segment was lost in "//./c" and at the fast/slow path
+ handoff ("//é").
+ - path.length drifted after dot segments and underflowed after ".." at
+ the root, so a later ".." in the slow path did not shorten the path
+ ("/a/b/../../../c/é/../../x" gave "/c/x"). The file host and path
+ start states did not update it either.
+ - An invalid percent sequence in the fast path skipped the next two
+ code points, so a following "/", "\", "?" or "#" was lost ("/%?q" put
+ "?q" into the path). Such a segment is now handed to the slow path,
+ which checks "%" one code point at a time.
+
+ Per reports from Tim Düsterhus (@TimWolla) and @NickSdot.
+
+ This relates to #409 issue on GitHub.
+ This relates to #412 issue on GitHub.
+ This relates to #411 PR on GitHub.
+
+ * URL: added regression tests for dot segments in path.
+
+ Added tests for "." and ".." path segments followed by "?", "#", "\"
+ and the end of input, in both the fast and the slow path, for the
+ invalid-reverse-solidus validation error, and for path.length against
+ the serialized path. Also added tests for invalid percent sequences
+ before dot segments and delimiters, both in parsing and in the pathname
+ setter.
+
+ This relates to #409 issue on GitHub.
+ This relates to #412 issue on GitHub.
+ ---
+ source/lexbor/url/url.c | 200 +++++++++++++++++++++++-----------------
+ 1 file changed, 113 insertions(+), 87 deletions(-)
+
+ diff --git a/source/lexbor/url/url.c b/source/lexbor/url/url.c
-index 9bfc2aa..55fe11f 100644
++index 06325cd..8187c0f 100644
+ --- a/source/lexbor/url/url.c
+ +++ b/source/lexbor/url/url.c
-@@ -546,15 +546,15 @@ lxb_url_path_slow_path(lxb_url_parser_t *parser, lxb_url_t *url,
++@@ -532,15 +532,15 @@ lxb_url_path_slow_path(lxb_url_parser_t *parser, lxb_url_t *url,
+ const lxb_char_t *data, const lxb_char_t *end, bool bqs);
+
+ static lxb_status_t
+ -lxb_url_path_try_dot(lxb_url_t *url, const lxb_char_t **begin,
+ - const lxb_char_t **last, const lxb_char_t **start,
+ - const lxb_char_t *end, bool bqs);
+ +lxb_url_path_try_dot(lxb_url_parser_t *parser, lxb_url_t *url,
+ + const lxb_char_t **begin, const lxb_char_t **last,
+ + const lxb_char_t **start, const lxb_char_t *end, bool bqs);
+
+ static const lxb_char_t *
+ -lxb_url_path_dot_count(lxb_url_t *url, const lxb_char_t *p,
+ - const lxb_char_t *end, const lxb_char_t *sbuf_begin,
+ - lxb_char_t **sbuf, lxb_char_t **last, size_t *path_count,
+ - bool bqs);
+ +lxb_url_path_dot_count(lxb_url_parser_t *parser, lxb_url_t *url,
+ + const lxb_char_t *p, const lxb_char_t *end,
+ + const lxb_char_t *sbuf_begin, lxb_char_t **sbuf,
+ + lxb_char_t **last, size_t *path_count, bool bqs);
+
+ static void
+ lxb_url_path_fix_windows_drive(lxb_url_t *url, lxb_char_t *sbuf,
-@@ -990,23 +990,25 @@ lxb_url_path_shorten(lxb_url_t *url)
++@@ -976,23 +976,25 @@ lxb_url_path_shorten(lxb_url_t *url)
+ }
+ }
+
+ - if (url->path.str.data != NULL) {
+ - url->path.length -= 1;
+ + if (url->path.length == 0 || str->data == NULL) {
+ + return;
+ + }
+
+ - begin = str->data;
+ - p = begin + str->length;
+ + url->path.length -= 1;
+
+ - while (p > begin) {
+ - p -= 1;
+ + begin = str->data;
+ + p = begin + str->length;
+
+ - if (*p == '/') {
+ - *p = '\0';
+ - break;
+ - }
+ - }
+ + while (p > begin) {
+ + p -= 1;
+
+ - str->length = p - begin;
+ + if (*p == '/') {
+ + *p = '\0';
+ + break;
+ + }
+ }
+ +
+ + str->length = p - begin;
+ }
+
+ static lxb_status_t
-@@ -2146,6 +2148,8 @@ again:
++@@ -2133,6 +2135,8 @@ again:
+ if (status != LXB_STATUS_OK) {
+ lxb_url_parse_return(orig_data, buf, status);
+ }
+ +
+ + url->path.length += 1;
+ }
+ }
+ }
-@@ -2287,7 +2291,13 @@ again:
++@@ -2274,7 +2278,13 @@ again:
+ && url->host.type == LXB_URL_HOST_TYPE__UNDEF)
+ {
+ status = lxb_url_path_append(url, mp_str.data, mp_str.length);
+ - lxb_url_parse_return(orig_data, buf, status);
+ + if (status != LXB_STATUS_OK) {
+ + lxb_url_parse_return(orig_data, buf, status);
+ + }
+ +
+ + url->path.length += 1;
+ +
+ + lxb_url_parse_return(orig_data, buf, LXB_STATUS_OK);
+ }
+
+ lxb_url_parse_return(orig_data, buf, LXB_STATUS_OK);
-@@ -2535,13 +2545,8 @@ lxb_url_path_fast_path(lxb_url_parser_t *parser, lxb_url_t *url,
++@@ -2523,13 +2533,8 @@ lxb_url_path_fast_path(lxb_url_parser_t *parser, lxb_url_t *url,
+ || lexbor_str_res_map_hex[p[1]] == 0xff
+ || lexbor_str_res_map_hex[p[2]] == 0xff)
+ {
+ - status = lxb_url_log_append(parser, p,
+ - LXB_URL_ERROR_TYPE_INVALID_URL_UNIT);
+ - if (status != LXB_STATUS_OK) {
+ - return NULL;
+ - }
+ -
+ - p = (end - p < 3) ? end - 1 : p + 2;
+ + /* Reprocess the segment without skipping delimiters. */
+ + goto slow;
+ }
+ else if (p[1] == '2' && (p[2] == 'e' || p[2] == 'E')
+ && (p == begin
-@@ -2550,8 +2555,8 @@ lxb_url_path_fast_path(lxb_url_parser_t *parser, lxb_url_t *url,
++@@ -2538,8 +2543,8 @@ lxb_url_path_fast_path(lxb_url_parser_t *parser, lxb_url_t *url,
+ {
+ url->path.length = count;
+
+ - status = lxb_url_path_try_dot(url, &begin, &last,
+ - &p, end, bqs);
+ + status = lxb_url_path_try_dot(parser, url, &begin,
+ + &last, &p, end, bqs);
+ if (status != LXB_STATUS_OK) {
+ return NULL;
+ }
-@@ -2589,8 +2594,8 @@ lxb_url_path_fast_path(lxb_url_parser_t *parser, lxb_url_t *url,
++@@ -2577,8 +2582,8 @@ lxb_url_path_fast_path(lxb_url_parser_t *parser, lxb_url_t *url,
+ {
+ url->path.length = count;
+
+ - status = lxb_url_path_try_dot(url, &begin, &last,
+ - &p, end, bqs);
+ + status = lxb_url_path_try_dot(parser, url, &begin,
+ + &last, &p, end, bqs);
+ if (status != LXB_STATUS_OK) {
+ return NULL;
+ }
-@@ -2599,17 +2604,7 @@ lxb_url_path_fast_path(lxb_url_parser_t *parser, lxb_url_t *url,
++@@ -2587,17 +2592,7 @@ lxb_url_path_fast_path(lxb_url_parser_t *parser, lxb_url_t *url,
+ }
+ }
+ else {
+ - url->path.length = count;
+ -
+ - if (last - 1 > begin) {
+ - status = lxb_url_path_append(url, begin,
+ - (last - 1) - begin);
+ - if (status != LXB_STATUS_OK) {
+ - return NULL;
+ - }
+ - }
+ -
+ - return lxb_url_path_slow_path(parser, url, last, end, bqs);
+ + goto slow;
+ }
+ }
+ }
-@@ -2619,13 +2614,22 @@ lxb_url_path_fast_path(lxb_url_parser_t *parser, lxb_url_t *url,
++@@ -2607,13 +2602,22 @@ lxb_url_path_fast_path(lxb_url_parser_t *parser, lxb_url_t *url,
+ return NULL;
+ }
+
+ - if (count == 0 || p != begin) {
+ - count += 1;
+ - }
+ + url->path.length = count + 1;
+ +
+ + return p;
+ +
+ +slow:
+
+ url->path.length = count;
+
+ - return p;
+ + if (last > begin) {
+ + status = lxb_url_path_append(url, begin, (last - 1) - begin);
+ + if (status != LXB_STATUS_OK) {
+ + return NULL;
+ + }
+ + }
+ +
+ + return lxb_url_path_slow_path(parser, url, last, end, bqs);
+ }
+
+ /*
-@@ -2721,10 +2725,6 @@ lxb_url_path_slow_path(lxb_url_parser_t *parser, lxb_url_t *url,
++@@ -2709,10 +2713,6 @@ lxb_url_path_slow_path(lxb_url_parser_t *parser, lxb_url_t *url,
+
+ count += 1;
+ last = sbuf;
+ -
+ - if (p + 1 >= end) {
+ - count += 1;
+ - }
+ }
+ else if (c == '\\' && lxb_url_is_special(url)) {
+ status = lxb_url_log_append(parser, p,
-@@ -2742,16 +2742,8 @@ lxb_url_path_slow_path(lxb_url_parser_t *parser, lxb_url_t *url,
++@@ -2730,16 +2730,8 @@ lxb_url_path_slow_path(lxb_url_parser_t *parser, lxb_url_t *url,
+
+ count += 1;
+ last = sbuf;
+ -
+ - if (p + 1 >= end) {
+ - count += 1;
+ - }
+ }
+ else if ((c == '?' || c == '#') && bqs) {
+ - lxb_url_path_fix_windows_drive(url, last, sbuf, count);
+ -
+ - count += 1;
+ - last = sbuf;
+ break;
+ }
+ else if (lxb_url_map[c] & LXB_URL_MAP_PATH) {
-@@ -2771,11 +2763,19 @@ lxb_url_path_slow_path(lxb_url_parser_t *parser, lxb_url_t *url,
++@@ -2759,11 +2751,19 @@ lxb_url_path_slow_path(lxb_url_parser_t *parser, lxb_url_t *url,
+ }
+ else if (c == '.') {
+ if (last == sbuf) {
+ - tmp = lxb_url_path_dot_count(url, p, end, sbuf_begin,
+ + tmp = lxb_url_path_dot_count(parser, url, p, end, sbuf_begin,
+ &sbuf, &last, &count, bqs);
+ + if (tmp == NULL) {
+ + goto failed;
+ + }
+
+ if (tmp != p) {
+ - p = tmp + 1;
+ + /* Skip '/' or '\', but leave '?' and '#' to the loop. */
+ + if (tmp < end && *tmp != '?' && *tmp != '#') {
+ + tmp += 1;
+ + }
+ +
+ + p = tmp;
+ continue;
+ }
+ }
-@@ -2800,11 +2800,19 @@ lxb_url_path_slow_path(lxb_url_parser_t *parser, lxb_url_t *url,
++@@ -2788,11 +2788,19 @@ lxb_url_path_slow_path(lxb_url_parser_t *parser, lxb_url_t *url,
+ else if (p[1] == '2' && (p[2] == 'e' || p[2] == 'E')
+ && last == sbuf)
+ {
+ - tmp = lxb_url_path_dot_count(url, p, end, sbuf_begin,
+ + tmp = lxb_url_path_dot_count(parser, url, p, end, sbuf_begin,
+ &sbuf, &last, &count, bqs);
+ + if (tmp == NULL) {
+ + goto failed;
+ + }
+
+ if (tmp != p) {
+ - p = tmp + 1;
+ + /* Skip '/' or '\', but leave '?' and '#' to the loop. */
+ + if (tmp < end && *tmp != '?' && *tmp != '#') {
+ + tmp += 1;
+ + }
+ +
+ + p = tmp;
+ continue;
+ }
+ }
-@@ -2833,12 +2841,9 @@ lxb_url_path_slow_path(lxb_url_parser_t *parser, lxb_url_t *url,
++@@ -2821,12 +2829,9 @@ lxb_url_path_slow_path(lxb_url_parser_t *parser, lxb_url_t *url,
+ p += 1;
+ }
+
+ - if (count == 0 || last < sbuf) {
+ - lxb_url_path_fix_windows_drive(url, last, sbuf, count);
+ - count += 1;
+ - }
+ + lxb_url_path_fix_windows_drive(url, last, sbuf, count);
+
+ - url->path.length = count;
+ + url->path.length = count + 1;
+
+ status = lxb_url_path_append_wo_slash(url, sbuf_begin, sbuf - sbuf_begin);
+ if (status != LXB_STATUS_OK) {
-@@ -2861,13 +2866,12 @@ failed:
++@@ -2849,13 +2854,12 @@ failed:
+ }
+
+ static lxb_status_t
+ -lxb_url_path_try_dot(lxb_url_t *url, const lxb_char_t **begin,
+ - const lxb_char_t **last, const lxb_char_t **start,
+ - const lxb_char_t *end, bool bqs)
+ +lxb_url_path_try_dot(lxb_url_parser_t *parser, lxb_url_t *url,
+ + const lxb_char_t **begin, const lxb_char_t **last,
+ + const lxb_char_t **start, const lxb_char_t *end, bool bqs)
+ {
+ unsigned count;
+ lxb_char_t c;
+ - lexbor_str_t *str;
+ lxb_status_t status;
+ const lxb_char_t *p;
+
-@@ -2912,40 +2916,54 @@ lxb_url_path_try_dot(lxb_url_t *url, const lxb_char_t **begin,
++@@ -2900,40 +2904,54 @@ lxb_url_path_try_dot(lxb_url_t *url, const lxb_char_t **begin,
+ }
+ }
+
+ - if (p < end) {
+ - *start = p;
+ - *begin = p + 1;
+ - *last = *begin;
+ + if (count == 2) {
+ + lxb_url_path_shorten(url);
+ }
+ - else {
+ +
+ + if (p >= end) {
+ + /* The caller appends the trailing empty segment. */
+ *start = end - 1;
+ *begin = end;
+ *last = end;
+ +
+ + return LXB_STATUS_OK;
+ }
+
+ - if (count == 2) {
+ - lxb_url_path_shorten(url);
+ + if (*p == '?' || *p == '#') {
+ + /* The caller's loop handles the delimiter and the empty segment. */
+ + *start = p - 1;
+ + *begin = p;
+ + *last = p;
+ +
+ + return LXB_STATUS_OK;
+ }
+ - else if (count == 1) {
+ - str = &url->path.str;
+
+ - if (str->length > 0 && str->data[str->length - 1] == '/') {
+ - str->length -= 1;
+ - str->data[str->length] = '\0';
+ + if (*p == '\\') {
+ + status = lxb_url_log_append(parser, p,
+ + LXB_URL_ERROR_TYPE_INVALID_REVERSE_SOLIDUS);
+ + if (status != LXB_STATUS_OK) {
+ + return status;
+ }
+ }
+
+ + /* Skip '/' or '\'. */
+ +
+ + *start = p;
+ + *begin = p + 1;
+ + *last = *begin;
+ +
+ return LXB_STATUS_OK;
+ }
+
+ static const lxb_char_t *
+ -lxb_url_path_dot_count(lxb_url_t *url, const lxb_char_t *p,
+ - const lxb_char_t *end, const lxb_char_t *sbuf_begin,
+ - lxb_char_t **sbuf, lxb_char_t **last, size_t *path_count,
+ - bool bqs)
+ +lxb_url_path_dot_count(lxb_url_parser_t *parser, lxb_url_t *url,
+ + const lxb_char_t *p, const lxb_char_t *end,
+ + const lxb_char_t *sbuf_begin, lxb_char_t **sbuf,
+ + lxb_char_t **last, size_t *path_count, bool bqs)
+ {
+ unsigned count;
+ lxb_char_t c, *last_p;
+ + lxb_status_t status;
+ const lxb_char_t *begin;
+
+ count = 0;
-@@ -2982,6 +3000,14 @@ lxb_url_path_dot_count(lxb_url_t *url, const lxb_char_t *p,
++@@ -2970,6 +2988,14 @@ lxb_url_path_dot_count(lxb_url_t *url, const lxb_char_t *p,
+ return begin;
+ }
+
+ + if (p < end && *p == '\\') {
+ + status = lxb_url_log_append(parser, p,
+ + LXB_URL_ERROR_TYPE_INVALID_REVERSE_SOLIDUS);
+ + if (status != LXB_STATUS_OK) {
+ + return NULL;
+ + }
+ + }
+ +
+ if (url->scheme.type == LXB_URL_SCHEMEL_TYPE_FILE
+ && *path_count == 1
+ && lxb_url_normalized_windows_drive_letter(sbuf_begin + 1, *last - 1))
diff --cc ext/lexbor/patches/0023-URL-Keep-replacement-file-drive-paths-hierarchical-4.patch
index 00000000000,9871cb65064..2caee494342
mode 000000,100644..100644
--- a/ext/lexbor/patches/0023-URL-Keep-replacement-file-drive-paths-hierarchical-4.patch
+++ b/ext/lexbor/patches/0023-URL-Keep-replacement-file-drive-paths-hierarchical-4.patch
@@@ -1,0 -1,23 +1,23 @@@
+ From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
+ From: =?UTF-8?q?M=C3=A1t=C3=A9=20Kocsis?= <kocsismate@woohoolabs.com>
+ Date: Mon, 21 Sep 2026 20:01:43 +0200
-Subject: [PATCH 20/21] URL: Keep replacement file drive paths hierarchical
++Subject: [PATCH 23/24] URL: Keep replacement file drive paths hierarchical
+ (#423)
+
+ WHATWG file state (https://url.spec.whatwg.org/#file-state) step 4.4.3.2 resets the path to an empty list. Do not mark it opaque, as that makes subsequent pathname updates silently do nothing.
+ ---
+ source/lexbor/url/url.c | 1 -
+ 1 file changed, 1 deletion(-)
+
+ diff --git a/source/lexbor/url/url.c b/source/lexbor/url/url.c
-index 55fe11f..2231816 100644
++index 8187c0f..dcbc410 100644
+ --- a/source/lexbor/url/url.c
+ +++ b/source/lexbor/url/url.c
-@@ -2099,7 +2099,6 @@ again:
++@@ -2086,7 +2086,6 @@ again:
+ }
+
+ lxb_url_path_set_null(url);
+ - url->path.opaque = true;
+ }
+ }
+
diff --cc ext/lexbor/patches/0024-URL-normalize-output-encoding-for-percent-encoding.patch
index 00000000000,715d7c8241d..c0d192d24ea
mode 000000,100644..100644
--- a/ext/lexbor/patches/0024-URL-normalize-output-encoding-for-percent-encoding.patch
+++ b/ext/lexbor/patches/0024-URL-normalize-output-encoding-for-percent-encoding.patch
@@@ -1,0 -1,80 +1,96 @@@
+ From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
+ From: Alexander Borisov <lex.borisov@gmail.com>
+ Date: Thu, 24 Sep 2026 15:05:51 +0300
-Subject: [PATCH 21/21] URL: normalize output encoding for percent-encoding.
++Subject: [PATCH 24/24] URL: normalize output encoding for percent-encoding.
+
+ ---
+ source/lexbor/url/url.c | 34 +++++++++++++++++++++++++---------
- 1 file changed, 25 insertions(+), 9 deletions(-)
++ source/lexbor/url/url.h | 4 ++++
++ 2 files changed, 29 insertions(+), 9 deletions(-)
+
+ diff --git a/source/lexbor/url/url.c b/source/lexbor/url/url.c
-index 2231816..146f0bd 100644
++index dcbc410..c88f8e7 100644
+ --- a/source/lexbor/url/url.c
+ +++ b/source/lexbor/url/url.c
-@@ -1228,6 +1228,26 @@ lxb_url_encoding_init(const lxb_encoding_data_t *encoding,
++@@ -1214,6 +1214,26 @@ lxb_url_encoding_init(const lxb_encoding_data_t *encoding,
+ (void) lxb_encoding_encode_init_single(encode, encoding);
+ }
+
+ +/*
+ + * https://encoding.spec.whatwg.org/#get-an-output-encoding
+ + */
+ +lxb_inline lxb_encoding_t
+ +lxb_url_output_encoding(lxb_encoding_t encoding)
+ +{
+ + switch (encoding) {
+ + case LXB_ENCODING_DEFAULT:
+ + case LXB_ENCODING_AUTO:
+ + case LXB_ENCODING_UNDEFINED:
+ + case LXB_ENCODING_REPLACEMENT:
+ + case LXB_ENCODING_UTF_16BE:
+ + case LXB_ENCODING_UTF_16LE:
+ + return LXB_ENCODING_UTF_8;
+ +
+ + default:
+ + return encoding;
+ + }
+ +}
+ +
+ static bool
+ lxb_url_start_windows_drive_letter(const lxb_char_t *data,
+ const lxb_char_t *end)
-@@ -1372,12 +1392,7 @@ lxb_url_parse_basic_h(lxb_url_parser_t *parser, lxb_url_t *url,
++@@ -1358,12 +1378,7 @@ lxb_url_parse_basic_h(lxb_url_parser_t *parser, lxb_url_t *url,
+ state = override_state;
+ }
+
+ - if (encoding <= LXB_ENCODING_UNDEFINED
+ - || encoding == LXB_ENCODING_UTF_16BE
+ - || encoding == LXB_ENCODING_UTF_16LE)
+ - {
+ - encoding = LXB_ENCODING_UTF_8;
+ - }
+ + encoding = lxb_url_output_encoding(encoding);
+
+ enc = lxb_encoding_data(encoding);
+ if (enc == NULL) {
-@@ -3221,7 +3236,7 @@ lxb_url_percent_encode_after_encoding(const lxb_char_t *data,
++@@ -3222,7 +3237,7 @@ lxb_url_percent_encode_after_encoding(const lxb_char_t *data,
+ const lxb_char_t *buf_end = buf + sizeof(buffer);
+ static const lexbor_str_t esc_str = lexbor_str("%26%23");
+
+ - if (encoding->encoding == LXB_ENCODING_UTF_8) {
+ + if (lxb_url_output_encoding(encoding->encoding) == LXB_ENCODING_UTF_8) {
+ return lxb_url_percent_encode_after_utf_8(data, end, str, mraw,
- enmap, space_as_plus);
- }
-@@ -3256,13 +3271,14 @@ lxb_url_percent_encode_after_encoding(const lxb_char_t *data,
++ url_map, enmap,
++ space_as_plus);
++@@ -3258,13 +3273,14 @@ lxb_url_percent_encode_after_encoding(const lxb_char_t *data,
+ len = encoding->encode_single(&encode, &buf, buf_end, cp);
+
+ if (len < LXB_ENCODING_ENCODE_OK) {
+ - size = lexbor_conv_int64_to_data((int64_t) cp, buf, buf_end - buf);
+ + size = lexbor_conv_int64_to_data((int64_t) cp, buffer,
+ + sizeof(buffer));
+
+ if (lexbor_str_append(str, mraw, esc_str.data, esc_str.length) == NULL) {
+ return LXB_STATUS_ERROR_MEMORY_ALLOCATION;
+ }
+
+ - if (lexbor_str_append(str, mraw, buf, size) == NULL) {
+ + if (lexbor_str_append(str, mraw, buffer, size) == NULL) {
+ return LXB_STATUS_ERROR_MEMORY_ALLOCATION;
+ }
+
++diff --git a/source/lexbor/url/url.h b/source/lexbor/url/url.h
++index a68cf65..5ecc1da 100644
++--- a/source/lexbor/url/url.h
+++++ b/source/lexbor/url/url.h
++@@ -400,6 +400,10 @@ lxb_url_percent_encode_utf_8(const lxb_char_t *data, size_t length,
++ * hexadecimal digits. If a code point cannot be represented in the target
++ * encoding, its percent-encoded numeric character reference is appended.
++ *
+++ * The output encoding of the target encoding is used: UTF-16BE, UTF-16LE and
+++ * replacement are replaced with UTF-8, see
+++ * https://encoding.spec.whatwg.org/#get-an-output-encoding
+++ *
++ * If encoding is UTF-8, no conversion is performed. If space_as_plus is true,
++ * an encoded U+0020 SPACE is replaced with '+' before the map is checked.
++ * The input is expected to be valid UTF-8; the function does not validate it.
diff --cc ext/uri/tests/whatwg/builder/path_success_opaque_spaces_with_fragment.phpt
index a5cc306c07f,00000000000..b6d72abded4
mode 100644,000000..100644
--- a/ext/uri/tests/whatwg/builder/path_success_opaque_spaces_with_fragment.phpt
+++ b/ext/uri/tests/whatwg/builder/path_success_opaque_spaces_with_fragment.phpt
@@@ -1,59 -1,0 +1,59 @@@
+--TEST--
+Test Uri\WhatWg\UrlBuilder::setPath() - success - trailing spaces before a fragment
+--FILE--
+<?php
+
+$builder = new Uri\WhatWg\UrlBuilder();
+$builder->setScheme("foo");
+$builder->setPath("abc ");
+$builder->setFragment("f");
+$softErrors = [];
+$url = $builder->build(softErrors: $softErrors);
+
+var_dump($url->toAsciiString());
+var_dump($url);
+var_dump($softErrors);
+var_dump($url->equals(new Uri\WhatWg\Url($url->toAsciiString()), Uri\UriComparisonMode::IncludeFragment));
+
+?>
+--EXPECTF--
- string(11) "foo:abc #f"
++string(13) "foo:abc %20#f"
+object(Uri\WhatWg\Url)#%d (%d) {
+ ["scheme"]=>
+ string(3) "foo"
+ ["username"]=>
+ NULL
+ ["password"]=>
+ NULL
+ ["host"]=>
+ NULL
+ ["port"]=>
+ NULL
+ ["path"]=>
- string(5) "abc "
++ string(7) "abc %20"
+ ["query"]=>
+ NULL
+ ["fragment"]=>
+ string(1) "f"
+}
+array(2) {
+ [0]=>
+ object(Uri\WhatWg\UrlValidationError)#%d (%d) {
+ ["context"]=>
+ string(2) " #"
+ ["type"]=>
+ enum(Uri\WhatWg\UrlValidationErrorType::InvalidUrlUnit)
+ ["failure"]=>
+ bool(false)
+ }
+ [1]=>
+ object(Uri\WhatWg\UrlValidationError)#%d (%d) {
+ ["context"]=>
+ string(3) " #"
+ ["type"]=>
+ enum(Uri\WhatWg\UrlValidationErrorType::InvalidUrlUnit)
+ ["failure"]=>
+ bool(false)
+ }
+}
+bool(true)
diff --cc ext/uri/tests/whatwg/builder/path_success_opaque_spaces_with_query.phpt
index 4ffa9701e10,00000000000..d719feafc3d
mode 100644,000000..100644
--- a/ext/uri/tests/whatwg/builder/path_success_opaque_spaces_with_query.phpt
+++ b/ext/uri/tests/whatwg/builder/path_success_opaque_spaces_with_query.phpt
@@@ -1,59 -1,0 +1,59 @@@
+--TEST--
+Test Uri\WhatWg\UrlBuilder::setPath() - success - trailing spaces before a query
+--FILE--
+<?php
+
+$builder = new Uri\WhatWg\UrlBuilder();
+$builder->setScheme("foo");
+$builder->setPath("abc ");
+$builder->setQuery("q");
+$softErrors = [];
+$url = $builder->build(softErrors: $softErrors);
+
+var_dump($url->toAsciiString());
+var_dump($url);
+var_dump($softErrors);
+var_dump($url->equals(new Uri\WhatWg\Url($url->toAsciiString())));
+
+?>
+--EXPECTF--
- string(11) "foo:abc ?q"
++string(13) "foo:abc %20?q"
+object(Uri\WhatWg\Url)#%d (%d) {
+ ["scheme"]=>
+ string(3) "foo"
+ ["username"]=>
+ NULL
+ ["password"]=>
+ NULL
+ ["host"]=>
+ NULL
+ ["port"]=>
+ NULL
+ ["path"]=>
- string(5) "abc "
++ string(7) "abc %20"
+ ["query"]=>
+ string(1) "q"
+ ["fragment"]=>
+ NULL
+}
+array(2) {
+ [0]=>
+ object(Uri\WhatWg\UrlValidationError)#%d (%d) {
+ ["context"]=>
+ string(2) " ?"
+ ["type"]=>
+ enum(Uri\WhatWg\UrlValidationErrorType::InvalidUrlUnit)
+ ["failure"]=>
+ bool(false)
+ }
+ [1]=>
+ object(Uri\WhatWg\UrlValidationError)#%d (%d) {
+ ["context"]=>
+ string(3) " ?"
+ ["type"]=>
+ enum(Uri\WhatWg\UrlValidationErrorType::InvalidUrlUnit)
+ ["failure"]=>
+ bool(false)
+ }
+}
+bool(true)