Commit beee5444d01 for php
commit beee5444d016b4699d1ef48376c9ef7659adbdd0
Author: Alexandre Daubois <alex.daubois@gmail.com>
Date: Mon Oct 5 14:18:04 2026 +0200
lexbor: Merge upstream WHATWG URL and IDNA fixes
lexbor/lexbor@859f100
lexbor/lexbor@a36e09a
lexbor/lexbor@b0f7412
lexbor/lexbor@1b215a8
lexbor/lexbor@385afff
lexbor/lexbor@e6c068f
lexbor/lexbor@327a8b6
lexbor/lexbor@917742f
lexbor/lexbor@e89c258
Only the source/lexbor/url part of lexbor/lexbor@327a8b6 and
lexbor/lexbor@e89c258 is carried: the first squash also contains unrelated
HTML tree construction changes, and the second documents a percent-encoder
API that this branch does not bundle.
diff --git a/NEWS b/NEWS
index 1039ef9a31e..a42642d601e 100644
--- a/NEWS
+++ b/NEWS
@@ -87,6 +87,15 @@ PHP NEWS
. Merge patches lexbor/lexbor@8a14bc0 and lexbor/lexbor@f67ce4b, fixing a
heap buffer overflow in :lexbor-contains() parsing and buffer overflows
in malformed decode replay. (alexandre-daubois)
+ . Merge patches lexbor/lexbor@859f100, lexbor/lexbor@a36e09a,
+ lexbor/lexbor@b0f7412, lexbor/lexbor@1b215a8, lexbor/lexbor@385afff,
+ lexbor/lexbor@e6c068f, lexbor/lexbor@327a8b6, lexbor/lexbor@917742f and
+ lexbor/lexbor@e89c258, fixing dropped usernames containing an at sign,
+ uninitialized memory in IDNA buffer growth, the encoding of a space
+ before a query or fragment in an opaque path, a query or fragment lost
+ after a dot segment in a path, replacement file drive paths, the output
+ encoding used for percent-encoding, the URLSearchParams tail pointer and
+ fragment serialization without a query. (alexandre-daubois)
- MBString:
. Fixed bug GH-23106 (mb_strpos() reads past the end of a haystack ending in
diff --git a/ext/lexbor/lexbor/unicode/idna.c b/ext/lexbor/lexbor/unicode/idna.c
index 754f6b2026d..b31ae3338b9 100644
--- a/ext/lexbor/lexbor/unicode/idna.c
+++ b/ext/lexbor/lexbor/unicode/idna.c
@@ -117,12 +117,14 @@ lxb_unicode_idna_realloc(lxb_codepoint_t *buf, const lxb_codepoint_t *buffer,
lxb_codepoint_t *tmp;
nlen = ((*buf_end - buf) * 4) + len;
-
+
if (buf == buffer) {
tmp = lexbor_malloc(nlen * sizeof(lxb_codepoint_t));
if (tmp == NULL) {
return NULL;
}
+
+ memcpy(tmp, buf, (*buf_p - buf) * sizeof(lxb_codepoint_t));
}
else {
tmp = lexbor_realloc(buf, nlen * sizeof(lxb_codepoint_t));
@@ -458,13 +460,17 @@ lxb_unicode_idna_ascii_puny_cb(const lxb_char_t *data, size_t length, void *ctx,
if (asc->buf == asc->buffer) {
tmp = lexbor_malloc(nlen);
+ if (tmp == NULL) {
+ return LXB_STATUS_ERROR_MEMORY_ALLOCATION;
+ }
+
+ memcpy(tmp, asc->buf, asc->p - asc->buf);
}
else {
tmp = lexbor_realloc(asc->buf, nlen);
- }
-
- if (tmp == NULL) {
- return LXB_STATUS_ERROR_MEMORY_ALLOCATION;
+ if (tmp == NULL) {
+ return LXB_STATUS_ERROR_MEMORY_ALLOCATION;
+ }
}
asc->p = tmp + (asc->p - asc->buf);
@@ -711,13 +717,17 @@ lxb_unicode_idna_to_unicode_cb(const lxb_codepoint_t *part, size_t len,
if (asc->buf == asc->buffer) {
tmp = lexbor_malloc(nlen);
+ if (tmp == NULL) {
+ return LXB_STATUS_ERROR_MEMORY_ALLOCATION;
+ }
+
+ memcpy(tmp, asc->buf, asc->p - asc->buf);
}
else {
tmp = lexbor_realloc(asc->buf, nlen);
- }
-
- if (tmp == NULL) {
- return LXB_STATUS_ERROR_MEMORY_ALLOCATION;
+ if (tmp == NULL) {
+ return LXB_STATUS_ERROR_MEMORY_ALLOCATION;
+ }
}
asc->p = tmp + (asc->p - asc->buf);
diff --git a/ext/lexbor/lexbor/url/url.c b/ext/lexbor/lexbor/url/url.c
index de19239936a..146f0bda292 100644
--- a/ext/lexbor/lexbor/url/url.c
+++ b/ext/lexbor/lexbor/url/url.c
@@ -546,15 +546,15 @@ lxb_url_path_slow_path(lxb_url_parser_t *parser, lxb_url_t *url,
const lxb_char_t *data, const lxb_char_t *end, bool bqs);
static lxb_status_t
-lxb_url_path_try_dot(lxb_url_t *url, const lxb_char_t **begin,
- const lxb_char_t **last, const lxb_char_t **start,
- const lxb_char_t *end, bool bqs);
+lxb_url_path_try_dot(lxb_url_parser_t *parser, lxb_url_t *url,
+ const lxb_char_t **begin, const lxb_char_t **last,
+ const lxb_char_t **start, const lxb_char_t *end, bool bqs);
static const lxb_char_t *
-lxb_url_path_dot_count(lxb_url_t *url, const lxb_char_t *p,
- const lxb_char_t *end, const lxb_char_t *sbuf_begin,
- lxb_char_t **sbuf, lxb_char_t **last, size_t *path_count,
- bool bqs);
+lxb_url_path_dot_count(lxb_url_parser_t *parser, lxb_url_t *url,
+ const lxb_char_t *p, const lxb_char_t *end,
+ const lxb_char_t *sbuf_begin, lxb_char_t **sbuf,
+ lxb_char_t **last, size_t *path_count, bool bqs);
static void
lxb_url_path_fix_windows_drive(lxb_url_t *url, lxb_char_t *sbuf,
@@ -923,7 +923,7 @@ lxb_url_scheme_copy_special(const lxb_url_scheme_data_t *src,
return lxb_url_str_copy(&src->name, &dst->name, dst_mraw);
}
-static void
+void
lxb_url_path_set_null(lxb_url_t *url)
{
if (url->path.str.data == NULL) {
@@ -990,23 +990,25 @@ lxb_url_path_shorten(lxb_url_t *url)
}
}
- if (url->path.str.data != NULL) {
- url->path.length -= 1;
+ if (url->path.length == 0 || str->data == NULL) {
+ return;
+ }
- begin = str->data;
- p = begin + str->length;
+ url->path.length -= 1;
- while (p > begin) {
- p -= 1;
+ begin = str->data;
+ p = begin + str->length;
- if (*p == '/') {
- *p = '\0';
- break;
- }
- }
+ while (p > begin) {
+ p -= 1;
- str->length = p - begin;
+ if (*p == '/') {
+ *p = '\0';
+ break;
+ }
}
+
+ str->length = p - begin;
}
static lxb_status_t
@@ -1147,7 +1149,7 @@ lxb_url_host_destroy(lxb_url_host_t *host, lexbor_mraw_t *mraw)
}
}
-static void
+void
lxb_url_host_set_empty(lxb_url_host_t *host, lexbor_mraw_t *mraw)
{
lxb_url_host_destroy(host, mraw);
@@ -1197,7 +1199,15 @@ lxb_url_port_set(lxb_url_t *url, uint16_t port)
url->has_port = true;
}
-static void
+void
+lxb_url_query_set_null(lxb_url_t *url)
+{
+ if (url->query.data != NULL) {
+ (void) lexbor_str_destroy(&url->query, url->mraw, false);
+ }
+}
+
+void
lxb_url_fragment_set_null(lxb_url_t *url)
{
if (url->fragment.data != NULL) {
@@ -1218,6 +1228,26 @@ lxb_url_encoding_init(const lxb_encoding_data_t *encoding,
(void) lxb_encoding_encode_init_single(encode, encoding);
}
+/*
+ * https://encoding.spec.whatwg.org/#get-an-output-encoding
+ */
+lxb_inline lxb_encoding_t
+lxb_url_output_encoding(lxb_encoding_t encoding)
+{
+ switch (encoding) {
+ case LXB_ENCODING_DEFAULT:
+ case LXB_ENCODING_AUTO:
+ case LXB_ENCODING_UNDEFINED:
+ case LXB_ENCODING_REPLACEMENT:
+ case LXB_ENCODING_UTF_16BE:
+ case LXB_ENCODING_UTF_16LE:
+ return LXB_ENCODING_UTF_8;
+
+ default:
+ return encoding;
+ }
+}
+
static bool
lxb_url_start_windows_drive_letter(const lxb_char_t *data,
const lxb_char_t *end)
@@ -1362,12 +1392,7 @@ lxb_url_parse_basic_h(lxb_url_parser_t *parser, lxb_url_t *url,
state = override_state;
}
- if (encoding <= LXB_ENCODING_UNDEFINED
- || encoding == LXB_ENCODING_UTF_16BE
- || encoding == LXB_ENCODING_UTF_16LE)
- {
- encoding = LXB_ENCODING_UTF_8;
- }
+ encoding = lxb_url_output_encoding(encoding);
enc = lxb_encoding_data(encoding);
if (enc == NULL) {
@@ -1753,16 +1778,13 @@ lxb_url_parse_basic_h(lxb_url_parser_t *parser, lxb_url_t *url,
break;
}
- if (pswd == NULL || !at_sign) {
- tmp = (pswd != NULL) ? pswd - 1 : p;
-
- if (tmp > begin) {
- status = lxb_url_percent_encode_after_utf_8(begin, tmp,
- &url->username, url->mraw,
- LXB_URL_MAP_USERINFO, false);
- if (status != LXB_STATUS_OK) {
- lxb_url_parse_return(orig_data, buf, status);
- }
+ tmp = (pswd != NULL) ? pswd - 1 : p;
+ if (tmp > begin) {
+ status = lxb_url_percent_encode_after_utf_8(begin, tmp,
+ &url->username, url->mraw,
+ LXB_URL_MAP_USERINFO, false);
+ if (status != LXB_STATUS_OK) {
+ lxb_url_parse_return(orig_data, buf, status);
}
}
@@ -2092,7 +2114,6 @@ lxb_url_parse_basic_h(lxb_url_parser_t *parser, lxb_url_t *url,
}
lxb_url_path_set_null(url);
- url->path.opaque = true;
}
}
@@ -2141,6 +2162,8 @@ lxb_url_parse_basic_h(lxb_url_parser_t *parser, lxb_url_t *url,
if (status != LXB_STATUS_OK) {
lxb_url_parse_return(orig_data, buf, status);
}
+
+ url->path.length += 1;
}
}
}
@@ -2282,7 +2305,13 @@ lxb_url_parse_basic_h(lxb_url_parser_t *parser, lxb_url_t *url,
&& url->host.type == LXB_URL_HOST_TYPE__UNDEF)
{
status = lxb_url_path_append(url, mp_str.data, mp_str.length);
- lxb_url_parse_return(orig_data, buf, status);
+ if (status != LXB_STATUS_OK) {
+ lxb_url_parse_return(orig_data, buf, status);
+ }
+
+ url->path.length += 1;
+
+ lxb_url_parse_return(orig_data, buf, LXB_STATUS_OK);
}
lxb_url_parse_return(orig_data, buf, LXB_STATUS_OK);
@@ -2343,6 +2372,17 @@ lxb_url_parse_basic_h(lxb_url_parser_t *parser, lxb_url_t *url,
lxb_url_parse_return(orig_data, buf, status);
}
+ /* Encode only the space immediately before a query or fragment. */
+ if (p > begin && p[-1] == ' ') {
+ tmp_str.length--;
+ if (lexbor_str_append(&tmp_str, url->mraw,
+ (const lxb_char_t *) "%20", 3) == NULL)
+ {
+ lxb_url_parse_return(orig_data, buf,
+ LXB_STATUS_ERROR_MEMORY_ALLOCATION);
+ }
+ }
+
status = lxb_url_path_list_push(url, &tmp_str);
if (status != LXB_STATUS_OK) {
lxb_url_parse_return(orig_data, buf, status);
@@ -2519,13 +2559,8 @@ lxb_url_path_fast_path(lxb_url_parser_t *parser, lxb_url_t *url,
|| lexbor_str_res_map_hex[p[1]] == 0xff
|| lexbor_str_res_map_hex[p[2]] == 0xff)
{
- status = lxb_url_log_append(parser, p,
- LXB_URL_ERROR_TYPE_INVALID_URL_UNIT);
- if (status != LXB_STATUS_OK) {
- return NULL;
- }
-
- p = (end - p < 3) ? end - 1 : p + 2;
+ /* Reprocess the segment without skipping delimiters. */
+ goto slow;
}
else if (p[1] == '2' && (p[2] == 'e' || p[2] == 'E')
&& (p == begin
@@ -2534,8 +2569,8 @@ lxb_url_path_fast_path(lxb_url_parser_t *parser, lxb_url_t *url,
{
url->path.length = count;
- status = lxb_url_path_try_dot(url, &begin, &last,
- &p, end, bqs);
+ status = lxb_url_path_try_dot(parser, url, &begin,
+ &last, &p, end, bqs);
if (status != LXB_STATUS_OK) {
return NULL;
}
@@ -2573,8 +2608,8 @@ lxb_url_path_fast_path(lxb_url_parser_t *parser, lxb_url_t *url,
{
url->path.length = count;
- status = lxb_url_path_try_dot(url, &begin, &last,
- &p, end, bqs);
+ status = lxb_url_path_try_dot(parser, url, &begin,
+ &last, &p, end, bqs);
if (status != LXB_STATUS_OK) {
return NULL;
}
@@ -2583,17 +2618,7 @@ lxb_url_path_fast_path(lxb_url_parser_t *parser, lxb_url_t *url,
}
}
else {
- url->path.length = count;
-
- if (last - 1 > begin) {
- status = lxb_url_path_append(url, begin,
- (last - 1) - begin);
- if (status != LXB_STATUS_OK) {
- return NULL;
- }
- }
-
- return lxb_url_path_slow_path(parser, url, last, end, bqs);
+ goto slow;
}
}
}
@@ -2603,13 +2628,22 @@ lxb_url_path_fast_path(lxb_url_parser_t *parser, lxb_url_t *url,
return NULL;
}
- if (count == 0 || p != begin) {
- count += 1;
- }
+ url->path.length = count + 1;
+
+ return p;
+
+slow:
url->path.length = count;
- return p;
+ if (last > begin) {
+ status = lxb_url_path_append(url, begin, (last - 1) - begin);
+ if (status != LXB_STATUS_OK) {
+ return NULL;
+ }
+ }
+
+ return lxb_url_path_slow_path(parser, url, last, end, bqs);
}
/*
@@ -2705,10 +2739,6 @@ lxb_url_path_slow_path(lxb_url_parser_t *parser, lxb_url_t *url,
count += 1;
last = sbuf;
-
- if (p + 1 >= end) {
- count += 1;
- }
}
else if (c == '\\' && lxb_url_is_special(url)) {
status = lxb_url_log_append(parser, p,
@@ -2726,16 +2756,8 @@ lxb_url_path_slow_path(lxb_url_parser_t *parser, lxb_url_t *url,
count += 1;
last = sbuf;
-
- if (p + 1 >= end) {
- count += 1;
- }
}
else if ((c == '?' || c == '#') && bqs) {
- lxb_url_path_fix_windows_drive(url, last, sbuf, count);
-
- count += 1;
- last = sbuf;
break;
}
else if (lxb_url_map[c] & LXB_URL_MAP_PATH) {
@@ -2755,11 +2777,19 @@ lxb_url_path_slow_path(lxb_url_parser_t *parser, lxb_url_t *url,
}
else if (c == '.') {
if (last == sbuf) {
- tmp = lxb_url_path_dot_count(url, p, end, sbuf_begin,
+ tmp = lxb_url_path_dot_count(parser, url, p, end, sbuf_begin,
&sbuf, &last, &count, bqs);
+ if (tmp == NULL) {
+ goto failed;
+ }
if (tmp != p) {
- p = tmp + 1;
+ /* Skip '/' or '\', but leave '?' and '#' to the loop. */
+ if (tmp < end && *tmp != '?' && *tmp != '#') {
+ tmp += 1;
+ }
+
+ p = tmp;
continue;
}
}
@@ -2784,11 +2814,19 @@ lxb_url_path_slow_path(lxb_url_parser_t *parser, lxb_url_t *url,
else if (p[1] == '2' && (p[2] == 'e' || p[2] == 'E')
&& last == sbuf)
{
- tmp = lxb_url_path_dot_count(url, p, end, sbuf_begin,
+ tmp = lxb_url_path_dot_count(parser, url, p, end, sbuf_begin,
&sbuf, &last, &count, bqs);
+ if (tmp == NULL) {
+ goto failed;
+ }
if (tmp != p) {
- p = tmp + 1;
+ /* Skip '/' or '\', but leave '?' and '#' to the loop. */
+ if (tmp < end && *tmp != '?' && *tmp != '#') {
+ tmp += 1;
+ }
+
+ p = tmp;
continue;
}
}
@@ -2817,12 +2855,9 @@ lxb_url_path_slow_path(lxb_url_parser_t *parser, lxb_url_t *url,
p += 1;
}
- if (count == 0 || last < sbuf) {
- lxb_url_path_fix_windows_drive(url, last, sbuf, count);
- count += 1;
- }
+ lxb_url_path_fix_windows_drive(url, last, sbuf, count);
- url->path.length = count;
+ url->path.length = count + 1;
status = lxb_url_path_append_wo_slash(url, sbuf_begin, sbuf - sbuf_begin);
if (status != LXB_STATUS_OK) {
@@ -2845,13 +2880,12 @@ lxb_url_path_slow_path(lxb_url_parser_t *parser, lxb_url_t *url,
}
static lxb_status_t
-lxb_url_path_try_dot(lxb_url_t *url, const lxb_char_t **begin,
- const lxb_char_t **last, const lxb_char_t **start,
- const lxb_char_t *end, bool bqs)
+lxb_url_path_try_dot(lxb_url_parser_t *parser, lxb_url_t *url,
+ const lxb_char_t **begin, const lxb_char_t **last,
+ const lxb_char_t **start, const lxb_char_t *end, bool bqs)
{
unsigned count;
lxb_char_t c;
- lexbor_str_t *str;
lxb_status_t status;
const lxb_char_t *p;
@@ -2896,40 +2930,54 @@ lxb_url_path_try_dot(lxb_url_t *url, const lxb_char_t **begin,
}
}
- if (p < end) {
- *start = p;
- *begin = p + 1;
- *last = *begin;
+ if (count == 2) {
+ lxb_url_path_shorten(url);
}
- else {
+
+ if (p >= end) {
+ /* The caller appends the trailing empty segment. */
*start = end - 1;
*begin = end;
*last = end;
+
+ return LXB_STATUS_OK;
}
- if (count == 2) {
- lxb_url_path_shorten(url);
+ if (*p == '?' || *p == '#') {
+ /* The caller's loop handles the delimiter and the empty segment. */
+ *start = p - 1;
+ *begin = p;
+ *last = p;
+
+ return LXB_STATUS_OK;
}
- else if (count == 1) {
- str = &url->path.str;
- if (str->length > 0 && str->data[str->length - 1] == '/') {
- str->length -= 1;
- str->data[str->length] = '\0';
+ if (*p == '\\') {
+ status = lxb_url_log_append(parser, p,
+ LXB_URL_ERROR_TYPE_INVALID_REVERSE_SOLIDUS);
+ if (status != LXB_STATUS_OK) {
+ return status;
}
}
+ /* Skip '/' or '\'. */
+
+ *start = p;
+ *begin = p + 1;
+ *last = *begin;
+
return LXB_STATUS_OK;
}
static const lxb_char_t *
-lxb_url_path_dot_count(lxb_url_t *url, const lxb_char_t *p,
- const lxb_char_t *end, const lxb_char_t *sbuf_begin,
- lxb_char_t **sbuf, lxb_char_t **last, size_t *path_count,
- bool bqs)
+lxb_url_path_dot_count(lxb_url_parser_t *parser, lxb_url_t *url,
+ const lxb_char_t *p, const lxb_char_t *end,
+ const lxb_char_t *sbuf_begin, lxb_char_t **sbuf,
+ lxb_char_t **last, size_t *path_count, bool bqs)
{
unsigned count;
lxb_char_t c, *last_p;
+ lxb_status_t status;
const lxb_char_t *begin;
count = 0;
@@ -2966,6 +3014,14 @@ lxb_url_path_dot_count(lxb_url_t *url, const lxb_char_t *p,
return begin;
}
+ if (p < end && *p == '\\') {
+ status = lxb_url_log_append(parser, p,
+ LXB_URL_ERROR_TYPE_INVALID_REVERSE_SOLIDUS);
+ if (status != LXB_STATUS_OK) {
+ return NULL;
+ }
+ }
+
if (url->scheme.type == LXB_URL_SCHEMEL_TYPE_FILE
&& *path_count == 1
&& lxb_url_normalized_windows_drive_letter(sbuf_begin + 1, *last - 1))
@@ -3180,7 +3236,7 @@ lxb_url_percent_encode_after_encoding(const lxb_char_t *data,
const lxb_char_t *buf_end = buf + sizeof(buffer);
static const lexbor_str_t esc_str = lexbor_str("%26%23");
- if (encoding->encoding == LXB_ENCODING_UTF_8) {
+ if (lxb_url_output_encoding(encoding->encoding) == LXB_ENCODING_UTF_8) {
return lxb_url_percent_encode_after_utf_8(data, end, str, mraw,
enmap, space_as_plus);
}
@@ -3215,13 +3271,14 @@ lxb_url_percent_encode_after_encoding(const lxb_char_t *data,
len = encoding->encode_single(&encode, &buf, buf_end, cp);
if (len < LXB_ENCODING_ENCODE_OK) {
- size = lexbor_conv_int64_to_data((int64_t) cp, buf, buf_end - buf);
+ size = lexbor_conv_int64_to_data((int64_t) cp, buffer,
+ sizeof(buffer));
if (lexbor_str_append(str, mraw, esc_str.data, esc_str.length) == NULL) {
return LXB_STATUS_ERROR_MEMORY_ALLOCATION;
}
- if (lexbor_str_append(str, mraw, buf, size) == NULL) {
+ if (lexbor_str_append(str, mraw, buffer, size) == NULL) {
return LXB_STATUS_ERROR_MEMORY_ALLOCATION;
}
@@ -4907,7 +4964,7 @@ lxb_status_t
lxb_url_serialize_fragment(const lxb_url_t *url,
lexbor_serialize_cb_f cb, void *ctx)
{
- if (url->query.data != NULL) {
+ if (url->fragment.data != NULL) {
return cb(url->fragment.data, url->fragment.length, ctx);
}
@@ -5106,6 +5163,8 @@ lxb_url_search_params_parse(lxb_url_search_params_t *search_params,
return status;
}
+ last = entry;
+
lexbor_str_init(&entry->value, mraw, 0);
if (entry->value.data == NULL) {
return LXB_STATUS_ERROR_MEMORY_ALLOCATION;
diff --git a/ext/lexbor/lexbor/url/url.h b/ext/lexbor/lexbor/url/url.h
index 4ed3f32aa64..b9e4973a674 100644
--- a/ext/lexbor/lexbor/url/url.h
+++ b/ext/lexbor/lexbor/url/url.h
@@ -763,6 +763,51 @@ LXB_API lxb_status_t
lxb_url_search_params_serialize(lxb_url_search_params_t *search_params,
lexbor_callback_f cb, void *ctx);
+/*
+ * Reset the URL path to an empty list.
+ *
+ * Frees the path buffer using url->mraw, resets the segment count and clears
+ * the opaque flag. Does nothing if the path buffer is already NULL.
+ *
+ * @param[in, out] URL object. Not NULL.
+ */
+LXB_API void
+lxb_url_path_set_null(lxb_url_t *url);
+
+/*
+ * Set the host to the empty host.
+ *
+ * Frees any domain or opaque host buffer using mraw and sets the host type
+ * to LXB_URL_HOST_TYPE_EMPTY.
+ *
+ * @param[in, out] Host object. Not NULL.
+ * @param[in] Memory object associated with the host. Not NULL.
+ */
+LXB_API void
+lxb_url_host_set_empty(lxb_url_host_t *host, lexbor_mraw_t *mraw);
+
+/*
+ * Set the URL query to null.
+ *
+ * Frees the query buffer using url->mraw. Does nothing if the query
+ * is already null.
+ *
+ * @param[in, out] URL object. Not NULL.
+ */
+LXB_API void
+lxb_url_query_set_null(lxb_url_t *url);
+
+/*
+ * Set the URL fragment to null.
+ *
+ * Frees the fragment buffer using url->mraw. Does nothing if the
+ * fragment is already null.
+ *
+ * @param[in, out] URL object. Not NULL.
+ */
+LXB_API void
+lxb_url_fragment_set_null(lxb_url_t *url);
+
/*
* Inline functions.
*/
diff --git a/ext/lexbor/patches/0001-Expose-line-and-column-information-for-use-in-PHP.patch b/ext/lexbor/patches/0001-Expose-line-and-column-information-for-use-in-PHP.patch
index 9aec14cca5d..56079940029 100644
--- a/ext/lexbor/patches/0001-Expose-line-and-column-information-for-use-in-PHP.patch
+++ b/ext/lexbor/patches/0001-Expose-line-and-column-information-for-use-in-PHP.patch
@@ -1,7 +1,7 @@
From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
From: Niels Dossche <7771979+nielsdos@users.noreply.github.com>
Date: Sat, 26 Aug 2023 15:08:59 +0200
-Subject: [PATCH 01/12] Expose line and column information for use in PHP
+Subject: [PATCH 01/21] Expose line and column information for use in PHP
---
source/lexbor/dom/interfaces/node.h | 2 ++
diff --git a/ext/lexbor/patches/0002-Track-implied-added-nodes-for-options-use-in-PHP.patch b/ext/lexbor/patches/0002-Track-implied-added-nodes-for-options-use-in-PHP.patch
index 9f4da029446..8dc8cb984d2 100644
--- a/ext/lexbor/patches/0002-Track-implied-added-nodes-for-options-use-in-PHP.patch
+++ b/ext/lexbor/patches/0002-Track-implied-added-nodes-for-options-use-in-PHP.patch
@@ -1,7 +1,7 @@
From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
From: Niels Dossche <7771979+nielsdos@users.noreply.github.com>
Date: Mon, 14 Aug 2023 20:18:51 +0200
-Subject: [PATCH 02/12] Track implied added nodes for options use in PHP
+Subject: [PATCH 02/21] Track implied added nodes for options use in PHP
---
source/lexbor/html/tree.h | 3 +++
diff --git a/ext/lexbor/patches/0003-Patch-utilities-and-data-structure-to-be-able-to-gen.patch b/ext/lexbor/patches/0003-Patch-utilities-and-data-structure-to-be-able-to-gen.patch
index fe7068d9bda..f93d9fe8f86 100644
--- a/ext/lexbor/patches/0003-Patch-utilities-and-data-structure-to-be-able-to-gen.patch
+++ b/ext/lexbor/patches/0003-Patch-utilities-and-data-structure-to-be-able-to-gen.patch
@@ -1,7 +1,7 @@
From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
From: Niels Dossche <7771979+nielsdos@users.noreply.github.com>
Date: Thu, 24 Aug 2023 22:57:48 +0200
-Subject: [PATCH 03/12] Patch utilities and data structure to be able to
+Subject: [PATCH 03/21] Patch utilities and data structure to be able to
generate smaller lookup tables
Changed the generation script to check if everything fits in 32-bits.
diff --git a/ext/lexbor/patches/0004-Remove-unused-upper-case-tag-static-data.patch b/ext/lexbor/patches/0004-Remove-unused-upper-case-tag-static-data.patch
index 35482388531..35bec95e6b9 100644
--- a/ext/lexbor/patches/0004-Remove-unused-upper-case-tag-static-data.patch
+++ b/ext/lexbor/patches/0004-Remove-unused-upper-case-tag-static-data.patch
@@ -1,7 +1,7 @@
From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
From: Niels Dossche <7771979+nielsdos@users.noreply.github.com>
Date: Wed, 29 Nov 2023 21:26:47 +0100
-Subject: [PATCH 04/12] Remove unused upper case tag static data
+Subject: [PATCH 04/21] Remove unused upper case tag static data
---
source/lexbor/tag/res.h | 2 ++
diff --git a/ext/lexbor/patches/0005-Shrink-size-of-static-binary-search-tree.patch b/ext/lexbor/patches/0005-Shrink-size-of-static-binary-search-tree.patch
index 8e9b9524104..68f2d4d379e 100644
--- a/ext/lexbor/patches/0005-Shrink-size-of-static-binary-search-tree.patch
+++ b/ext/lexbor/patches/0005-Shrink-size-of-static-binary-search-tree.patch
@@ -1,7 +1,7 @@
From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
From: Niels Dossche <7771979+nielsdos@users.noreply.github.com>
Date: Wed, 29 Nov 2023 21:29:31 +0100
-Subject: [PATCH 05/12] Shrink size of static binary search tree
+Subject: [PATCH 05/21] Shrink size of static binary search tree
This also makes it more efficient on the data cache.
---
diff --git a/ext/lexbor/patches/0006-Patch-out-unused-CSS-style-code.patch b/ext/lexbor/patches/0006-Patch-out-unused-CSS-style-code.patch
index 0f4e4cd8661..5d63d17116f 100644
--- a/ext/lexbor/patches/0006-Patch-out-unused-CSS-style-code.patch
+++ b/ext/lexbor/patches/0006-Patch-out-unused-CSS-style-code.patch
@@ -1,7 +1,7 @@
From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
From: Niels Dossche <7771979+nielsdos@users.noreply.github.com>
Date: Sun, 7 Jan 2024 21:59:28 +0100
-Subject: [PATCH 06/12] Patch out unused CSS style code
+Subject: [PATCH 06/21] Patch out unused CSS style code
---
source/lexbor/css/rule.h | 2 ++
diff --git a/ext/lexbor/patches/0007-URL-fixed-setters-for-empty-hosts.patch b/ext/lexbor/patches/0007-URL-fixed-setters-for-empty-hosts.patch
index 44f0f918458..3e69d07f95c 100644
--- a/ext/lexbor/patches/0007-URL-fixed-setters-for-empty-hosts.patch
+++ b/ext/lexbor/patches/0007-URL-fixed-setters-for-empty-hosts.patch
@@ -1,7 +1,7 @@
From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
From: Alexander Borisov <lex.borisov@gmail.com>
Date: Fri, 26 Jun 2026 18:55:56 +0300
-Subject: [PATCH 07/12] URL: fixed setters for empty hosts.
+Subject: [PATCH 07/21] URL: fixed setters for empty hosts.
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
diff --git a/ext/lexbor/patches/0008-URL-fixed-uninitialized-memory-in-the-path-buffer-gr.patch b/ext/lexbor/patches/0008-URL-fixed-uninitialized-memory-in-the-path-buffer-gr.patch
index df1f6a1c2ac..3f7d7e1eab3 100644
--- a/ext/lexbor/patches/0008-URL-fixed-uninitialized-memory-in-the-path-buffer-gr.patch
+++ b/ext/lexbor/patches/0008-URL-fixed-uninitialized-memory-in-the-path-buffer-gr.patch
@@ -1,7 +1,7 @@
From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
From: Alexander Borisov <lex.borisov@gmail.com>
Date: Fri, 5 Jun 2026 22:13:32 +0300
-Subject: [PATCH 08/12] URL: fixed uninitialized memory in the path buffer
+Subject: [PATCH 08/21] URL: fixed uninitialized memory in the path buffer
growth.
When a path was long enough to outgrow the on-stack buffer, the first
diff --git a/ext/lexbor/patches/0009-Fix-parsing-for-URL-containing-empty-host-and-userin.patch b/ext/lexbor/patches/0009-Fix-parsing-for-URL-containing-empty-host-and-userin.patch
index 9abcd37fe3f..c94e76622a4 100644
--- a/ext/lexbor/patches/0009-Fix-parsing-for-URL-containing-empty-host-and-userin.patch
+++ b/ext/lexbor/patches/0009-Fix-parsing-for-URL-containing-empty-host-and-userin.patch
@@ -1,7 +1,7 @@
From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?M=C3=A1t=C3=A9=20Kocsis?= <kocsismate@woohoolabs.com>
Date: Thu, 9 Jul 2026 21:51:05 +0200
-Subject: [PATCH 09/12] Fix parsing for URL containing empty host and userinfo
+Subject: [PATCH 09/21] Fix parsing for URL containing empty host and userinfo
The returned error code (LXB_URL_ERROR_TYPE_INVALID_CREDENTIALS) apparently contradicts the specification:
diff --git a/ext/lexbor/patches/0010-Percent-encode-the-caret-in-the-path.patch b/ext/lexbor/patches/0010-Percent-encode-the-caret-in-the-path.patch
index 48777d5de6d..69156c42159 100644
--- a/ext/lexbor/patches/0010-Percent-encode-the-caret-in-the-path.patch
+++ b/ext/lexbor/patches/0010-Percent-encode-the-caret-in-the-path.patch
@@ -1,7 +1,7 @@
From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?M=C3=A1t=C3=A9=20Kocsis?= <kocsismate@woohoolabs.com>
Date: Fri, 10 Jul 2026 22:31:16 +0200
-Subject: [PATCH 10/12] Percent-encode the caret in the path
+Subject: [PATCH 10/21] Percent-encode the caret in the path
The caret (^) is part of the path percent-encode set:
diff --git a/ext/lexbor/patches/0011-CSS-fixed-heap-buffer-overflow-in-lexbor-contains-pa.patch b/ext/lexbor/patches/0011-CSS-fixed-heap-buffer-overflow-in-lexbor-contains-pa.patch
index 3928daa7eab..32d52517acb 100644
--- a/ext/lexbor/patches/0011-CSS-fixed-heap-buffer-overflow-in-lexbor-contains-pa.patch
+++ b/ext/lexbor/patches/0011-CSS-fixed-heap-buffer-overflow-in-lexbor-contains-pa.patch
@@ -1,7 +1,7 @@
From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
From: Alexander Borisov <lex.borisov@gmail.com>
Date: Fri, 5 Jun 2026 22:34:23 +0300
-Subject: [PATCH 11/12] CSS: fixed heap buffer overflow in :lexbor-contains()
+Subject: [PATCH 11/21] CSS: fixed heap buffer overflow in :lexbor-contains()
parsing.
The contains string buffer was allocated by the size of the string
diff --git a/ext/lexbor/patches/0012-Encoding-fixed-buffer-overflows-in-malformed-decode-.patch b/ext/lexbor/patches/0012-Encoding-fixed-buffer-overflows-in-malformed-decode-.patch
index 32f3aff4ee7..6214cfbda10 100644
--- a/ext/lexbor/patches/0012-Encoding-fixed-buffer-overflows-in-malformed-decode-.patch
+++ b/ext/lexbor/patches/0012-Encoding-fixed-buffer-overflows-in-malformed-decode-.patch
@@ -1,7 +1,7 @@
From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
From: Alexander Borisov <lex.borisov@gmail.com>
Date: Wed, 10 Jun 2026 19:50:10 +0300
-Subject: [PATCH 12/12] Encoding: fixed buffer overflows in malformed decode
+Subject: [PATCH 12/21] Encoding: fixed buffer overflows in malformed decode
replay.
Fixed out-of-bounds writes in buffering decoders when replacement output
diff --git a/ext/lexbor/patches/0013-URL-fixed-tail-pointer-in-URLSearchParams-for-delimi.patch b/ext/lexbor/patches/0013-URL-fixed-tail-pointer-in-URLSearchParams-for-delimi.patch
new file mode 100644
index 00000000000..89adb8132fc
--- /dev/null
+++ b/ext/lexbor/patches/0013-URL-fixed-tail-pointer-in-URLSearchParams-for-delimi.patch
@@ -0,0 +1,29 @@
+From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
+From: Alexander Borisov <lex.borisov@gmail.com>
+Date: Fri, 5 Jun 2026 21:46:26 +0300
+Subject: [PATCH 13/21] URL: fixed tail pointer in URLSearchParams for
+ delimiter-free query.
+
+When a query had a single token without '=' or '&' (e.g. "?abc"), the
+internal tail pointer wasn't updated, so a later append() could lose the
+added parameter (and write through a stale pointer). Fixed by keeping the
+tail pointer in sync.
+
+Per report from Xiansheng Cao (@HMF2021)
+---
+ source/lexbor/url/url.c | 2 ++
+ 1 file changed, 2 insertions(+)
+
+diff --git a/source/lexbor/url/url.c b/source/lexbor/url/url.c
+index de19239..fcae2d6 100644
+--- a/source/lexbor/url/url.c
++++ b/source/lexbor/url/url.c
+@@ -5106,6 +5106,8 @@ lxb_url_search_params_parse(lxb_url_search_params_t *search_params,
+ return status;
+ }
+
++ last = entry;
++
+ lexbor_str_init(&entry->value, mraw, 0);
+ if (entry->value.data == NULL) {
+ return LXB_STATUS_ERROR_MEMORY_ALLOCATION;
diff --git a/ext/lexbor/patches/0014-URL-Fix-parsing-of-usernames-containing.patch b/ext/lexbor/patches/0014-URL-Fix-parsing-of-usernames-containing.patch
new file mode 100644
index 00000000000..9aa3b403994
--- /dev/null
+++ b/ext/lexbor/patches/0014-URL-Fix-parsing-of-usernames-containing.patch
@@ -0,0 +1,38 @@
+From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
+From: =?UTF-8?q?Tim=20D=C3=BCsterhus?= <tim@bastelstu.be>
+Date: Fri, 31 Jul 2026 20:56:18 +0200
+Subject: [PATCH 14/21] URL: Fix parsing of usernames containing `@`
+
+Fixes lexbor/lexbor#399.
+---
+ source/lexbor/url/url.c | 17 +++++++----------
+ 1 file changed, 7 insertions(+), 10 deletions(-)
+
+diff --git a/source/lexbor/url/url.c b/source/lexbor/url/url.c
+index fcae2d6..654e2e6 100644
+--- a/source/lexbor/url/url.c
++++ b/source/lexbor/url/url.c
+@@ -1753,16 +1753,13 @@ again:
+ break;
+ }
+
+- if (pswd == NULL || !at_sign) {
+- tmp = (pswd != NULL) ? pswd - 1 : p;
+-
+- if (tmp > begin) {
+- status = lxb_url_percent_encode_after_utf_8(begin, tmp,
+- &url->username, url->mraw,
+- LXB_URL_MAP_USERINFO, false);
+- if (status != LXB_STATUS_OK) {
+- lxb_url_parse_return(orig_data, buf, status);
+- }
++ tmp = (pswd != NULL) ? pswd - 1 : p;
++ if (tmp > begin) {
++ status = lxb_url_percent_encode_after_utf_8(begin, tmp,
++ &url->username, url->mraw,
++ LXB_URL_MAP_USERINFO, false);
++ if (status != LXB_STATUS_OK) {
++ lxb_url_parse_return(orig_data, buf, status);
+ }
+ }
+
diff --git a/ext/lexbor/patches/0015-Unicode-fixed-uninitialized-memory-in-IDNA-buffer-gr.patch b/ext/lexbor/patches/0015-Unicode-fixed-uninitialized-memory-in-IDNA-buffer-gr.patch
new file mode 100644
index 00000000000..177022aaa53
--- /dev/null
+++ b/ext/lexbor/patches/0015-Unicode-fixed-uninitialized-memory-in-IDNA-buffer-gr.patch
@@ -0,0 +1,81 @@
+From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
+From: Alexander Borisov <lex.borisov@gmail.com>
+Date: Mon, 7 Sep 2026 22:11:39 +0300
+Subject: [PATCH 15/21] Unicode: fixed uninitialized memory in IDNA buffer
+ growth.
+
+When an IDNA buffer outgrew the stack allocation, the move to the heap
+did not copy the existing contents. The converted domain could therefore
+contain uninitialized heap data.
+
+Fixed copying of the codepoint, ASCII and UTF-8 buffers.
+
+Per report from Muhammad Daffa (@daffainfo).
+---
+ source/lexbor/unicode/idna.c | 28 +++++++++++++++++++---------
+ 1 file changed, 19 insertions(+), 9 deletions(-)
+
+diff --git a/source/lexbor/unicode/idna.c b/source/lexbor/unicode/idna.c
+index 754f6b2..b31ae33 100644
+--- a/source/lexbor/unicode/idna.c
++++ b/source/lexbor/unicode/idna.c
+@@ -117,12 +117,14 @@ lxb_unicode_idna_realloc(lxb_codepoint_t *buf, const lxb_codepoint_t *buffer,
+ lxb_codepoint_t *tmp;
+
+ nlen = ((*buf_end - buf) * 4) + len;
+-
++
+ if (buf == buffer) {
+ tmp = lexbor_malloc(nlen * sizeof(lxb_codepoint_t));
+ if (tmp == NULL) {
+ return NULL;
+ }
++
++ memcpy(tmp, buf, (*buf_p - buf) * sizeof(lxb_codepoint_t));
+ }
+ else {
+ tmp = lexbor_realloc(buf, nlen * sizeof(lxb_codepoint_t));
+@@ -458,13 +460,17 @@ lxb_unicode_idna_ascii_puny_cb(const lxb_char_t *data, size_t length, void *ctx,
+
+ if (asc->buf == asc->buffer) {
+ tmp = lexbor_malloc(nlen);
++ if (tmp == NULL) {
++ return LXB_STATUS_ERROR_MEMORY_ALLOCATION;
++ }
++
++ memcpy(tmp, asc->buf, asc->p - asc->buf);
+ }
+ else {
+ tmp = lexbor_realloc(asc->buf, nlen);
+- }
+-
+- if (tmp == NULL) {
+- return LXB_STATUS_ERROR_MEMORY_ALLOCATION;
++ if (tmp == NULL) {
++ return LXB_STATUS_ERROR_MEMORY_ALLOCATION;
++ }
+ }
+
+ asc->p = tmp + (asc->p - asc->buf);
+@@ -711,13 +717,17 @@ lxb_unicode_idna_to_unicode_cb(const lxb_codepoint_t *part, size_t len,
+
+ if (asc->buf == asc->buffer) {
+ tmp = lexbor_malloc(nlen);
++ if (tmp == NULL) {
++ return LXB_STATUS_ERROR_MEMORY_ALLOCATION;
++ }
++
++ memcpy(tmp, asc->buf, asc->p - asc->buf);
+ }
+ else {
+ tmp = lexbor_realloc(asc->buf, nlen);
+- }
+-
+- if (tmp == NULL) {
+- return LXB_STATUS_ERROR_MEMORY_ALLOCATION;
++ if (tmp == NULL) {
++ return LXB_STATUS_ERROR_MEMORY_ALLOCATION;
++ }
+ }
+
+ asc->p = tmp + (asc->p - asc->buf);
diff --git a/ext/lexbor/patches/0016-URL-encode-opaque-path-spaces-before-query-and-fragm.patch b/ext/lexbor/patches/0016-URL-encode-opaque-path-spaces-before-query-and-fragm.patch
new file mode 100644
index 00000000000..4484e113231
--- /dev/null
+++ b/ext/lexbor/patches/0016-URL-encode-opaque-path-spaces-before-query-and-fragm.patch
@@ -0,0 +1,38 @@
+From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
+From: =?UTF-8?q?M=C3=A1t=C3=A9=20Kocsis?= <kocsismate@woohoolabs.com>
+Date: Fri, 11 Sep 2026 21:58:45 +0200
+Subject: [PATCH 16/21] URL: encode opaque path spaces before query and
+ fragment. (#405)
+MIME-Version: 1.0
+Content-Type: text/plain; charset=UTF-8
+Content-Transfer-Encoding: 8bit
+
+Percent-encode only the space immediately preceding a query or fragment delimiter, while preserving validation errors for every parsed space.
+
+This way, Lexbor will correctly follow "If remaining starts with U+003F (?) or U+0023 (#), then append "%20" to url’s path." in the "opaque path state".
+---
+ source/lexbor/url/url.c | 11 +++++++++++
+ 1 file changed, 11 insertions(+)
+
+diff --git a/source/lexbor/url/url.c b/source/lexbor/url/url.c
+index 654e2e6..98ee304 100644
+--- a/source/lexbor/url/url.c
++++ b/source/lexbor/url/url.c
+@@ -2340,6 +2340,17 @@ again:
+ lxb_url_parse_return(orig_data, buf, status);
+ }
+
++ /* Encode only the space immediately before a query or fragment. */
++ if (p > begin && p[-1] == ' ') {
++ tmp_str.length--;
++ if (lexbor_str_append(&tmp_str, url->mraw,
++ (const lxb_char_t *) "%20", 3) == NULL)
++ {
++ lxb_url_parse_return(orig_data, buf,
++ LXB_STATUS_ERROR_MEMORY_ALLOCATION);
++ }
++ }
++
+ status = lxb_url_path_list_push(url, &tmp_str);
+ if (status != LXB_STATUS_OK) {
+ lxb_url_parse_return(orig_data, buf, status);
diff --git a/ext/lexbor/patches/0017-URL-Fix-lxb_url_serialize_fragment-without-a-query-4.patch b/ext/lexbor/patches/0017-URL-Fix-lxb_url_serialize_fragment-without-a-query-4.patch
new file mode 100644
index 00000000000..98139927410
--- /dev/null
+++ b/ext/lexbor/patches/0017-URL-Fix-lxb_url_serialize_fragment-without-a-query-4.patch
@@ -0,0 +1,25 @@
+From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
+From: =?UTF-8?q?Tim=20D=C3=BCsterhus?= <tim@tideways-gmbh.com>
+Date: Fri, 11 Sep 2026 22:45:33 +0200
+Subject: [PATCH 17/21] URL: Fix `lxb_url_serialize_fragment()` without a query
+ (#410)
+
+The `lxb_url_serialize_fragment()` function checked if the query contained data
+before correctly serializing the fragment. Check for the fragment instead.
+---
+ source/lexbor/url/url.c | 2 +-
+ 1 file changed, 1 insertion(+), 1 deletion(-)
+
+diff --git a/source/lexbor/url/url.c b/source/lexbor/url/url.c
+index 98ee304..faf553b 100644
+--- a/source/lexbor/url/url.c
++++ b/source/lexbor/url/url.c
+@@ -4915,7 +4915,7 @@ lxb_status_t
+ lxb_url_serialize_fragment(const lxb_url_t *url,
+ lexbor_serialize_cb_f cb, void *ctx)
+ {
+- if (url->query.data != NULL) {
++ if (url->fragment.data != NULL) {
+ return cb(url->fragment.data, url->fragment.length, ctx);
+ }
+
diff --git a/ext/lexbor/patches/0018-URL-expose-component-reset-functions.-415.patch b/ext/lexbor/patches/0018-URL-expose-component-reset-functions.-415.patch
new file mode 100644
index 00000000000..c259de69dff
--- /dev/null
+++ b/ext/lexbor/patches/0018-URL-expose-component-reset-functions.-415.patch
@@ -0,0 +1,110 @@
+From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
+From: =?UTF-8?q?M=C3=A1t=C3=A9=20Kocsis?= <kocsismate@woohoolabs.com>
+Date: Tue, 22 Sep 2026 22:20:28 +0200
+Subject: [PATCH 18/21] URL: expose component reset functions. (#415)
+
+* URL: expose component reset functions.
+
+Make the path, host and fragment reset functions public and add lxb_url_query_set_null().
+
+* Add documentation for the newly exposed functions
+---
+ source/lexbor/url/url.c | 14 ++++++++++---
+ source/lexbor/url/url.h | 45 +++++++++++++++++++++++++++++++++++++++++
+ 2 files changed, 56 insertions(+), 3 deletions(-)
+
+diff --git a/source/lexbor/url/url.c b/source/lexbor/url/url.c
+index faf553b..9bfc2aa 100644
+--- a/source/lexbor/url/url.c
++++ b/source/lexbor/url/url.c
+@@ -923,7 +923,7 @@ lxb_url_scheme_copy_special(const lxb_url_scheme_data_t *src,
+ return lxb_url_str_copy(&src->name, &dst->name, dst_mraw);
+ }
+
+-static void
++void
+ lxb_url_path_set_null(lxb_url_t *url)
+ {
+ if (url->path.str.data == NULL) {
+@@ -1147,7 +1147,7 @@ lxb_url_host_destroy(lxb_url_host_t *host, lexbor_mraw_t *mraw)
+ }
+ }
+
+-static void
++void
+ lxb_url_host_set_empty(lxb_url_host_t *host, lexbor_mraw_t *mraw)
+ {
+ lxb_url_host_destroy(host, mraw);
+@@ -1197,7 +1197,15 @@ lxb_url_port_set(lxb_url_t *url, uint16_t port)
+ url->has_port = true;
+ }
+
+-static void
++void
++lxb_url_query_set_null(lxb_url_t *url)
++{
++ if (url->query.data != NULL) {
++ (void) lexbor_str_destroy(&url->query, url->mraw, false);
++ }
++}
++
++void
+ lxb_url_fragment_set_null(lxb_url_t *url)
+ {
+ if (url->fragment.data != NULL) {
+diff --git a/source/lexbor/url/url.h b/source/lexbor/url/url.h
+index 4ed3f32..b9e4973 100644
+--- a/source/lexbor/url/url.h
++++ b/source/lexbor/url/url.h
+@@ -763,6 +763,51 @@ LXB_API lxb_status_t
+ lxb_url_search_params_serialize(lxb_url_search_params_t *search_params,
+ lexbor_callback_f cb, void *ctx);
+
++/*
++ * Reset the URL path to an empty list.
++ *
++ * Frees the path buffer using url->mraw, resets the segment count and clears
++ * the opaque flag. Does nothing if the path buffer is already NULL.
++ *
++ * @param[in, out] URL object. Not NULL.
++ */
++LXB_API void
++lxb_url_path_set_null(lxb_url_t *url);
++
++/*
++ * Set the host to the empty host.
++ *
++ * Frees any domain or opaque host buffer using mraw and sets the host type
++ * to LXB_URL_HOST_TYPE_EMPTY.
++ *
++ * @param[in, out] Host object. Not NULL.
++ * @param[in] Memory object associated with the host. Not NULL.
++ */
++LXB_API void
++lxb_url_host_set_empty(lxb_url_host_t *host, lexbor_mraw_t *mraw);
++
++/*
++ * Set the URL query to null.
++ *
++ * Frees the query buffer using url->mraw. Does nothing if the query
++ * is already null.
++ *
++ * @param[in, out] URL object. Not NULL.
++ */
++LXB_API void
++lxb_url_query_set_null(lxb_url_t *url);
++
++/*
++ * Set the URL fragment to null.
++ *
++ * Frees the fragment buffer using url->mraw. Does nothing if the
++ * fragment is already null.
++ *
++ * @param[in, out] URL object. Not NULL.
++ */
++LXB_API void
++lxb_url_fragment_set_null(lxb_url_t *url);
++
+ /*
+ * Inline functions.
+ */
diff --git a/ext/lexbor/patches/0019-URL-Fix-parsing-of-query-and-fragment-after-a-dot-co.patch b/ext/lexbor/patches/0019-URL-Fix-parsing-of-query-and-fragment-after-a-dot-co.patch
new file mode 100644
index 00000000000..c2dd006b4b8
--- /dev/null
+++ b/ext/lexbor/patches/0019-URL-Fix-parsing-of-query-and-fragment-after-a-dot-co.patch
@@ -0,0 +1,416 @@
+From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
+From: =?UTF-8?q?Tim=20D=C3=BCsterhus?= <tim@tideways-gmbh.com>
+Date: Sat, 26 Sep 2026 18:41:05 +0200
+Subject: [PATCH 19/21] URL: Fix parsing of query and fragment after a
+ dot-component in path (#411)
+MIME-Version: 1.0
+Content-Type: text/plain; charset=UTF-8
+Content-Transfer-Encoding: 8bit
+
+* URL: fixed dot segment terminators in path parsing.
+
+After a "." or ".." path segment the parser skipped the next code point
+unconditionally. For "?" and "#" this lost the delimiter, so the query
+or the fragment became part of the path: "https://example.com/..#frag"
+was parsed as "https://example.com/frag". The fast path fix from #411
+did not cover the slow path, which is taken after a code point that must
+be percent-encoded, for example "https://example.com/café/..#frag".
+
+Also fixed in the same code:
+- "\" after a dot segment in a special URL did not report
+ invalid-reverse-solidus.
+- The empty segment was lost in "//./c" and at the fast/slow path
+ handoff ("//é").
+- path.length drifted after dot segments and underflowed after ".." at
+ the root, so a later ".." in the slow path did not shorten the path
+ ("/a/b/../../../c/é/../../x" gave "/c/x"). The file host and path
+ start states did not update it either.
+- An invalid percent sequence in the fast path skipped the next two
+ code points, so a following "/", "\", "?" or "#" was lost ("/%?q" put
+ "?q" into the path). Such a segment is now handed to the slow path,
+ which checks "%" one code point at a time.
+
+Per reports from Tim Düsterhus (@TimWolla) and @NickSdot.
+
+This relates to #409 issue on GitHub.
+This relates to #412 issue on GitHub.
+This relates to #411 PR on GitHub.
+
+* URL: added regression tests for dot segments in path.
+
+Added tests for "." and ".." path segments followed by "?", "#", "\"
+and the end of input, in both the fast and the slow path, for the
+invalid-reverse-solidus validation error, and for path.length against
+the serialized path. Also added tests for invalid percent sequences
+before dot segments and delimiters, both in parsing and in the pathname
+setter.
+
+This relates to #409 issue on GitHub.
+This relates to #412 issue on GitHub.
+---
+ source/lexbor/url/url.c | 200 +++++++++++++++++++++++-----------------
+ 1 file changed, 113 insertions(+), 87 deletions(-)
+
+diff --git a/source/lexbor/url/url.c b/source/lexbor/url/url.c
+index 9bfc2aa..55fe11f 100644
+--- a/source/lexbor/url/url.c
++++ b/source/lexbor/url/url.c
+@@ -546,15 +546,15 @@ lxb_url_path_slow_path(lxb_url_parser_t *parser, lxb_url_t *url,
+ const lxb_char_t *data, const lxb_char_t *end, bool bqs);
+
+ static lxb_status_t
+-lxb_url_path_try_dot(lxb_url_t *url, const lxb_char_t **begin,
+- const lxb_char_t **last, const lxb_char_t **start,
+- const lxb_char_t *end, bool bqs);
++lxb_url_path_try_dot(lxb_url_parser_t *parser, lxb_url_t *url,
++ const lxb_char_t **begin, const lxb_char_t **last,
++ const lxb_char_t **start, const lxb_char_t *end, bool bqs);
+
+ static const lxb_char_t *
+-lxb_url_path_dot_count(lxb_url_t *url, const lxb_char_t *p,
+- const lxb_char_t *end, const lxb_char_t *sbuf_begin,
+- lxb_char_t **sbuf, lxb_char_t **last, size_t *path_count,
+- bool bqs);
++lxb_url_path_dot_count(lxb_url_parser_t *parser, lxb_url_t *url,
++ const lxb_char_t *p, const lxb_char_t *end,
++ const lxb_char_t *sbuf_begin, lxb_char_t **sbuf,
++ lxb_char_t **last, size_t *path_count, bool bqs);
+
+ static void
+ lxb_url_path_fix_windows_drive(lxb_url_t *url, lxb_char_t *sbuf,
+@@ -990,23 +990,25 @@ lxb_url_path_shorten(lxb_url_t *url)
+ }
+ }
+
+- if (url->path.str.data != NULL) {
+- url->path.length -= 1;
++ if (url->path.length == 0 || str->data == NULL) {
++ return;
++ }
+
+- begin = str->data;
+- p = begin + str->length;
++ url->path.length -= 1;
+
+- while (p > begin) {
+- p -= 1;
++ begin = str->data;
++ p = begin + str->length;
+
+- if (*p == '/') {
+- *p = '\0';
+- break;
+- }
+- }
++ while (p > begin) {
++ p -= 1;
+
+- str->length = p - begin;
++ if (*p == '/') {
++ *p = '\0';
++ break;
++ }
+ }
++
++ str->length = p - begin;
+ }
+
+ static lxb_status_t
+@@ -2146,6 +2148,8 @@ again:
+ if (status != LXB_STATUS_OK) {
+ lxb_url_parse_return(orig_data, buf, status);
+ }
++
++ url->path.length += 1;
+ }
+ }
+ }
+@@ -2287,7 +2291,13 @@ again:
+ && url->host.type == LXB_URL_HOST_TYPE__UNDEF)
+ {
+ status = lxb_url_path_append(url, mp_str.data, mp_str.length);
+- lxb_url_parse_return(orig_data, buf, status);
++ if (status != LXB_STATUS_OK) {
++ lxb_url_parse_return(orig_data, buf, status);
++ }
++
++ url->path.length += 1;
++
++ lxb_url_parse_return(orig_data, buf, LXB_STATUS_OK);
+ }
+
+ lxb_url_parse_return(orig_data, buf, LXB_STATUS_OK);
+@@ -2535,13 +2545,8 @@ lxb_url_path_fast_path(lxb_url_parser_t *parser, lxb_url_t *url,
+ || lexbor_str_res_map_hex[p[1]] == 0xff
+ || lexbor_str_res_map_hex[p[2]] == 0xff)
+ {
+- status = lxb_url_log_append(parser, p,
+- LXB_URL_ERROR_TYPE_INVALID_URL_UNIT);
+- if (status != LXB_STATUS_OK) {
+- return NULL;
+- }
+-
+- p = (end - p < 3) ? end - 1 : p + 2;
++ /* Reprocess the segment without skipping delimiters. */
++ goto slow;
+ }
+ else if (p[1] == '2' && (p[2] == 'e' || p[2] == 'E')
+ && (p == begin
+@@ -2550,8 +2555,8 @@ lxb_url_path_fast_path(lxb_url_parser_t *parser, lxb_url_t *url,
+ {
+ url->path.length = count;
+
+- status = lxb_url_path_try_dot(url, &begin, &last,
+- &p, end, bqs);
++ status = lxb_url_path_try_dot(parser, url, &begin,
++ &last, &p, end, bqs);
+ if (status != LXB_STATUS_OK) {
+ return NULL;
+ }
+@@ -2589,8 +2594,8 @@ lxb_url_path_fast_path(lxb_url_parser_t *parser, lxb_url_t *url,
+ {
+ url->path.length = count;
+
+- status = lxb_url_path_try_dot(url, &begin, &last,
+- &p, end, bqs);
++ status = lxb_url_path_try_dot(parser, url, &begin,
++ &last, &p, end, bqs);
+ if (status != LXB_STATUS_OK) {
+ return NULL;
+ }
+@@ -2599,17 +2604,7 @@ lxb_url_path_fast_path(lxb_url_parser_t *parser, lxb_url_t *url,
+ }
+ }
+ else {
+- url->path.length = count;
+-
+- if (last - 1 > begin) {
+- status = lxb_url_path_append(url, begin,
+- (last - 1) - begin);
+- if (status != LXB_STATUS_OK) {
+- return NULL;
+- }
+- }
+-
+- return lxb_url_path_slow_path(parser, url, last, end, bqs);
++ goto slow;
+ }
+ }
+ }
+@@ -2619,13 +2614,22 @@ lxb_url_path_fast_path(lxb_url_parser_t *parser, lxb_url_t *url,
+ return NULL;
+ }
+
+- if (count == 0 || p != begin) {
+- count += 1;
+- }
++ url->path.length = count + 1;
++
++ return p;
++
++slow:
+
+ url->path.length = count;
+
+- return p;
++ if (last > begin) {
++ status = lxb_url_path_append(url, begin, (last - 1) - begin);
++ if (status != LXB_STATUS_OK) {
++ return NULL;
++ }
++ }
++
++ return lxb_url_path_slow_path(parser, url, last, end, bqs);
+ }
+
+ /*
+@@ -2721,10 +2725,6 @@ lxb_url_path_slow_path(lxb_url_parser_t *parser, lxb_url_t *url,
+
+ count += 1;
+ last = sbuf;
+-
+- if (p + 1 >= end) {
+- count += 1;
+- }
+ }
+ else if (c == '\\' && lxb_url_is_special(url)) {
+ status = lxb_url_log_append(parser, p,
+@@ -2742,16 +2742,8 @@ lxb_url_path_slow_path(lxb_url_parser_t *parser, lxb_url_t *url,
+
+ count += 1;
+ last = sbuf;
+-
+- if (p + 1 >= end) {
+- count += 1;
+- }
+ }
+ else if ((c == '?' || c == '#') && bqs) {
+- lxb_url_path_fix_windows_drive(url, last, sbuf, count);
+-
+- count += 1;
+- last = sbuf;
+ break;
+ }
+ else if (lxb_url_map[c] & LXB_URL_MAP_PATH) {
+@@ -2771,11 +2763,19 @@ lxb_url_path_slow_path(lxb_url_parser_t *parser, lxb_url_t *url,
+ }
+ else if (c == '.') {
+ if (last == sbuf) {
+- tmp = lxb_url_path_dot_count(url, p, end, sbuf_begin,
++ tmp = lxb_url_path_dot_count(parser, url, p, end, sbuf_begin,
+ &sbuf, &last, &count, bqs);
++ if (tmp == NULL) {
++ goto failed;
++ }
+
+ if (tmp != p) {
+- p = tmp + 1;
++ /* Skip '/' or '\', but leave '?' and '#' to the loop. */
++ if (tmp < end && *tmp != '?' && *tmp != '#') {
++ tmp += 1;
++ }
++
++ p = tmp;
+ continue;
+ }
+ }
+@@ -2800,11 +2800,19 @@ lxb_url_path_slow_path(lxb_url_parser_t *parser, lxb_url_t *url,
+ else if (p[1] == '2' && (p[2] == 'e' || p[2] == 'E')
+ && last == sbuf)
+ {
+- tmp = lxb_url_path_dot_count(url, p, end, sbuf_begin,
++ tmp = lxb_url_path_dot_count(parser, url, p, end, sbuf_begin,
+ &sbuf, &last, &count, bqs);
++ if (tmp == NULL) {
++ goto failed;
++ }
+
+ if (tmp != p) {
+- p = tmp + 1;
++ /* Skip '/' or '\', but leave '?' and '#' to the loop. */
++ if (tmp < end && *tmp != '?' && *tmp != '#') {
++ tmp += 1;
++ }
++
++ p = tmp;
+ continue;
+ }
+ }
+@@ -2833,12 +2841,9 @@ lxb_url_path_slow_path(lxb_url_parser_t *parser, lxb_url_t *url,
+ p += 1;
+ }
+
+- if (count == 0 || last < sbuf) {
+- lxb_url_path_fix_windows_drive(url, last, sbuf, count);
+- count += 1;
+- }
++ lxb_url_path_fix_windows_drive(url, last, sbuf, count);
+
+- url->path.length = count;
++ url->path.length = count + 1;
+
+ status = lxb_url_path_append_wo_slash(url, sbuf_begin, sbuf - sbuf_begin);
+ if (status != LXB_STATUS_OK) {
+@@ -2861,13 +2866,12 @@ failed:
+ }
+
+ static lxb_status_t
+-lxb_url_path_try_dot(lxb_url_t *url, const lxb_char_t **begin,
+- const lxb_char_t **last, const lxb_char_t **start,
+- const lxb_char_t *end, bool bqs)
++lxb_url_path_try_dot(lxb_url_parser_t *parser, lxb_url_t *url,
++ const lxb_char_t **begin, const lxb_char_t **last,
++ const lxb_char_t **start, const lxb_char_t *end, bool bqs)
+ {
+ unsigned count;
+ lxb_char_t c;
+- lexbor_str_t *str;
+ lxb_status_t status;
+ const lxb_char_t *p;
+
+@@ -2912,40 +2916,54 @@ lxb_url_path_try_dot(lxb_url_t *url, const lxb_char_t **begin,
+ }
+ }
+
+- if (p < end) {
+- *start = p;
+- *begin = p + 1;
+- *last = *begin;
++ if (count == 2) {
++ lxb_url_path_shorten(url);
+ }
+- else {
++
++ if (p >= end) {
++ /* The caller appends the trailing empty segment. */
+ *start = end - 1;
+ *begin = end;
+ *last = end;
++
++ return LXB_STATUS_OK;
+ }
+
+- if (count == 2) {
+- lxb_url_path_shorten(url);
++ if (*p == '?' || *p == '#') {
++ /* The caller's loop handles the delimiter and the empty segment. */
++ *start = p - 1;
++ *begin = p;
++ *last = p;
++
++ return LXB_STATUS_OK;
+ }
+- else if (count == 1) {
+- str = &url->path.str;
+
+- if (str->length > 0 && str->data[str->length - 1] == '/') {
+- str->length -= 1;
+- str->data[str->length] = '\0';
++ if (*p == '\\') {
++ status = lxb_url_log_append(parser, p,
++ LXB_URL_ERROR_TYPE_INVALID_REVERSE_SOLIDUS);
++ if (status != LXB_STATUS_OK) {
++ return status;
+ }
+ }
+
++ /* Skip '/' or '\'. */
++
++ *start = p;
++ *begin = p + 1;
++ *last = *begin;
++
+ return LXB_STATUS_OK;
+ }
+
+ static const lxb_char_t *
+-lxb_url_path_dot_count(lxb_url_t *url, const lxb_char_t *p,
+- const lxb_char_t *end, const lxb_char_t *sbuf_begin,
+- lxb_char_t **sbuf, lxb_char_t **last, size_t *path_count,
+- bool bqs)
++lxb_url_path_dot_count(lxb_url_parser_t *parser, lxb_url_t *url,
++ const lxb_char_t *p, const lxb_char_t *end,
++ const lxb_char_t *sbuf_begin, lxb_char_t **sbuf,
++ lxb_char_t **last, size_t *path_count, bool bqs)
+ {
+ unsigned count;
+ lxb_char_t c, *last_p;
++ lxb_status_t status;
+ const lxb_char_t *begin;
+
+ count = 0;
+@@ -2982,6 +3000,14 @@ lxb_url_path_dot_count(lxb_url_t *url, const lxb_char_t *p,
+ return begin;
+ }
+
++ if (p < end && *p == '\\') {
++ status = lxb_url_log_append(parser, p,
++ LXB_URL_ERROR_TYPE_INVALID_REVERSE_SOLIDUS);
++ if (status != LXB_STATUS_OK) {
++ return NULL;
++ }
++ }
++
+ if (url->scheme.type == LXB_URL_SCHEMEL_TYPE_FILE
+ && *path_count == 1
+ && lxb_url_normalized_windows_drive_letter(sbuf_begin + 1, *last - 1))
diff --git a/ext/lexbor/patches/0020-URL-Keep-replacement-file-drive-paths-hierarchical-4.patch b/ext/lexbor/patches/0020-URL-Keep-replacement-file-drive-paths-hierarchical-4.patch
new file mode 100644
index 00000000000..9871cb65064
--- /dev/null
+++ b/ext/lexbor/patches/0020-URL-Keep-replacement-file-drive-paths-hierarchical-4.patch
@@ -0,0 +1,23 @@
+From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
+From: =?UTF-8?q?M=C3=A1t=C3=A9=20Kocsis?= <kocsismate@woohoolabs.com>
+Date: Mon, 21 Sep 2026 20:01:43 +0200
+Subject: [PATCH 20/21] URL: Keep replacement file drive paths hierarchical
+ (#423)
+
+WHATWG file state (https://url.spec.whatwg.org/#file-state) step 4.4.3.2 resets the path to an empty list. Do not mark it opaque, as that makes subsequent pathname updates silently do nothing.
+---
+ source/lexbor/url/url.c | 1 -
+ 1 file changed, 1 deletion(-)
+
+diff --git a/source/lexbor/url/url.c b/source/lexbor/url/url.c
+index 55fe11f..2231816 100644
+--- a/source/lexbor/url/url.c
++++ b/source/lexbor/url/url.c
+@@ -2099,7 +2099,6 @@ again:
+ }
+
+ lxb_url_path_set_null(url);
+- url->path.opaque = true;
+ }
+ }
+
diff --git a/ext/lexbor/patches/0021-URL-normalize-output-encoding-for-percent-encoding.patch b/ext/lexbor/patches/0021-URL-normalize-output-encoding-for-percent-encoding.patch
new file mode 100644
index 00000000000..715d7c8241d
--- /dev/null
+++ b/ext/lexbor/patches/0021-URL-normalize-output-encoding-for-percent-encoding.patch
@@ -0,0 +1,80 @@
+From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
+From: Alexander Borisov <lex.borisov@gmail.com>
+Date: Thu, 24 Sep 2026 15:05:51 +0300
+Subject: [PATCH 21/21] URL: normalize output encoding for percent-encoding.
+
+---
+ source/lexbor/url/url.c | 34 +++++++++++++++++++++++++---------
+ 1 file changed, 25 insertions(+), 9 deletions(-)
+
+diff --git a/source/lexbor/url/url.c b/source/lexbor/url/url.c
+index 2231816..146f0bd 100644
+--- a/source/lexbor/url/url.c
++++ b/source/lexbor/url/url.c
+@@ -1228,6 +1228,26 @@ lxb_url_encoding_init(const lxb_encoding_data_t *encoding,
+ (void) lxb_encoding_encode_init_single(encode, encoding);
+ }
+
++/*
++ * https://encoding.spec.whatwg.org/#get-an-output-encoding
++ */
++lxb_inline lxb_encoding_t
++lxb_url_output_encoding(lxb_encoding_t encoding)
++{
++ switch (encoding) {
++ case LXB_ENCODING_DEFAULT:
++ case LXB_ENCODING_AUTO:
++ case LXB_ENCODING_UNDEFINED:
++ case LXB_ENCODING_REPLACEMENT:
++ case LXB_ENCODING_UTF_16BE:
++ case LXB_ENCODING_UTF_16LE:
++ return LXB_ENCODING_UTF_8;
++
++ default:
++ return encoding;
++ }
++}
++
+ static bool
+ lxb_url_start_windows_drive_letter(const lxb_char_t *data,
+ const lxb_char_t *end)
+@@ -1372,12 +1392,7 @@ lxb_url_parse_basic_h(lxb_url_parser_t *parser, lxb_url_t *url,
+ state = override_state;
+ }
+
+- if (encoding <= LXB_ENCODING_UNDEFINED
+- || encoding == LXB_ENCODING_UTF_16BE
+- || encoding == LXB_ENCODING_UTF_16LE)
+- {
+- encoding = LXB_ENCODING_UTF_8;
+- }
++ encoding = lxb_url_output_encoding(encoding);
+
+ enc = lxb_encoding_data(encoding);
+ if (enc == NULL) {
+@@ -3221,7 +3236,7 @@ lxb_url_percent_encode_after_encoding(const lxb_char_t *data,
+ const lxb_char_t *buf_end = buf + sizeof(buffer);
+ static const lexbor_str_t esc_str = lexbor_str("%26%23");
+
+- if (encoding->encoding == LXB_ENCODING_UTF_8) {
++ if (lxb_url_output_encoding(encoding->encoding) == LXB_ENCODING_UTF_8) {
+ return lxb_url_percent_encode_after_utf_8(data, end, str, mraw,
+ enmap, space_as_plus);
+ }
+@@ -3256,13 +3271,14 @@ lxb_url_percent_encode_after_encoding(const lxb_char_t *data,
+ len = encoding->encode_single(&encode, &buf, buf_end, cp);
+
+ if (len < LXB_ENCODING_ENCODE_OK) {
+- size = lexbor_conv_int64_to_data((int64_t) cp, buf, buf_end - buf);
++ size = lexbor_conv_int64_to_data((int64_t) cp, buffer,
++ sizeof(buffer));
+
+ if (lexbor_str_append(str, mraw, esc_str.data, esc_str.length) == NULL) {
+ return LXB_STATUS_ERROR_MEMORY_ALLOCATION;
+ }
+
+- if (lexbor_str_append(str, mraw, buf, size) == NULL) {
++ if (lexbor_str_append(str, mraw, buffer, size) == NULL) {
+ return LXB_STATUS_ERROR_MEMORY_ALLOCATION;
+ }
+
diff --git a/ext/uri/tests/whatwg/modification/withPath_success_file_invalid_drive_letter.phpt b/ext/uri/tests/whatwg/modification/withPath_success_file_invalid_drive_letter.phpt
new file mode 100644
index 00000000000..39ed5d30a4f
--- /dev/null
+++ b/ext/uri/tests/whatwg/modification/withPath_success_file_invalid_drive_letter.phpt
@@ -0,0 +1,21 @@
+--TEST--
+Test Uri\WhatWg\Url::withPath() - file URL whose path was reset by an invalid drive letter
+--FILE--
+<?php
+
+$url = Uri\WhatWg\Url::parse("c|/x", new Uri\WhatWg\Url("file:///d:/a/b"), $errors);
+
+var_dump($url->getPath());
+var_dump(array_map(static fn (Uri\WhatWg\UrlValidationError $error): string => $error->type->name, $errors));
+var_dump($url->withPath("/zz")->getPath());
+
+?>
+--EXPECT--
+string(5) "/c:/x"
+array(2) {
+ [0]=>
+ string(14) "InvalidUrlUnit"
+ [1]=>
+ string(29) "FileInvalidWindowsDriveLetter"
+}
+string(3) "/zz"
diff --git a/ext/uri/tests/whatwg/parsing/host_success_multibyte_long.phpt b/ext/uri/tests/whatwg/parsing/host_success_multibyte_long.phpt
new file mode 100644
index 00000000000..109cb39de1f
--- /dev/null
+++ b/ext/uri/tests/whatwg/parsing/host_success_multibyte_long.phpt
@@ -0,0 +1,21 @@
+--TEST--
+Test Uri\WhatWg\Url parsing - host - IDN host longer than the IDNA on-stack buffer
+--FILE--
+<?php
+
+$url = Uri\WhatWg\Url::parse("https://" . str_repeat("é", 5000) . ".com/");
+$host = $url->getAsciiHost();
+
+var_dump(strlen($host));
+var_dump(substr($host, 0, 12));
+var_dump(substr($host, -6));
+var_dump(substr_count($host, "a"));
+var_dump($url->getUnicodeHost() === str_repeat("é", 5000) . ".com");
+
+?>
+--EXPECT--
+int(5010)
+string(12) "xn--9caaaaaa"
+string(6) "aa.com"
+int(5000)
+bool(true)
diff --git a/ext/uri/tests/whatwg/parsing/path_success_dot_segment_delimiters.phpt b/ext/uri/tests/whatwg/parsing/path_success_dot_segment_delimiters.phpt
new file mode 100644
index 00000000000..648edbf2653
--- /dev/null
+++ b/ext/uri/tests/whatwg/parsing/path_success_dot_segment_delimiters.phpt
@@ -0,0 +1,29 @@
+--TEST--
+Test Uri\WhatWg\Url parsing - path - query and fragment after a dot segment
+--FILE--
+<?php
+
+foreach ([
+ "https://example.com/..#frag",
+ "https://example.com/caf\u{e9}/..#frag",
+ "https://example.com/..?q=1",
+ "https://example.com/%?q",
+] as $input) {
+ $url = new Uri\WhatWg\Url($input);
+ var_dump($url->getPath(), $url->getQuery(), $url->getFragment());
+}
+
+?>
+--EXPECT--
+string(1) "/"
+NULL
+string(4) "frag"
+string(1) "/"
+NULL
+string(4) "frag"
+string(1) "/"
+string(3) "q=1"
+NULL
+string(2) "/%"
+string(1) "q"
+NULL
diff --git a/ext/uri/tests/whatwg/parsing/path_success_opaque_space_before_query.phpt b/ext/uri/tests/whatwg/parsing/path_success_opaque_space_before_query.phpt
new file mode 100644
index 00000000000..3ffdd94f54c
--- /dev/null
+++ b/ext/uri/tests/whatwg/parsing/path_success_opaque_space_before_query.phpt
@@ -0,0 +1,16 @@
+--TEST--
+Test Uri\WhatWg\Url parsing - path - space before the query or fragment of an opaque path
+--FILE--
+<?php
+
+foreach (["data:x ?q", "data:x #f", "foo:a b ?c", "data:x ?q"] as $input) {
+ $url = new Uri\WhatWg\Url($input);
+ var_dump($url->getPath());
+}
+
+?>
+--EXPECT--
+string(4) "x%20"
+string(4) "x%20"
+string(6) "a b%20"
+string(5) "x %20"
diff --git a/ext/uri/tests/whatwg/parsing/username_success_at_sign.phpt b/ext/uri/tests/whatwg/parsing/username_success_at_sign.phpt
new file mode 100644
index 00000000000..910d984858f
--- /dev/null
+++ b/ext/uri/tests/whatwg/parsing/username_success_at_sign.phpt
@@ -0,0 +1,18 @@
+--TEST--
+Test Uri\WhatWg\Url parsing - username - at sign in the username and the password
+--FILE--
+<?php
+
+$url = new Uri\WhatWg\Url("http://user@name:pass@word@localhost/");
+
+var_dump($url->getUsername());
+var_dump($url->getPassword());
+var_dump($url->getAsciiHost());
+var_dump($url->toAsciiString());
+
+?>
+--EXPECT--
+string(11) "user%40name"
+string(11) "pass%40word"
+string(9) "localhost"
+string(41) "http://user%40name:pass%40word@localhost/"