Commit beee5444d01 for php

commit beee5444d016b4699d1ef48376c9ef7659adbdd0
Author: Alexandre Daubois <alex.daubois@gmail.com>
Date:   Mon Oct 5 14:18:04 2026 +0200

    lexbor: Merge upstream WHATWG URL and IDNA fixes

    lexbor/lexbor@859f100
    lexbor/lexbor@a36e09a
    lexbor/lexbor@b0f7412
    lexbor/lexbor@1b215a8
    lexbor/lexbor@385afff
    lexbor/lexbor@e6c068f
    lexbor/lexbor@327a8b6
    lexbor/lexbor@917742f
    lexbor/lexbor@e89c258

    Only the source/lexbor/url part of lexbor/lexbor@327a8b6 and
    lexbor/lexbor@e89c258 is carried: the first squash also contains unrelated
    HTML tree construction changes, and the second documents a percent-encoder
    API that this branch does not bundle.

diff --git a/NEWS b/NEWS
index 1039ef9a31e..a42642d601e 100644
--- a/NEWS
+++ b/NEWS
@@ -87,6 +87,15 @@ PHP                                                                        NEWS
   . Merge patches lexbor/lexbor@8a14bc0 and lexbor/lexbor@f67ce4b, fixing a
     heap buffer overflow in :lexbor-contains() parsing and buffer overflows
     in malformed decode replay. (alexandre-daubois)
+  . Merge patches lexbor/lexbor@859f100, lexbor/lexbor@a36e09a,
+    lexbor/lexbor@b0f7412, lexbor/lexbor@1b215a8, lexbor/lexbor@385afff,
+    lexbor/lexbor@e6c068f, lexbor/lexbor@327a8b6, lexbor/lexbor@917742f and
+    lexbor/lexbor@e89c258, fixing dropped usernames containing an at sign,
+    uninitialized memory in IDNA buffer growth, the encoding of a space
+    before a query or fragment in an opaque path, a query or fragment lost
+    after a dot segment in a path, replacement file drive paths, the output
+    encoding used for percent-encoding, the URLSearchParams tail pointer and
+    fragment serialization without a query. (alexandre-daubois)

 - MBString:
   . Fixed bug GH-23106 (mb_strpos() reads past the end of a haystack ending in
diff --git a/ext/lexbor/lexbor/unicode/idna.c b/ext/lexbor/lexbor/unicode/idna.c
index 754f6b2026d..b31ae3338b9 100644
--- a/ext/lexbor/lexbor/unicode/idna.c
+++ b/ext/lexbor/lexbor/unicode/idna.c
@@ -117,12 +117,14 @@ lxb_unicode_idna_realloc(lxb_codepoint_t *buf, const lxb_codepoint_t *buffer,
     lxb_codepoint_t *tmp;

     nlen = ((*buf_end - buf) * 4) + len;
-
+
     if (buf == buffer) {
         tmp = lexbor_malloc(nlen * sizeof(lxb_codepoint_t));
         if (tmp == NULL) {
             return NULL;
         }
+
+        memcpy(tmp, buf, (*buf_p - buf) * sizeof(lxb_codepoint_t));
     }
     else {
         tmp = lexbor_realloc(buf, nlen * sizeof(lxb_codepoint_t));
@@ -458,13 +460,17 @@ lxb_unicode_idna_ascii_puny_cb(const lxb_char_t *data, size_t length, void *ctx,

         if (asc->buf == asc->buffer) {
             tmp = lexbor_malloc(nlen);
+            if (tmp == NULL) {
+                return LXB_STATUS_ERROR_MEMORY_ALLOCATION;
+            }
+
+            memcpy(tmp, asc->buf, asc->p - asc->buf);
         }
         else {
             tmp = lexbor_realloc(asc->buf, nlen);
-        }
-
-        if (tmp == NULL) {
-            return LXB_STATUS_ERROR_MEMORY_ALLOCATION;
+            if (tmp == NULL) {
+                return LXB_STATUS_ERROR_MEMORY_ALLOCATION;
+            }
         }

         asc->p = tmp + (asc->p - asc->buf);
@@ -711,13 +717,17 @@ lxb_unicode_idna_to_unicode_cb(const lxb_codepoint_t *part, size_t len,

         if (asc->buf == asc->buffer) {
             tmp = lexbor_malloc(nlen);
+            if (tmp == NULL) {
+                return LXB_STATUS_ERROR_MEMORY_ALLOCATION;
+            }
+
+            memcpy(tmp, asc->buf, asc->p - asc->buf);
         }
         else {
             tmp = lexbor_realloc(asc->buf, nlen);
-        }
-
-        if (tmp == NULL) {
-            return LXB_STATUS_ERROR_MEMORY_ALLOCATION;
+            if (tmp == NULL) {
+                return LXB_STATUS_ERROR_MEMORY_ALLOCATION;
+            }
         }

         asc->p = tmp + (asc->p - asc->buf);
diff --git a/ext/lexbor/lexbor/url/url.c b/ext/lexbor/lexbor/url/url.c
index de19239936a..146f0bda292 100644
--- a/ext/lexbor/lexbor/url/url.c
+++ b/ext/lexbor/lexbor/url/url.c
@@ -546,15 +546,15 @@ lxb_url_path_slow_path(lxb_url_parser_t *parser, lxb_url_t *url,
                        const lxb_char_t *data, const lxb_char_t *end, bool bqs);

 static lxb_status_t
-lxb_url_path_try_dot(lxb_url_t *url, const lxb_char_t **begin,
-                     const lxb_char_t **last, const lxb_char_t **start,
-                     const lxb_char_t *end, bool bqs);
+lxb_url_path_try_dot(lxb_url_parser_t *parser, lxb_url_t *url,
+                     const lxb_char_t **begin, const lxb_char_t **last,
+                     const lxb_char_t **start, const lxb_char_t *end, bool bqs);

 static const lxb_char_t *
-lxb_url_path_dot_count(lxb_url_t *url, const lxb_char_t *p,
-                       const lxb_char_t *end, const lxb_char_t *sbuf_begin,
-                       lxb_char_t **sbuf, lxb_char_t **last, size_t *path_count,
-                       bool bqs);
+lxb_url_path_dot_count(lxb_url_parser_t *parser, lxb_url_t *url,
+                       const lxb_char_t *p, const lxb_char_t *end,
+                       const lxb_char_t *sbuf_begin, lxb_char_t **sbuf,
+                       lxb_char_t **last, size_t *path_count, bool bqs);

 static void
 lxb_url_path_fix_windows_drive(lxb_url_t *url, lxb_char_t *sbuf,
@@ -923,7 +923,7 @@ lxb_url_scheme_copy_special(const lxb_url_scheme_data_t *src,
     return lxb_url_str_copy(&src->name, &dst->name, dst_mraw);
 }

-static void
+void
 lxb_url_path_set_null(lxb_url_t *url)
 {
     if (url->path.str.data == NULL) {
@@ -990,23 +990,25 @@ lxb_url_path_shorten(lxb_url_t *url)
         }
     }

-    if (url->path.str.data != NULL) {
-        url->path.length -= 1;
+    if (url->path.length == 0 || str->data == NULL) {
+        return;
+    }

-        begin = str->data;
-        p = begin + str->length;
+    url->path.length -= 1;

-        while (p > begin) {
-            p -= 1;
+    begin = str->data;
+    p = begin + str->length;

-            if (*p == '/') {
-                *p = '\0';
-                break;
-            }
-        }
+    while (p > begin) {
+        p -= 1;

-        str->length = p - begin;
+        if (*p == '/') {
+            *p = '\0';
+            break;
+        }
     }
+
+    str->length = p - begin;
 }

 static lxb_status_t
@@ -1147,7 +1149,7 @@ lxb_url_host_destroy(lxb_url_host_t *host, lexbor_mraw_t *mraw)
     }
 }

-static void
+void
 lxb_url_host_set_empty(lxb_url_host_t *host, lexbor_mraw_t *mraw)
 {
     lxb_url_host_destroy(host, mraw);
@@ -1197,7 +1199,15 @@ lxb_url_port_set(lxb_url_t *url, uint16_t port)
     url->has_port = true;
 }

-static void
+void
+lxb_url_query_set_null(lxb_url_t *url)
+{
+    if (url->query.data != NULL) {
+        (void) lexbor_str_destroy(&url->query, url->mraw, false);
+    }
+}
+
+void
 lxb_url_fragment_set_null(lxb_url_t *url)
 {
     if (url->fragment.data != NULL) {
@@ -1218,6 +1228,26 @@ lxb_url_encoding_init(const lxb_encoding_data_t *encoding,
     (void) lxb_encoding_encode_init_single(encode, encoding);
 }

+/*
+ * https://encoding.spec.whatwg.org/#get-an-output-encoding
+ */
+lxb_inline lxb_encoding_t
+lxb_url_output_encoding(lxb_encoding_t encoding)
+{
+    switch (encoding) {
+        case LXB_ENCODING_DEFAULT:
+        case LXB_ENCODING_AUTO:
+        case LXB_ENCODING_UNDEFINED:
+        case LXB_ENCODING_REPLACEMENT:
+        case LXB_ENCODING_UTF_16BE:
+        case LXB_ENCODING_UTF_16LE:
+            return LXB_ENCODING_UTF_8;
+
+        default:
+            return encoding;
+    }
+}
+
 static bool
 lxb_url_start_windows_drive_letter(const lxb_char_t *data,
                                    const lxb_char_t *end)
@@ -1362,12 +1392,7 @@ lxb_url_parse_basic_h(lxb_url_parser_t *parser, lxb_url_t *url,
         state = override_state;
     }

-    if (encoding <= LXB_ENCODING_UNDEFINED
-        || encoding == LXB_ENCODING_UTF_16BE
-        || encoding == LXB_ENCODING_UTF_16LE)
-    {
-        encoding = LXB_ENCODING_UTF_8;
-    }
+    encoding = lxb_url_output_encoding(encoding);

     enc = lxb_encoding_data(encoding);
     if (enc == NULL) {
@@ -1753,16 +1778,13 @@ lxb_url_parse_basic_h(lxb_url_parser_t *parser, lxb_url_t *url,
                         break;
                     }

-                    if (pswd == NULL || !at_sign) {
-                        tmp = (pswd != NULL) ? pswd - 1 : p;
-
-                        if (tmp > begin) {
-                            status = lxb_url_percent_encode_after_utf_8(begin, tmp,
-                                                        &url->username, url->mraw,
-                                                        LXB_URL_MAP_USERINFO, false);
-                            if (status != LXB_STATUS_OK) {
-                                lxb_url_parse_return(orig_data, buf, status);
-                            }
+                    tmp = (pswd != NULL) ? pswd - 1 : p;
+                    if (tmp > begin) {
+                        status = lxb_url_percent_encode_after_utf_8(begin, tmp,
+                                                    &url->username, url->mraw,
+                                                    LXB_URL_MAP_USERINFO, false);
+                        if (status != LXB_STATUS_OK) {
+                            lxb_url_parse_return(orig_data, buf, status);
                         }
                     }

@@ -2092,7 +2114,6 @@ lxb_url_parse_basic_h(lxb_url_parser_t *parser, lxb_url_t *url,
                 }

                 lxb_url_path_set_null(url);
-                url->path.opaque = true;
             }
         }

@@ -2141,6 +2162,8 @@ lxb_url_parse_basic_h(lxb_url_parser_t *parser, lxb_url_t *url,
                     if (status != LXB_STATUS_OK) {
                         lxb_url_parse_return(orig_data, buf, status);
                     }
+
+                    url->path.length += 1;
                 }
             }
         }
@@ -2282,7 +2305,13 @@ lxb_url_parse_basic_h(lxb_url_parser_t *parser, lxb_url_t *url,
             && url->host.type == LXB_URL_HOST_TYPE__UNDEF)
         {
             status = lxb_url_path_append(url, mp_str.data, mp_str.length);
-            lxb_url_parse_return(orig_data, buf, status);
+            if (status != LXB_STATUS_OK) {
+                lxb_url_parse_return(orig_data, buf, status);
+            }
+
+            url->path.length += 1;
+
+            lxb_url_parse_return(orig_data, buf, LXB_STATUS_OK);
         }

         lxb_url_parse_return(orig_data, buf, LXB_STATUS_OK);
@@ -2343,6 +2372,17 @@ lxb_url_parse_basic_h(lxb_url_parser_t *parser, lxb_url_t *url,
                     lxb_url_parse_return(orig_data, buf, status);
                 }

+                /* Encode only the space immediately before a query or fragment. */
+                if (p > begin && p[-1] == ' ') {
+                    tmp_str.length--;
+                    if (lexbor_str_append(&tmp_str, url->mraw,
+                                          (const lxb_char_t *) "%20", 3) == NULL)
+                    {
+                        lxb_url_parse_return(orig_data, buf,
+                                             LXB_STATUS_ERROR_MEMORY_ALLOCATION);
+                    }
+                }
+
                 status = lxb_url_path_list_push(url, &tmp_str);
                 if (status != LXB_STATUS_OK) {
                     lxb_url_parse_return(orig_data, buf, status);
@@ -2519,13 +2559,8 @@ lxb_url_path_fast_path(lxb_url_parser_t *parser, lxb_url_t *url,
                     || lexbor_str_res_map_hex[p[1]] == 0xff
                     || lexbor_str_res_map_hex[p[2]] == 0xff)
                 {
-                    status = lxb_url_log_append(parser, p,
-                                                LXB_URL_ERROR_TYPE_INVALID_URL_UNIT);
-                    if (status != LXB_STATUS_OK) {
-                        return NULL;
-                    }
-
-                    p = (end - p < 3) ? end - 1 : p + 2;
+                    /* Reprocess the segment without skipping delimiters. */
+                    goto slow;
                 }
                 else if (p[1] == '2' && (p[2] == 'e' || p[2] == 'E')
                          && (p == begin
@@ -2534,8 +2569,8 @@ lxb_url_path_fast_path(lxb_url_parser_t *parser, lxb_url_t *url,
                 {
                     url->path.length = count;

-                    status = lxb_url_path_try_dot(url, &begin, &last,
-                                                  &p, end, bqs);
+                    status = lxb_url_path_try_dot(parser, url, &begin,
+                                                  &last, &p, end, bqs);
                     if (status != LXB_STATUS_OK) {
                         return NULL;
                     }
@@ -2573,8 +2608,8 @@ lxb_url_path_fast_path(lxb_url_parser_t *parser, lxb_url_t *url,
                 {
                     url->path.length = count;

-                    status = lxb_url_path_try_dot(url, &begin, &last,
-                                                  &p, end, bqs);
+                    status = lxb_url_path_try_dot(parser, url, &begin,
+                                                  &last, &p, end, bqs);
                     if (status != LXB_STATUS_OK) {
                         return NULL;
                     }
@@ -2583,17 +2618,7 @@ lxb_url_path_fast_path(lxb_url_parser_t *parser, lxb_url_t *url,
                 }
             }
             else {
-                url->path.length = count;
-
-                if (last - 1 > begin) {
-                    status = lxb_url_path_append(url, begin,
-                                                 (last - 1) - begin);
-                    if (status != LXB_STATUS_OK) {
-                        return NULL;
-                    }
-                }
-
-                return lxb_url_path_slow_path(parser, url, last, end, bqs);
+                goto slow;
             }
         }
     }
@@ -2603,13 +2628,22 @@ lxb_url_path_fast_path(lxb_url_parser_t *parser, lxb_url_t *url,
         return NULL;
     }

-    if (count == 0 || p != begin) {
-        count += 1;
-    }
+    url->path.length = count + 1;
+
+    return p;
+
+slow:

     url->path.length = count;

-    return p;
+    if (last > begin) {
+        status = lxb_url_path_append(url, begin, (last - 1) - begin);
+        if (status != LXB_STATUS_OK) {
+            return NULL;
+        }
+    }
+
+    return lxb_url_path_slow_path(parser, url, last, end, bqs);
 }

 /*
@@ -2705,10 +2739,6 @@ lxb_url_path_slow_path(lxb_url_parser_t *parser, lxb_url_t *url,

             count += 1;
             last = sbuf;
-
-            if (p + 1 >= end) {
-                count += 1;
-            }
         }
         else if (c == '\\' && lxb_url_is_special(url)) {
             status = lxb_url_log_append(parser, p,
@@ -2726,16 +2756,8 @@ lxb_url_path_slow_path(lxb_url_parser_t *parser, lxb_url_t *url,

             count += 1;
             last = sbuf;
-
-            if (p + 1 >= end) {
-                count += 1;
-            }
         }
         else if ((c == '?' || c == '#') && bqs) {
-            lxb_url_path_fix_windows_drive(url, last, sbuf, count);
-
-            count += 1;
-            last = sbuf;
             break;
         }
         else if (lxb_url_map[c] & LXB_URL_MAP_PATH) {
@@ -2755,11 +2777,19 @@ lxb_url_path_slow_path(lxb_url_parser_t *parser, lxb_url_t *url,
         }
         else if (c == '.') {
             if (last == sbuf) {
-                tmp = lxb_url_path_dot_count(url, p, end, sbuf_begin,
+                tmp = lxb_url_path_dot_count(parser, url, p, end, sbuf_begin,
                                              &sbuf, &last, &count, bqs);
+                if (tmp == NULL) {
+                    goto failed;
+                }

                 if (tmp != p) {
-                    p = tmp + 1;
+                    /* Skip '/' or '\', but leave '?' and '#' to the loop. */
+                    if (tmp < end && *tmp != '?' && *tmp != '#') {
+                        tmp += 1;
+                    }
+
+                    p = tmp;
                     continue;
                 }
             }
@@ -2784,11 +2814,19 @@ lxb_url_path_slow_path(lxb_url_parser_t *parser, lxb_url_t *url,
             else if (p[1] == '2' && (p[2] == 'e' || p[2] == 'E')
                      && last == sbuf)
             {
-                tmp = lxb_url_path_dot_count(url, p, end, sbuf_begin,
+                tmp = lxb_url_path_dot_count(parser, url, p, end, sbuf_begin,
                                              &sbuf, &last, &count, bqs);
+                if (tmp == NULL) {
+                    goto failed;
+                }

                 if (tmp != p) {
-                    p = tmp + 1;
+                    /* Skip '/' or '\', but leave '?' and '#' to the loop. */
+                    if (tmp < end && *tmp != '?' && *tmp != '#') {
+                        tmp += 1;
+                    }
+
+                    p = tmp;
                     continue;
                 }
             }
@@ -2817,12 +2855,9 @@ lxb_url_path_slow_path(lxb_url_parser_t *parser, lxb_url_t *url,
         p += 1;
     }

-    if (count == 0 || last < sbuf) {
-        lxb_url_path_fix_windows_drive(url, last, sbuf, count);
-        count += 1;
-    }
+    lxb_url_path_fix_windows_drive(url, last, sbuf, count);

-    url->path.length = count;
+    url->path.length = count + 1;

     status = lxb_url_path_append_wo_slash(url, sbuf_begin, sbuf - sbuf_begin);
     if (status != LXB_STATUS_OK) {
@@ -2845,13 +2880,12 @@ lxb_url_path_slow_path(lxb_url_parser_t *parser, lxb_url_t *url,
 }

 static lxb_status_t
-lxb_url_path_try_dot(lxb_url_t *url, const lxb_char_t **begin,
-                     const lxb_char_t **last, const lxb_char_t **start,
-                     const lxb_char_t *end, bool bqs)
+lxb_url_path_try_dot(lxb_url_parser_t *parser, lxb_url_t *url,
+                     const lxb_char_t **begin, const lxb_char_t **last,
+                     const lxb_char_t **start, const lxb_char_t *end, bool bqs)
 {
     unsigned count;
     lxb_char_t c;
-    lexbor_str_t *str;
     lxb_status_t status;
     const lxb_char_t *p;

@@ -2896,40 +2930,54 @@ lxb_url_path_try_dot(lxb_url_t *url, const lxb_char_t **begin,
         }
     }

-    if (p < end) {
-        *start = p;
-        *begin = p + 1;
-        *last = *begin;
+    if (count == 2) {
+        lxb_url_path_shorten(url);
     }
-    else {
+
+    if (p >= end) {
+        /* The caller appends the trailing empty segment. */
         *start = end - 1;
         *begin = end;
         *last = end;
+
+        return LXB_STATUS_OK;
     }

-    if (count == 2) {
-        lxb_url_path_shorten(url);
+    if (*p == '?' || *p == '#') {
+        /* The caller's loop handles the delimiter and the empty segment. */
+        *start = p - 1;
+        *begin = p;
+        *last = p;
+
+        return LXB_STATUS_OK;
     }
-    else if (count == 1) {
-        str = &url->path.str;

-        if (str->length > 0 && str->data[str->length - 1] == '/') {
-            str->length -= 1;
-            str->data[str->length] = '\0';
+    if (*p == '\\') {
+        status = lxb_url_log_append(parser, p,
+                                    LXB_URL_ERROR_TYPE_INVALID_REVERSE_SOLIDUS);
+        if (status != LXB_STATUS_OK) {
+            return status;
         }
     }

+    /* Skip '/' or '\'. */
+
+    *start = p;
+    *begin = p + 1;
+    *last = *begin;
+
     return LXB_STATUS_OK;
 }

 static const lxb_char_t *
-lxb_url_path_dot_count(lxb_url_t *url, const lxb_char_t *p,
-                       const lxb_char_t *end, const lxb_char_t *sbuf_begin,
-                       lxb_char_t **sbuf, lxb_char_t **last, size_t *path_count,
-                       bool bqs)
+lxb_url_path_dot_count(lxb_url_parser_t *parser, lxb_url_t *url,
+                       const lxb_char_t *p, const lxb_char_t *end,
+                       const lxb_char_t *sbuf_begin, lxb_char_t **sbuf,
+                       lxb_char_t **last, size_t *path_count, bool bqs)
 {
     unsigned count;
     lxb_char_t c, *last_p;
+    lxb_status_t status;
     const lxb_char_t *begin;

     count = 0;
@@ -2966,6 +3014,14 @@ lxb_url_path_dot_count(lxb_url_t *url, const lxb_char_t *p,
         return begin;
     }

+    if (p < end && *p == '\\') {
+        status = lxb_url_log_append(parser, p,
+                                    LXB_URL_ERROR_TYPE_INVALID_REVERSE_SOLIDUS);
+        if (status != LXB_STATUS_OK) {
+            return NULL;
+        }
+    }
+
     if (url->scheme.type == LXB_URL_SCHEMEL_TYPE_FILE
         && *path_count == 1
         && lxb_url_normalized_windows_drive_letter(sbuf_begin + 1, *last - 1))
@@ -3180,7 +3236,7 @@ lxb_url_percent_encode_after_encoding(const lxb_char_t *data,
     const lxb_char_t *buf_end = buf + sizeof(buffer);
     static const lexbor_str_t esc_str = lexbor_str("%26%23");

-    if (encoding->encoding == LXB_ENCODING_UTF_8) {
+    if (lxb_url_output_encoding(encoding->encoding) == LXB_ENCODING_UTF_8) {
         return lxb_url_percent_encode_after_utf_8(data, end, str, mraw,
                                                   enmap, space_as_plus);
     }
@@ -3215,13 +3271,14 @@ lxb_url_percent_encode_after_encoding(const lxb_char_t *data,
         len = encoding->encode_single(&encode, &buf, buf_end, cp);

         if (len < LXB_ENCODING_ENCODE_OK) {
-            size = lexbor_conv_int64_to_data((int64_t) cp, buf, buf_end - buf);
+            size = lexbor_conv_int64_to_data((int64_t) cp, buffer,
+                                             sizeof(buffer));

             if (lexbor_str_append(str, mraw, esc_str.data, esc_str.length) == NULL) {
                 return LXB_STATUS_ERROR_MEMORY_ALLOCATION;
             }

-            if (lexbor_str_append(str, mraw, buf, size) == NULL) {
+            if (lexbor_str_append(str, mraw, buffer, size) == NULL) {
                 return LXB_STATUS_ERROR_MEMORY_ALLOCATION;
             }

@@ -4907,7 +4964,7 @@ lxb_status_t
 lxb_url_serialize_fragment(const lxb_url_t *url,
                            lexbor_serialize_cb_f cb, void *ctx)
 {
-    if (url->query.data != NULL) {
+    if (url->fragment.data != NULL) {
         return cb(url->fragment.data, url->fragment.length, ctx);
     }

@@ -5106,6 +5163,8 @@ lxb_url_search_params_parse(lxb_url_search_params_t *search_params,
             return status;
         }

+        last = entry;
+
         lexbor_str_init(&entry->value, mraw, 0);
         if (entry->value.data == NULL) {
             return LXB_STATUS_ERROR_MEMORY_ALLOCATION;
diff --git a/ext/lexbor/lexbor/url/url.h b/ext/lexbor/lexbor/url/url.h
index 4ed3f32aa64..b9e4973a674 100644
--- a/ext/lexbor/lexbor/url/url.h
+++ b/ext/lexbor/lexbor/url/url.h
@@ -763,6 +763,51 @@ LXB_API lxb_status_t
 lxb_url_search_params_serialize(lxb_url_search_params_t *search_params,
                                 lexbor_callback_f cb, void *ctx);

+/*
+ * Reset the URL path to an empty list.
+ *
+ * Frees the path buffer using url->mraw, resets the segment count and clears
+ * the opaque flag. Does nothing if the path buffer is already NULL.
+ *
+ * @param[in, out] URL object. Not NULL.
+ */
+LXB_API void
+lxb_url_path_set_null(lxb_url_t *url);
+
+/*
+ * Set the host to the empty host.
+ *
+ * Frees any domain or opaque host buffer using mraw and sets the host type
+ * to LXB_URL_HOST_TYPE_EMPTY.
+ *
+ * @param[in, out] Host object. Not NULL.
+ * @param[in] Memory object associated with the host. Not NULL.
+ */
+LXB_API void
+lxb_url_host_set_empty(lxb_url_host_t *host, lexbor_mraw_t *mraw);
+
+/*
+ * Set the URL query to null.
+ *
+ * Frees the query buffer using url->mraw. Does nothing if the query
+ * is already null.
+ *
+ * @param[in, out] URL object. Not NULL.
+ */
+LXB_API void
+lxb_url_query_set_null(lxb_url_t *url);
+
+/*
+ * Set the URL fragment to null.
+ *
+ * Frees the fragment buffer using url->mraw. Does nothing if the
+ * fragment is already null.
+ *
+ * @param[in, out] URL object. Not NULL.
+ */
+LXB_API void
+lxb_url_fragment_set_null(lxb_url_t *url);
+
 /*
  * Inline functions.
  */
diff --git a/ext/lexbor/patches/0001-Expose-line-and-column-information-for-use-in-PHP.patch b/ext/lexbor/patches/0001-Expose-line-and-column-information-for-use-in-PHP.patch
index 9aec14cca5d..56079940029 100644
--- a/ext/lexbor/patches/0001-Expose-line-and-column-information-for-use-in-PHP.patch
+++ b/ext/lexbor/patches/0001-Expose-line-and-column-information-for-use-in-PHP.patch
@@ -1,7 +1,7 @@
 From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
 From: Niels Dossche <7771979+nielsdos@users.noreply.github.com>
 Date: Sat, 26 Aug 2023 15:08:59 +0200
-Subject: [PATCH 01/12] Expose line and column information for use in PHP
+Subject: [PATCH 01/21] Expose line and column information for use in PHP

 ---
  source/lexbor/dom/interfaces/node.h  |  2 ++
diff --git a/ext/lexbor/patches/0002-Track-implied-added-nodes-for-options-use-in-PHP.patch b/ext/lexbor/patches/0002-Track-implied-added-nodes-for-options-use-in-PHP.patch
index 9f4da029446..8dc8cb984d2 100644
--- a/ext/lexbor/patches/0002-Track-implied-added-nodes-for-options-use-in-PHP.patch
+++ b/ext/lexbor/patches/0002-Track-implied-added-nodes-for-options-use-in-PHP.patch
@@ -1,7 +1,7 @@
 From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
 From: Niels Dossche <7771979+nielsdos@users.noreply.github.com>
 Date: Mon, 14 Aug 2023 20:18:51 +0200
-Subject: [PATCH 02/12] Track implied added nodes for options use in PHP
+Subject: [PATCH 02/21] Track implied added nodes for options use in PHP

 ---
  source/lexbor/html/tree.h                            | 3 +++
diff --git a/ext/lexbor/patches/0003-Patch-utilities-and-data-structure-to-be-able-to-gen.patch b/ext/lexbor/patches/0003-Patch-utilities-and-data-structure-to-be-able-to-gen.patch
index fe7068d9bda..f93d9fe8f86 100644
--- a/ext/lexbor/patches/0003-Patch-utilities-and-data-structure-to-be-able-to-gen.patch
+++ b/ext/lexbor/patches/0003-Patch-utilities-and-data-structure-to-be-able-to-gen.patch
@@ -1,7 +1,7 @@
 From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
 From: Niels Dossche <7771979+nielsdos@users.noreply.github.com>
 Date: Thu, 24 Aug 2023 22:57:48 +0200
-Subject: [PATCH 03/12] Patch utilities and data structure to be able to
+Subject: [PATCH 03/21] Patch utilities and data structure to be able to
  generate smaller lookup tables

 Changed the generation script to check if everything fits in 32-bits.
diff --git a/ext/lexbor/patches/0004-Remove-unused-upper-case-tag-static-data.patch b/ext/lexbor/patches/0004-Remove-unused-upper-case-tag-static-data.patch
index 35482388531..35bec95e6b9 100644
--- a/ext/lexbor/patches/0004-Remove-unused-upper-case-tag-static-data.patch
+++ b/ext/lexbor/patches/0004-Remove-unused-upper-case-tag-static-data.patch
@@ -1,7 +1,7 @@
 From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
 From: Niels Dossche <7771979+nielsdos@users.noreply.github.com>
 Date: Wed, 29 Nov 2023 21:26:47 +0100
-Subject: [PATCH 04/12] Remove unused upper case tag static data
+Subject: [PATCH 04/21] Remove unused upper case tag static data

 ---
  source/lexbor/tag/res.h | 2 ++
diff --git a/ext/lexbor/patches/0005-Shrink-size-of-static-binary-search-tree.patch b/ext/lexbor/patches/0005-Shrink-size-of-static-binary-search-tree.patch
index 8e9b9524104..68f2d4d379e 100644
--- a/ext/lexbor/patches/0005-Shrink-size-of-static-binary-search-tree.patch
+++ b/ext/lexbor/patches/0005-Shrink-size-of-static-binary-search-tree.patch
@@ -1,7 +1,7 @@
 From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
 From: Niels Dossche <7771979+nielsdos@users.noreply.github.com>
 Date: Wed, 29 Nov 2023 21:29:31 +0100
-Subject: [PATCH 05/12] Shrink size of static binary search tree
+Subject: [PATCH 05/21] Shrink size of static binary search tree

 This also makes it more efficient on the data cache.
 ---
diff --git a/ext/lexbor/patches/0006-Patch-out-unused-CSS-style-code.patch b/ext/lexbor/patches/0006-Patch-out-unused-CSS-style-code.patch
index 0f4e4cd8661..5d63d17116f 100644
--- a/ext/lexbor/patches/0006-Patch-out-unused-CSS-style-code.patch
+++ b/ext/lexbor/patches/0006-Patch-out-unused-CSS-style-code.patch
@@ -1,7 +1,7 @@
 From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
 From: Niels Dossche <7771979+nielsdos@users.noreply.github.com>
 Date: Sun, 7 Jan 2024 21:59:28 +0100
-Subject: [PATCH 06/12] Patch out unused CSS style code
+Subject: [PATCH 06/21] Patch out unused CSS style code

 ---
  source/lexbor/css/rule.h | 2 ++
diff --git a/ext/lexbor/patches/0007-URL-fixed-setters-for-empty-hosts.patch b/ext/lexbor/patches/0007-URL-fixed-setters-for-empty-hosts.patch
index 44f0f918458..3e69d07f95c 100644
--- a/ext/lexbor/patches/0007-URL-fixed-setters-for-empty-hosts.patch
+++ b/ext/lexbor/patches/0007-URL-fixed-setters-for-empty-hosts.patch
@@ -1,7 +1,7 @@
 From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
 From: Alexander Borisov <lex.borisov@gmail.com>
 Date: Fri, 26 Jun 2026 18:55:56 +0300
-Subject: [PATCH 07/12] URL: fixed setters for empty hosts.
+Subject: [PATCH 07/21] URL: fixed setters for empty hosts.
 MIME-Version: 1.0
 Content-Type: text/plain; charset=UTF-8
 Content-Transfer-Encoding: 8bit
diff --git a/ext/lexbor/patches/0008-URL-fixed-uninitialized-memory-in-the-path-buffer-gr.patch b/ext/lexbor/patches/0008-URL-fixed-uninitialized-memory-in-the-path-buffer-gr.patch
index df1f6a1c2ac..3f7d7e1eab3 100644
--- a/ext/lexbor/patches/0008-URL-fixed-uninitialized-memory-in-the-path-buffer-gr.patch
+++ b/ext/lexbor/patches/0008-URL-fixed-uninitialized-memory-in-the-path-buffer-gr.patch
@@ -1,7 +1,7 @@
 From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
 From: Alexander Borisov <lex.borisov@gmail.com>
 Date: Fri, 5 Jun 2026 22:13:32 +0300
-Subject: [PATCH 08/12] URL: fixed uninitialized memory in the path buffer
+Subject: [PATCH 08/21] URL: fixed uninitialized memory in the path buffer
  growth.

 When a path was long enough to outgrow the on-stack buffer, the first
diff --git a/ext/lexbor/patches/0009-Fix-parsing-for-URL-containing-empty-host-and-userin.patch b/ext/lexbor/patches/0009-Fix-parsing-for-URL-containing-empty-host-and-userin.patch
index 9abcd37fe3f..c94e76622a4 100644
--- a/ext/lexbor/patches/0009-Fix-parsing-for-URL-containing-empty-host-and-userin.patch
+++ b/ext/lexbor/patches/0009-Fix-parsing-for-URL-containing-empty-host-and-userin.patch
@@ -1,7 +1,7 @@
 From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
 From: =?UTF-8?q?M=C3=A1t=C3=A9=20Kocsis?= <kocsismate@woohoolabs.com>
 Date: Thu, 9 Jul 2026 21:51:05 +0200
-Subject: [PATCH 09/12] Fix parsing for URL containing empty host and userinfo
+Subject: [PATCH 09/21] Fix parsing for URL containing empty host and userinfo

 The returned error code (LXB_URL_ERROR_TYPE_INVALID_CREDENTIALS) apparently contradicts the specification:

diff --git a/ext/lexbor/patches/0010-Percent-encode-the-caret-in-the-path.patch b/ext/lexbor/patches/0010-Percent-encode-the-caret-in-the-path.patch
index 48777d5de6d..69156c42159 100644
--- a/ext/lexbor/patches/0010-Percent-encode-the-caret-in-the-path.patch
+++ b/ext/lexbor/patches/0010-Percent-encode-the-caret-in-the-path.patch
@@ -1,7 +1,7 @@
 From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
 From: =?UTF-8?q?M=C3=A1t=C3=A9=20Kocsis?= <kocsismate@woohoolabs.com>
 Date: Fri, 10 Jul 2026 22:31:16 +0200
-Subject: [PATCH 10/12] Percent-encode the caret in the path
+Subject: [PATCH 10/21] Percent-encode the caret in the path

 The caret (^) is part of the path percent-encode set:

diff --git a/ext/lexbor/patches/0011-CSS-fixed-heap-buffer-overflow-in-lexbor-contains-pa.patch b/ext/lexbor/patches/0011-CSS-fixed-heap-buffer-overflow-in-lexbor-contains-pa.patch
index 3928daa7eab..32d52517acb 100644
--- a/ext/lexbor/patches/0011-CSS-fixed-heap-buffer-overflow-in-lexbor-contains-pa.patch
+++ b/ext/lexbor/patches/0011-CSS-fixed-heap-buffer-overflow-in-lexbor-contains-pa.patch
@@ -1,7 +1,7 @@
 From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
 From: Alexander Borisov <lex.borisov@gmail.com>
 Date: Fri, 5 Jun 2026 22:34:23 +0300
-Subject: [PATCH 11/12] CSS: fixed heap buffer overflow in :lexbor-contains()
+Subject: [PATCH 11/21] CSS: fixed heap buffer overflow in :lexbor-contains()
  parsing.

 The contains string buffer was allocated by the size of the string
diff --git a/ext/lexbor/patches/0012-Encoding-fixed-buffer-overflows-in-malformed-decode-.patch b/ext/lexbor/patches/0012-Encoding-fixed-buffer-overflows-in-malformed-decode-.patch
index 32f3aff4ee7..6214cfbda10 100644
--- a/ext/lexbor/patches/0012-Encoding-fixed-buffer-overflows-in-malformed-decode-.patch
+++ b/ext/lexbor/patches/0012-Encoding-fixed-buffer-overflows-in-malformed-decode-.patch
@@ -1,7 +1,7 @@
 From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
 From: Alexander Borisov <lex.borisov@gmail.com>
 Date: Wed, 10 Jun 2026 19:50:10 +0300
-Subject: [PATCH 12/12] Encoding: fixed buffer overflows in malformed decode
+Subject: [PATCH 12/21] Encoding: fixed buffer overflows in malformed decode
  replay.

 Fixed out-of-bounds writes in buffering decoders when replacement output
diff --git a/ext/lexbor/patches/0013-URL-fixed-tail-pointer-in-URLSearchParams-for-delimi.patch b/ext/lexbor/patches/0013-URL-fixed-tail-pointer-in-URLSearchParams-for-delimi.patch
new file mode 100644
index 00000000000..89adb8132fc
--- /dev/null
+++ b/ext/lexbor/patches/0013-URL-fixed-tail-pointer-in-URLSearchParams-for-delimi.patch
@@ -0,0 +1,29 @@
+From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
+From: Alexander Borisov <lex.borisov@gmail.com>
+Date: Fri, 5 Jun 2026 21:46:26 +0300
+Subject: [PATCH 13/21] URL: fixed tail pointer in URLSearchParams for
+ delimiter-free query.
+
+When a query had a single token without '=' or '&' (e.g. "?abc"), the
+internal tail pointer wasn't updated, so a later append() could lose the
+added parameter (and write through a stale pointer). Fixed by keeping the
+tail pointer in sync.
+
+Per report from Xiansheng Cao (@HMF2021)
+---
+ source/lexbor/url/url.c | 2 ++
+ 1 file changed, 2 insertions(+)
+
+diff --git a/source/lexbor/url/url.c b/source/lexbor/url/url.c
+index de19239..fcae2d6 100644
+--- a/source/lexbor/url/url.c
++++ b/source/lexbor/url/url.c
+@@ -5106,6 +5106,8 @@ lxb_url_search_params_parse(lxb_url_search_params_t *search_params,
+             return status;
+         }
+
++        last = entry;
++
+         lexbor_str_init(&entry->value, mraw, 0);
+         if (entry->value.data == NULL) {
+             return LXB_STATUS_ERROR_MEMORY_ALLOCATION;
diff --git a/ext/lexbor/patches/0014-URL-Fix-parsing-of-usernames-containing.patch b/ext/lexbor/patches/0014-URL-Fix-parsing-of-usernames-containing.patch
new file mode 100644
index 00000000000..9aa3b403994
--- /dev/null
+++ b/ext/lexbor/patches/0014-URL-Fix-parsing-of-usernames-containing.patch
@@ -0,0 +1,38 @@
+From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
+From: =?UTF-8?q?Tim=20D=C3=BCsterhus?= <tim@bastelstu.be>
+Date: Fri, 31 Jul 2026 20:56:18 +0200
+Subject: [PATCH 14/21] URL: Fix parsing of usernames containing `@`
+
+Fixes lexbor/lexbor#399.
+---
+ source/lexbor/url/url.c | 17 +++++++----------
+ 1 file changed, 7 insertions(+), 10 deletions(-)
+
+diff --git a/source/lexbor/url/url.c b/source/lexbor/url/url.c
+index fcae2d6..654e2e6 100644
+--- a/source/lexbor/url/url.c
++++ b/source/lexbor/url/url.c
+@@ -1753,16 +1753,13 @@ again:
+                         break;
+                     }
+
+-                    if (pswd == NULL || !at_sign) {
+-                        tmp = (pswd != NULL) ? pswd - 1 : p;
+-
+-                        if (tmp > begin) {
+-                            status = lxb_url_percent_encode_after_utf_8(begin, tmp,
+-                                                        &url->username, url->mraw,
+-                                                        LXB_URL_MAP_USERINFO, false);
+-                            if (status != LXB_STATUS_OK) {
+-                                lxb_url_parse_return(orig_data, buf, status);
+-                            }
++                    tmp = (pswd != NULL) ? pswd - 1 : p;
++                    if (tmp > begin) {
++                        status = lxb_url_percent_encode_after_utf_8(begin, tmp,
++                                                    &url->username, url->mraw,
++                                                    LXB_URL_MAP_USERINFO, false);
++                        if (status != LXB_STATUS_OK) {
++                            lxb_url_parse_return(orig_data, buf, status);
+                         }
+                     }
+
diff --git a/ext/lexbor/patches/0015-Unicode-fixed-uninitialized-memory-in-IDNA-buffer-gr.patch b/ext/lexbor/patches/0015-Unicode-fixed-uninitialized-memory-in-IDNA-buffer-gr.patch
new file mode 100644
index 00000000000..177022aaa53
--- /dev/null
+++ b/ext/lexbor/patches/0015-Unicode-fixed-uninitialized-memory-in-IDNA-buffer-gr.patch
@@ -0,0 +1,81 @@
+From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
+From: Alexander Borisov <lex.borisov@gmail.com>
+Date: Mon, 7 Sep 2026 22:11:39 +0300
+Subject: [PATCH 15/21] Unicode: fixed uninitialized memory in IDNA buffer
+ growth.
+
+When an IDNA buffer outgrew the stack allocation, the move to the heap
+did not copy the existing contents. The converted domain could therefore
+contain uninitialized heap data.
+
+Fixed copying of the codepoint, ASCII and UTF-8 buffers.
+
+Per report from Muhammad Daffa (@daffainfo).
+---
+ source/lexbor/unicode/idna.c | 28 +++++++++++++++++++---------
+ 1 file changed, 19 insertions(+), 9 deletions(-)
+
+diff --git a/source/lexbor/unicode/idna.c b/source/lexbor/unicode/idna.c
+index 754f6b2..b31ae33 100644
+--- a/source/lexbor/unicode/idna.c
++++ b/source/lexbor/unicode/idna.c
+@@ -117,12 +117,14 @@ lxb_unicode_idna_realloc(lxb_codepoint_t *buf, const lxb_codepoint_t *buffer,
+     lxb_codepoint_t *tmp;
+
+     nlen = ((*buf_end - buf) * 4) + len;
+-
++
+     if (buf == buffer) {
+         tmp = lexbor_malloc(nlen * sizeof(lxb_codepoint_t));
+         if (tmp == NULL) {
+             return NULL;
+         }
++
++        memcpy(tmp, buf, (*buf_p - buf) * sizeof(lxb_codepoint_t));
+     }
+     else {
+         tmp = lexbor_realloc(buf, nlen * sizeof(lxb_codepoint_t));
+@@ -458,13 +460,17 @@ lxb_unicode_idna_ascii_puny_cb(const lxb_char_t *data, size_t length, void *ctx,
+
+         if (asc->buf == asc->buffer) {
+             tmp = lexbor_malloc(nlen);
++            if (tmp == NULL) {
++                return LXB_STATUS_ERROR_MEMORY_ALLOCATION;
++            }
++
++            memcpy(tmp, asc->buf, asc->p - asc->buf);
+         }
+         else {
+             tmp = lexbor_realloc(asc->buf, nlen);
+-        }
+-
+-        if (tmp == NULL) {
+-            return LXB_STATUS_ERROR_MEMORY_ALLOCATION;
++            if (tmp == NULL) {
++                return LXB_STATUS_ERROR_MEMORY_ALLOCATION;
++            }
+         }
+
+         asc->p = tmp + (asc->p - asc->buf);
+@@ -711,13 +717,17 @@ lxb_unicode_idna_to_unicode_cb(const lxb_codepoint_t *part, size_t len,
+
+         if (asc->buf == asc->buffer) {
+             tmp = lexbor_malloc(nlen);
++            if (tmp == NULL) {
++                return LXB_STATUS_ERROR_MEMORY_ALLOCATION;
++            }
++
++            memcpy(tmp, asc->buf, asc->p - asc->buf);
+         }
+         else {
+             tmp = lexbor_realloc(asc->buf, nlen);
+-        }
+-
+-        if (tmp == NULL) {
+-            return LXB_STATUS_ERROR_MEMORY_ALLOCATION;
++            if (tmp == NULL) {
++                return LXB_STATUS_ERROR_MEMORY_ALLOCATION;
++            }
+         }
+
+         asc->p = tmp + (asc->p - asc->buf);
diff --git a/ext/lexbor/patches/0016-URL-encode-opaque-path-spaces-before-query-and-fragm.patch b/ext/lexbor/patches/0016-URL-encode-opaque-path-spaces-before-query-and-fragm.patch
new file mode 100644
index 00000000000..4484e113231
--- /dev/null
+++ b/ext/lexbor/patches/0016-URL-encode-opaque-path-spaces-before-query-and-fragm.patch
@@ -0,0 +1,38 @@
+From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
+From: =?UTF-8?q?M=C3=A1t=C3=A9=20Kocsis?= <kocsismate@woohoolabs.com>
+Date: Fri, 11 Sep 2026 21:58:45 +0200
+Subject: [PATCH 16/21] URL: encode opaque path spaces before query and
+ fragment. (#405)
+MIME-Version: 1.0
+Content-Type: text/plain; charset=UTF-8
+Content-Transfer-Encoding: 8bit
+
+Percent-encode only the space immediately preceding a query or fragment delimiter, while preserving validation errors for every parsed space.
+
+This way, Lexbor will correctly follow "If remaining starts with U+003F (?) or U+0023 (#), then append "%20" to url’s path." in the "opaque path state".
+---
+ source/lexbor/url/url.c | 11 +++++++++++
+ 1 file changed, 11 insertions(+)
+
+diff --git a/source/lexbor/url/url.c b/source/lexbor/url/url.c
+index 654e2e6..98ee304 100644
+--- a/source/lexbor/url/url.c
++++ b/source/lexbor/url/url.c
+@@ -2340,6 +2340,17 @@ again:
+                     lxb_url_parse_return(orig_data, buf, status);
+                 }
+
++                /* Encode only the space immediately before a query or fragment. */
++                if (p > begin && p[-1] == ' ') {
++                    tmp_str.length--;
++                    if (lexbor_str_append(&tmp_str, url->mraw,
++                                          (const lxb_char_t *) "%20", 3) == NULL)
++                    {
++                        lxb_url_parse_return(orig_data, buf,
++                                             LXB_STATUS_ERROR_MEMORY_ALLOCATION);
++                    }
++                }
++
+                 status = lxb_url_path_list_push(url, &tmp_str);
+                 if (status != LXB_STATUS_OK) {
+                     lxb_url_parse_return(orig_data, buf, status);
diff --git a/ext/lexbor/patches/0017-URL-Fix-lxb_url_serialize_fragment-without-a-query-4.patch b/ext/lexbor/patches/0017-URL-Fix-lxb_url_serialize_fragment-without-a-query-4.patch
new file mode 100644
index 00000000000..98139927410
--- /dev/null
+++ b/ext/lexbor/patches/0017-URL-Fix-lxb_url_serialize_fragment-without-a-query-4.patch
@@ -0,0 +1,25 @@
+From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
+From: =?UTF-8?q?Tim=20D=C3=BCsterhus?= <tim@tideways-gmbh.com>
+Date: Fri, 11 Sep 2026 22:45:33 +0200
+Subject: [PATCH 17/21] URL: Fix `lxb_url_serialize_fragment()` without a query
+ (#410)
+
+The `lxb_url_serialize_fragment()` function checked if the query contained data
+before correctly serializing the fragment. Check for the fragment instead.
+---
+ source/lexbor/url/url.c | 2 +-
+ 1 file changed, 1 insertion(+), 1 deletion(-)
+
+diff --git a/source/lexbor/url/url.c b/source/lexbor/url/url.c
+index 98ee304..faf553b 100644
+--- a/source/lexbor/url/url.c
++++ b/source/lexbor/url/url.c
+@@ -4915,7 +4915,7 @@ lxb_status_t
+ lxb_url_serialize_fragment(const lxb_url_t *url,
+                            lexbor_serialize_cb_f cb, void *ctx)
+ {
+-    if (url->query.data != NULL) {
++    if (url->fragment.data != NULL) {
+         return cb(url->fragment.data, url->fragment.length, ctx);
+     }
+
diff --git a/ext/lexbor/patches/0018-URL-expose-component-reset-functions.-415.patch b/ext/lexbor/patches/0018-URL-expose-component-reset-functions.-415.patch
new file mode 100644
index 00000000000..c259de69dff
--- /dev/null
+++ b/ext/lexbor/patches/0018-URL-expose-component-reset-functions.-415.patch
@@ -0,0 +1,110 @@
+From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
+From: =?UTF-8?q?M=C3=A1t=C3=A9=20Kocsis?= <kocsismate@woohoolabs.com>
+Date: Tue, 22 Sep 2026 22:20:28 +0200
+Subject: [PATCH 18/21] URL: expose component reset functions. (#415)
+
+* URL: expose component reset functions.
+
+Make the path, host and fragment reset functions public and add lxb_url_query_set_null().
+
+* Add documentation for the newly exposed functions
+---
+ source/lexbor/url/url.c | 14 ++++++++++---
+ source/lexbor/url/url.h | 45 +++++++++++++++++++++++++++++++++++++++++
+ 2 files changed, 56 insertions(+), 3 deletions(-)
+
+diff --git a/source/lexbor/url/url.c b/source/lexbor/url/url.c
+index faf553b..9bfc2aa 100644
+--- a/source/lexbor/url/url.c
++++ b/source/lexbor/url/url.c
+@@ -923,7 +923,7 @@ lxb_url_scheme_copy_special(const lxb_url_scheme_data_t *src,
+     return lxb_url_str_copy(&src->name, &dst->name, dst_mraw);
+ }
+
+-static void
++void
+ lxb_url_path_set_null(lxb_url_t *url)
+ {
+     if (url->path.str.data == NULL) {
+@@ -1147,7 +1147,7 @@ lxb_url_host_destroy(lxb_url_host_t *host, lexbor_mraw_t *mraw)
+     }
+ }
+
+-static void
++void
+ lxb_url_host_set_empty(lxb_url_host_t *host, lexbor_mraw_t *mraw)
+ {
+     lxb_url_host_destroy(host, mraw);
+@@ -1197,7 +1197,15 @@ lxb_url_port_set(lxb_url_t *url, uint16_t port)
+     url->has_port = true;
+ }
+
+-static void
++void
++lxb_url_query_set_null(lxb_url_t *url)
++{
++    if (url->query.data != NULL) {
++        (void) lexbor_str_destroy(&url->query, url->mraw, false);
++    }
++}
++
++void
+ lxb_url_fragment_set_null(lxb_url_t *url)
+ {
+     if (url->fragment.data != NULL) {
+diff --git a/source/lexbor/url/url.h b/source/lexbor/url/url.h
+index 4ed3f32..b9e4973 100644
+--- a/source/lexbor/url/url.h
++++ b/source/lexbor/url/url.h
+@@ -763,6 +763,51 @@ LXB_API lxb_status_t
+ lxb_url_search_params_serialize(lxb_url_search_params_t *search_params,
+                                 lexbor_callback_f cb, void *ctx);
+
++/*
++ * Reset the URL path to an empty list.
++ *
++ * Frees the path buffer using url->mraw, resets the segment count and clears
++ * the opaque flag. Does nothing if the path buffer is already NULL.
++ *
++ * @param[in, out] URL object. Not NULL.
++ */
++LXB_API void
++lxb_url_path_set_null(lxb_url_t *url);
++
++/*
++ * Set the host to the empty host.
++ *
++ * Frees any domain or opaque host buffer using mraw and sets the host type
++ * to LXB_URL_HOST_TYPE_EMPTY.
++ *
++ * @param[in, out] Host object. Not NULL.
++ * @param[in] Memory object associated with the host. Not NULL.
++ */
++LXB_API void
++lxb_url_host_set_empty(lxb_url_host_t *host, lexbor_mraw_t *mraw);
++
++/*
++ * Set the URL query to null.
++ *
++ * Frees the query buffer using url->mraw. Does nothing if the query
++ * is already null.
++ *
++ * @param[in, out] URL object. Not NULL.
++ */
++LXB_API void
++lxb_url_query_set_null(lxb_url_t *url);
++
++/*
++ * Set the URL fragment to null.
++ *
++ * Frees the fragment buffer using url->mraw. Does nothing if the
++ * fragment is already null.
++ *
++ * @param[in, out] URL object. Not NULL.
++ */
++LXB_API void
++lxb_url_fragment_set_null(lxb_url_t *url);
++
+ /*
+  * Inline functions.
+  */
diff --git a/ext/lexbor/patches/0019-URL-Fix-parsing-of-query-and-fragment-after-a-dot-co.patch b/ext/lexbor/patches/0019-URL-Fix-parsing-of-query-and-fragment-after-a-dot-co.patch
new file mode 100644
index 00000000000..c2dd006b4b8
--- /dev/null
+++ b/ext/lexbor/patches/0019-URL-Fix-parsing-of-query-and-fragment-after-a-dot-co.patch
@@ -0,0 +1,416 @@
+From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
+From: =?UTF-8?q?Tim=20D=C3=BCsterhus?= <tim@tideways-gmbh.com>
+Date: Sat, 26 Sep 2026 18:41:05 +0200
+Subject: [PATCH 19/21] URL: Fix parsing of query and fragment after a
+ dot-component in path (#411)
+MIME-Version: 1.0
+Content-Type: text/plain; charset=UTF-8
+Content-Transfer-Encoding: 8bit
+
+* URL: fixed dot segment terminators in path parsing.
+
+After a "." or ".." path segment the parser skipped the next code point
+unconditionally.  For "?" and "#" this lost the delimiter, so the query
+or the fragment became part of the path: "https://example.com/..#frag"
+was parsed as "https://example.com/frag".  The fast path fix from #411
+did not cover the slow path, which is taken after a code point that must
+be percent-encoded, for example "https://example.com/café/..#frag".
+
+Also fixed in the same code:
+- "\" after a dot segment in a special URL did not report
+  invalid-reverse-solidus.
+- The empty segment was lost in "//./c" and at the fast/slow path
+  handoff ("//é").
+- path.length drifted after dot segments and underflowed after ".." at
+  the root, so a later ".." in the slow path did not shorten the path
+  ("/a/b/../../../c/é/../../x" gave "/c/x").  The file host and path
+  start states did not update it either.
+- An invalid percent sequence in the fast path skipped the next two
+  code points, so a following "/", "\", "?" or "#" was lost ("/%?q" put
+  "?q" into the path).  Such a segment is now handed to the slow path,
+  which checks "%" one code point at a time.
+
+Per reports from Tim Düsterhus (@TimWolla) and @NickSdot.
+
+This relates to #409 issue on GitHub.
+This relates to #412 issue on GitHub.
+This relates to #411 PR on GitHub.
+
+* URL: added regression tests for dot segments in path.
+
+Added tests for "." and ".." path segments followed by "?", "#", "\"
+and the end of input, in both the fast and the slow path, for the
+invalid-reverse-solidus validation error, and for path.length against
+the serialized path.  Also added tests for invalid percent sequences
+before dot segments and delimiters, both in parsing and in the pathname
+setter.
+
+This relates to #409 issue on GitHub.
+This relates to #412 issue on GitHub.
+---
+ source/lexbor/url/url.c | 200 +++++++++++++++++++++++-----------------
+ 1 file changed, 113 insertions(+), 87 deletions(-)
+
+diff --git a/source/lexbor/url/url.c b/source/lexbor/url/url.c
+index 9bfc2aa..55fe11f 100644
+--- a/source/lexbor/url/url.c
++++ b/source/lexbor/url/url.c
+@@ -546,15 +546,15 @@ lxb_url_path_slow_path(lxb_url_parser_t *parser, lxb_url_t *url,
+                        const lxb_char_t *data, const lxb_char_t *end, bool bqs);
+
+ static lxb_status_t
+-lxb_url_path_try_dot(lxb_url_t *url, const lxb_char_t **begin,
+-                     const lxb_char_t **last, const lxb_char_t **start,
+-                     const lxb_char_t *end, bool bqs);
++lxb_url_path_try_dot(lxb_url_parser_t *parser, lxb_url_t *url,
++                     const lxb_char_t **begin, const lxb_char_t **last,
++                     const lxb_char_t **start, const lxb_char_t *end, bool bqs);
+
+ static const lxb_char_t *
+-lxb_url_path_dot_count(lxb_url_t *url, const lxb_char_t *p,
+-                       const lxb_char_t *end, const lxb_char_t *sbuf_begin,
+-                       lxb_char_t **sbuf, lxb_char_t **last, size_t *path_count,
+-                       bool bqs);
++lxb_url_path_dot_count(lxb_url_parser_t *parser, lxb_url_t *url,
++                       const lxb_char_t *p, const lxb_char_t *end,
++                       const lxb_char_t *sbuf_begin, lxb_char_t **sbuf,
++                       lxb_char_t **last, size_t *path_count, bool bqs);
+
+ static void
+ lxb_url_path_fix_windows_drive(lxb_url_t *url, lxb_char_t *sbuf,
+@@ -990,23 +990,25 @@ lxb_url_path_shorten(lxb_url_t *url)
+         }
+     }
+
+-    if (url->path.str.data != NULL) {
+-        url->path.length -= 1;
++    if (url->path.length == 0 || str->data == NULL) {
++        return;
++    }
+
+-        begin = str->data;
+-        p = begin + str->length;
++    url->path.length -= 1;
+
+-        while (p > begin) {
+-            p -= 1;
++    begin = str->data;
++    p = begin + str->length;
+
+-            if (*p == '/') {
+-                *p = '\0';
+-                break;
+-            }
+-        }
++    while (p > begin) {
++        p -= 1;
+
+-        str->length = p - begin;
++        if (*p == '/') {
++            *p = '\0';
++            break;
++        }
+     }
++
++    str->length = p - begin;
+ }
+
+ static lxb_status_t
+@@ -2146,6 +2148,8 @@ again:
+                     if (status != LXB_STATUS_OK) {
+                         lxb_url_parse_return(orig_data, buf, status);
+                     }
++
++                    url->path.length += 1;
+                 }
+             }
+         }
+@@ -2287,7 +2291,13 @@ again:
+             && url->host.type == LXB_URL_HOST_TYPE__UNDEF)
+         {
+             status = lxb_url_path_append(url, mp_str.data, mp_str.length);
+-            lxb_url_parse_return(orig_data, buf, status);
++            if (status != LXB_STATUS_OK) {
++                lxb_url_parse_return(orig_data, buf, status);
++            }
++
++            url->path.length += 1;
++
++            lxb_url_parse_return(orig_data, buf, LXB_STATUS_OK);
+         }
+
+         lxb_url_parse_return(orig_data, buf, LXB_STATUS_OK);
+@@ -2535,13 +2545,8 @@ lxb_url_path_fast_path(lxb_url_parser_t *parser, lxb_url_t *url,
+                     || lexbor_str_res_map_hex[p[1]] == 0xff
+                     || lexbor_str_res_map_hex[p[2]] == 0xff)
+                 {
+-                    status = lxb_url_log_append(parser, p,
+-                                                LXB_URL_ERROR_TYPE_INVALID_URL_UNIT);
+-                    if (status != LXB_STATUS_OK) {
+-                        return NULL;
+-                    }
+-
+-                    p = (end - p < 3) ? end - 1 : p + 2;
++                    /* Reprocess the segment without skipping delimiters. */
++                    goto slow;
+                 }
+                 else if (p[1] == '2' && (p[2] == 'e' || p[2] == 'E')
+                          && (p == begin
+@@ -2550,8 +2555,8 @@ lxb_url_path_fast_path(lxb_url_parser_t *parser, lxb_url_t *url,
+                 {
+                     url->path.length = count;
+
+-                    status = lxb_url_path_try_dot(url, &begin, &last,
+-                                                  &p, end, bqs);
++                    status = lxb_url_path_try_dot(parser, url, &begin,
++                                                  &last, &p, end, bqs);
+                     if (status != LXB_STATUS_OK) {
+                         return NULL;
+                     }
+@@ -2589,8 +2594,8 @@ lxb_url_path_fast_path(lxb_url_parser_t *parser, lxb_url_t *url,
+                 {
+                     url->path.length = count;
+
+-                    status = lxb_url_path_try_dot(url, &begin, &last,
+-                                                  &p, end, bqs);
++                    status = lxb_url_path_try_dot(parser, url, &begin,
++                                                  &last, &p, end, bqs);
+                     if (status != LXB_STATUS_OK) {
+                         return NULL;
+                     }
+@@ -2599,17 +2604,7 @@ lxb_url_path_fast_path(lxb_url_parser_t *parser, lxb_url_t *url,
+                 }
+             }
+             else {
+-                url->path.length = count;
+-
+-                if (last - 1 > begin) {
+-                    status = lxb_url_path_append(url, begin,
+-                                                 (last - 1) - begin);
+-                    if (status != LXB_STATUS_OK) {
+-                        return NULL;
+-                    }
+-                }
+-
+-                return lxb_url_path_slow_path(parser, url, last, end, bqs);
++                goto slow;
+             }
+         }
+     }
+@@ -2619,13 +2614,22 @@ lxb_url_path_fast_path(lxb_url_parser_t *parser, lxb_url_t *url,
+         return NULL;
+     }
+
+-    if (count == 0 || p != begin) {
+-        count += 1;
+-    }
++    url->path.length = count + 1;
++
++    return p;
++
++slow:
+
+     url->path.length = count;
+
+-    return p;
++    if (last > begin) {
++        status = lxb_url_path_append(url, begin, (last - 1) - begin);
++        if (status != LXB_STATUS_OK) {
++            return NULL;
++        }
++    }
++
++    return lxb_url_path_slow_path(parser, url, last, end, bqs);
+ }
+
+ /*
+@@ -2721,10 +2725,6 @@ lxb_url_path_slow_path(lxb_url_parser_t *parser, lxb_url_t *url,
+
+             count += 1;
+             last = sbuf;
+-
+-            if (p + 1 >= end) {
+-                count += 1;
+-            }
+         }
+         else if (c == '\\' && lxb_url_is_special(url)) {
+             status = lxb_url_log_append(parser, p,
+@@ -2742,16 +2742,8 @@ lxb_url_path_slow_path(lxb_url_parser_t *parser, lxb_url_t *url,
+
+             count += 1;
+             last = sbuf;
+-
+-            if (p + 1 >= end) {
+-                count += 1;
+-            }
+         }
+         else if ((c == '?' || c == '#') && bqs) {
+-            lxb_url_path_fix_windows_drive(url, last, sbuf, count);
+-
+-            count += 1;
+-            last = sbuf;
+             break;
+         }
+         else if (lxb_url_map[c] & LXB_URL_MAP_PATH) {
+@@ -2771,11 +2763,19 @@ lxb_url_path_slow_path(lxb_url_parser_t *parser, lxb_url_t *url,
+         }
+         else if (c == '.') {
+             if (last == sbuf) {
+-                tmp = lxb_url_path_dot_count(url, p, end, sbuf_begin,
++                tmp = lxb_url_path_dot_count(parser, url, p, end, sbuf_begin,
+                                              &sbuf, &last, &count, bqs);
++                if (tmp == NULL) {
++                    goto failed;
++                }
+
+                 if (tmp != p) {
+-                    p = tmp + 1;
++                    /* Skip '/' or '\', but leave '?' and '#' to the loop. */
++                    if (tmp < end && *tmp != '?' && *tmp != '#') {
++                        tmp += 1;
++                    }
++
++                    p = tmp;
+                     continue;
+                 }
+             }
+@@ -2800,11 +2800,19 @@ lxb_url_path_slow_path(lxb_url_parser_t *parser, lxb_url_t *url,
+             else if (p[1] == '2' && (p[2] == 'e' || p[2] == 'E')
+                      && last == sbuf)
+             {
+-                tmp = lxb_url_path_dot_count(url, p, end, sbuf_begin,
++                tmp = lxb_url_path_dot_count(parser, url, p, end, sbuf_begin,
+                                              &sbuf, &last, &count, bqs);
++                if (tmp == NULL) {
++                    goto failed;
++                }
+
+                 if (tmp != p) {
+-                    p = tmp + 1;
++                    /* Skip '/' or '\', but leave '?' and '#' to the loop. */
++                    if (tmp < end && *tmp != '?' && *tmp != '#') {
++                        tmp += 1;
++                    }
++
++                    p = tmp;
+                     continue;
+                 }
+             }
+@@ -2833,12 +2841,9 @@ lxb_url_path_slow_path(lxb_url_parser_t *parser, lxb_url_t *url,
+         p += 1;
+     }
+
+-    if (count == 0 || last < sbuf) {
+-        lxb_url_path_fix_windows_drive(url, last, sbuf, count);
+-        count += 1;
+-    }
++    lxb_url_path_fix_windows_drive(url, last, sbuf, count);
+
+-    url->path.length = count;
++    url->path.length = count + 1;
+
+     status = lxb_url_path_append_wo_slash(url, sbuf_begin, sbuf - sbuf_begin);
+     if (status != LXB_STATUS_OK) {
+@@ -2861,13 +2866,12 @@ failed:
+ }
+
+ static lxb_status_t
+-lxb_url_path_try_dot(lxb_url_t *url, const lxb_char_t **begin,
+-                     const lxb_char_t **last, const lxb_char_t **start,
+-                     const lxb_char_t *end, bool bqs)
++lxb_url_path_try_dot(lxb_url_parser_t *parser, lxb_url_t *url,
++                     const lxb_char_t **begin, const lxb_char_t **last,
++                     const lxb_char_t **start, const lxb_char_t *end, bool bqs)
+ {
+     unsigned count;
+     lxb_char_t c;
+-    lexbor_str_t *str;
+     lxb_status_t status;
+     const lxb_char_t *p;
+
+@@ -2912,40 +2916,54 @@ lxb_url_path_try_dot(lxb_url_t *url, const lxb_char_t **begin,
+         }
+     }
+
+-    if (p < end) {
+-        *start = p;
+-        *begin = p + 1;
+-        *last = *begin;
++    if (count == 2) {
++        lxb_url_path_shorten(url);
+     }
+-    else {
++
++    if (p >= end) {
++        /* The caller appends the trailing empty segment. */
+         *start = end - 1;
+         *begin = end;
+         *last = end;
++
++        return LXB_STATUS_OK;
+     }
+
+-    if (count == 2) {
+-        lxb_url_path_shorten(url);
++    if (*p == '?' || *p == '#') {
++        /* The caller's loop handles the delimiter and the empty segment. */
++        *start = p - 1;
++        *begin = p;
++        *last = p;
++
++        return LXB_STATUS_OK;
+     }
+-    else if (count == 1) {
+-        str = &url->path.str;
+
+-        if (str->length > 0 && str->data[str->length - 1] == '/') {
+-            str->length -= 1;
+-            str->data[str->length] = '\0';
++    if (*p == '\\') {
++        status = lxb_url_log_append(parser, p,
++                                    LXB_URL_ERROR_TYPE_INVALID_REVERSE_SOLIDUS);
++        if (status != LXB_STATUS_OK) {
++            return status;
+         }
+     }
+
++    /* Skip '/' or '\'. */
++
++    *start = p;
++    *begin = p + 1;
++    *last = *begin;
++
+     return LXB_STATUS_OK;
+ }
+
+ static const lxb_char_t *
+-lxb_url_path_dot_count(lxb_url_t *url, const lxb_char_t *p,
+-                       const lxb_char_t *end, const lxb_char_t *sbuf_begin,
+-                       lxb_char_t **sbuf, lxb_char_t **last, size_t *path_count,
+-                       bool bqs)
++lxb_url_path_dot_count(lxb_url_parser_t *parser, lxb_url_t *url,
++                       const lxb_char_t *p, const lxb_char_t *end,
++                       const lxb_char_t *sbuf_begin, lxb_char_t **sbuf,
++                       lxb_char_t **last, size_t *path_count, bool bqs)
+ {
+     unsigned count;
+     lxb_char_t c, *last_p;
++    lxb_status_t status;
+     const lxb_char_t *begin;
+
+     count = 0;
+@@ -2982,6 +3000,14 @@ lxb_url_path_dot_count(lxb_url_t *url, const lxb_char_t *p,
+         return begin;
+     }
+
++    if (p < end && *p == '\\') {
++        status = lxb_url_log_append(parser, p,
++                                    LXB_URL_ERROR_TYPE_INVALID_REVERSE_SOLIDUS);
++        if (status != LXB_STATUS_OK) {
++            return NULL;
++        }
++    }
++
+     if (url->scheme.type == LXB_URL_SCHEMEL_TYPE_FILE
+         && *path_count == 1
+         && lxb_url_normalized_windows_drive_letter(sbuf_begin + 1, *last - 1))
diff --git a/ext/lexbor/patches/0020-URL-Keep-replacement-file-drive-paths-hierarchical-4.patch b/ext/lexbor/patches/0020-URL-Keep-replacement-file-drive-paths-hierarchical-4.patch
new file mode 100644
index 00000000000..9871cb65064
--- /dev/null
+++ b/ext/lexbor/patches/0020-URL-Keep-replacement-file-drive-paths-hierarchical-4.patch
@@ -0,0 +1,23 @@
+From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
+From: =?UTF-8?q?M=C3=A1t=C3=A9=20Kocsis?= <kocsismate@woohoolabs.com>
+Date: Mon, 21 Sep 2026 20:01:43 +0200
+Subject: [PATCH 20/21] URL: Keep replacement file drive paths hierarchical
+ (#423)
+
+WHATWG file state (https://url.spec.whatwg.org/#file-state) step 4.4.3.2 resets the path to an empty list. Do not mark it opaque, as that makes subsequent pathname updates silently do nothing.
+---
+ source/lexbor/url/url.c | 1 -
+ 1 file changed, 1 deletion(-)
+
+diff --git a/source/lexbor/url/url.c b/source/lexbor/url/url.c
+index 55fe11f..2231816 100644
+--- a/source/lexbor/url/url.c
++++ b/source/lexbor/url/url.c
+@@ -2099,7 +2099,6 @@ again:
+                 }
+
+                 lxb_url_path_set_null(url);
+-                url->path.opaque = true;
+             }
+         }
+
diff --git a/ext/lexbor/patches/0021-URL-normalize-output-encoding-for-percent-encoding.patch b/ext/lexbor/patches/0021-URL-normalize-output-encoding-for-percent-encoding.patch
new file mode 100644
index 00000000000..715d7c8241d
--- /dev/null
+++ b/ext/lexbor/patches/0021-URL-normalize-output-encoding-for-percent-encoding.patch
@@ -0,0 +1,80 @@
+From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
+From: Alexander Borisov <lex.borisov@gmail.com>
+Date: Thu, 24 Sep 2026 15:05:51 +0300
+Subject: [PATCH 21/21] URL: normalize output encoding for percent-encoding.
+
+---
+ source/lexbor/url/url.c | 34 +++++++++++++++++++++++++---------
+ 1 file changed, 25 insertions(+), 9 deletions(-)
+
+diff --git a/source/lexbor/url/url.c b/source/lexbor/url/url.c
+index 2231816..146f0bd 100644
+--- a/source/lexbor/url/url.c
++++ b/source/lexbor/url/url.c
+@@ -1228,6 +1228,26 @@ lxb_url_encoding_init(const lxb_encoding_data_t *encoding,
+     (void) lxb_encoding_encode_init_single(encode, encoding);
+ }
+
++/*
++ * https://encoding.spec.whatwg.org/#get-an-output-encoding
++ */
++lxb_inline lxb_encoding_t
++lxb_url_output_encoding(lxb_encoding_t encoding)
++{
++    switch (encoding) {
++        case LXB_ENCODING_DEFAULT:
++        case LXB_ENCODING_AUTO:
++        case LXB_ENCODING_UNDEFINED:
++        case LXB_ENCODING_REPLACEMENT:
++        case LXB_ENCODING_UTF_16BE:
++        case LXB_ENCODING_UTF_16LE:
++            return LXB_ENCODING_UTF_8;
++
++        default:
++            return encoding;
++    }
++}
++
+ static bool
+ lxb_url_start_windows_drive_letter(const lxb_char_t *data,
+                                    const lxb_char_t *end)
+@@ -1372,12 +1392,7 @@ lxb_url_parse_basic_h(lxb_url_parser_t *parser, lxb_url_t *url,
+         state = override_state;
+     }
+
+-    if (encoding <= LXB_ENCODING_UNDEFINED
+-        || encoding == LXB_ENCODING_UTF_16BE
+-        || encoding == LXB_ENCODING_UTF_16LE)
+-    {
+-        encoding = LXB_ENCODING_UTF_8;
+-    }
++    encoding = lxb_url_output_encoding(encoding);
+
+     enc = lxb_encoding_data(encoding);
+     if (enc == NULL) {
+@@ -3221,7 +3236,7 @@ lxb_url_percent_encode_after_encoding(const lxb_char_t *data,
+     const lxb_char_t *buf_end = buf + sizeof(buffer);
+     static const lexbor_str_t esc_str = lexbor_str("%26%23");
+
+-    if (encoding->encoding == LXB_ENCODING_UTF_8) {
++    if (lxb_url_output_encoding(encoding->encoding) == LXB_ENCODING_UTF_8) {
+         return lxb_url_percent_encode_after_utf_8(data, end, str, mraw,
+                                                   enmap, space_as_plus);
+     }
+@@ -3256,13 +3271,14 @@ lxb_url_percent_encode_after_encoding(const lxb_char_t *data,
+         len = encoding->encode_single(&encode, &buf, buf_end, cp);
+
+         if (len < LXB_ENCODING_ENCODE_OK) {
+-            size = lexbor_conv_int64_to_data((int64_t) cp, buf, buf_end - buf);
++            size = lexbor_conv_int64_to_data((int64_t) cp, buffer,
++                                             sizeof(buffer));
+
+             if (lexbor_str_append(str, mraw, esc_str.data, esc_str.length) == NULL) {
+                 return LXB_STATUS_ERROR_MEMORY_ALLOCATION;
+             }
+
+-            if (lexbor_str_append(str, mraw, buf, size) == NULL) {
++            if (lexbor_str_append(str, mraw, buffer, size) == NULL) {
+                 return LXB_STATUS_ERROR_MEMORY_ALLOCATION;
+             }
+
diff --git a/ext/uri/tests/whatwg/modification/withPath_success_file_invalid_drive_letter.phpt b/ext/uri/tests/whatwg/modification/withPath_success_file_invalid_drive_letter.phpt
new file mode 100644
index 00000000000..39ed5d30a4f
--- /dev/null
+++ b/ext/uri/tests/whatwg/modification/withPath_success_file_invalid_drive_letter.phpt
@@ -0,0 +1,21 @@
+--TEST--
+Test Uri\WhatWg\Url::withPath() - file URL whose path was reset by an invalid drive letter
+--FILE--
+<?php
+
+$url = Uri\WhatWg\Url::parse("c|/x", new Uri\WhatWg\Url("file:///d:/a/b"), $errors);
+
+var_dump($url->getPath());
+var_dump(array_map(static fn (Uri\WhatWg\UrlValidationError $error): string => $error->type->name, $errors));
+var_dump($url->withPath("/zz")->getPath());
+
+?>
+--EXPECT--
+string(5) "/c:/x"
+array(2) {
+  [0]=>
+  string(14) "InvalidUrlUnit"
+  [1]=>
+  string(29) "FileInvalidWindowsDriveLetter"
+}
+string(3) "/zz"
diff --git a/ext/uri/tests/whatwg/parsing/host_success_multibyte_long.phpt b/ext/uri/tests/whatwg/parsing/host_success_multibyte_long.phpt
new file mode 100644
index 00000000000..109cb39de1f
--- /dev/null
+++ b/ext/uri/tests/whatwg/parsing/host_success_multibyte_long.phpt
@@ -0,0 +1,21 @@
+--TEST--
+Test Uri\WhatWg\Url parsing - host - IDN host longer than the IDNA on-stack buffer
+--FILE--
+<?php
+
+$url = Uri\WhatWg\Url::parse("https://" . str_repeat("é", 5000) . ".com/");
+$host = $url->getAsciiHost();
+
+var_dump(strlen($host));
+var_dump(substr($host, 0, 12));
+var_dump(substr($host, -6));
+var_dump(substr_count($host, "a"));
+var_dump($url->getUnicodeHost() === str_repeat("é", 5000) . ".com");
+
+?>
+--EXPECT--
+int(5010)
+string(12) "xn--9caaaaaa"
+string(6) "aa.com"
+int(5000)
+bool(true)
diff --git a/ext/uri/tests/whatwg/parsing/path_success_dot_segment_delimiters.phpt b/ext/uri/tests/whatwg/parsing/path_success_dot_segment_delimiters.phpt
new file mode 100644
index 00000000000..648edbf2653
--- /dev/null
+++ b/ext/uri/tests/whatwg/parsing/path_success_dot_segment_delimiters.phpt
@@ -0,0 +1,29 @@
+--TEST--
+Test Uri\WhatWg\Url parsing - path - query and fragment after a dot segment
+--FILE--
+<?php
+
+foreach ([
+    "https://example.com/..#frag",
+    "https://example.com/caf\u{e9}/..#frag",
+    "https://example.com/..?q=1",
+    "https://example.com/%?q",
+] as $input) {
+    $url = new Uri\WhatWg\Url($input);
+    var_dump($url->getPath(), $url->getQuery(), $url->getFragment());
+}
+
+?>
+--EXPECT--
+string(1) "/"
+NULL
+string(4) "frag"
+string(1) "/"
+NULL
+string(4) "frag"
+string(1) "/"
+string(3) "q=1"
+NULL
+string(2) "/%"
+string(1) "q"
+NULL
diff --git a/ext/uri/tests/whatwg/parsing/path_success_opaque_space_before_query.phpt b/ext/uri/tests/whatwg/parsing/path_success_opaque_space_before_query.phpt
new file mode 100644
index 00000000000..3ffdd94f54c
--- /dev/null
+++ b/ext/uri/tests/whatwg/parsing/path_success_opaque_space_before_query.phpt
@@ -0,0 +1,16 @@
+--TEST--
+Test Uri\WhatWg\Url parsing - path - space before the query or fragment of an opaque path
+--FILE--
+<?php
+
+foreach (["data:x ?q", "data:x #f", "foo:a b ?c", "data:x  ?q"] as $input) {
+    $url = new Uri\WhatWg\Url($input);
+    var_dump($url->getPath());
+}
+
+?>
+--EXPECT--
+string(4) "x%20"
+string(4) "x%20"
+string(6) "a b%20"
+string(5) "x %20"
diff --git a/ext/uri/tests/whatwg/parsing/username_success_at_sign.phpt b/ext/uri/tests/whatwg/parsing/username_success_at_sign.phpt
new file mode 100644
index 00000000000..910d984858f
--- /dev/null
+++ b/ext/uri/tests/whatwg/parsing/username_success_at_sign.phpt
@@ -0,0 +1,18 @@
+--TEST--
+Test Uri\WhatWg\Url parsing - username - at sign in the username and the password
+--FILE--
+<?php
+
+$url = new Uri\WhatWg\Url("http://user@name:pass@word@localhost/");
+
+var_dump($url->getUsername());
+var_dump($url->getPassword());
+var_dump($url->getAsciiHost());
+var_dump($url->toAsciiString());
+
+?>
+--EXPECT--
+string(11) "user%40name"
+string(11) "pass%40word"
+string(9) "localhost"
+string(41) "http://user%40name:pass%40word@localhost/"