Commit 0cf8668eb2e for php

commit 0cf8668eb2ebe36d2aed723b13949d123dd1a13e
Author: Nicolas Grekas <nicolas.grekas@gmail.com>
Date:   Mon Sep 28 09:56:16 2026 +0200

    ext/mbstring: Don't emit surrogates when encoding to UTF-8

    The UTF-8 encoder turned surrogate codepoints into 3-byte sequences like
    "\xED\xA0\x80", which are not valid UTF-8. Such codepoints come from
    UCS-2, UCS-4 or HTML-ENTITIES input, or from mb_decode_numericentity().
    Since mbstring marks the UTF-8 strings it returns as valid UTF-8,
    functions that trust that flag, like mb_check_encoding(), mb_scrub() or
    preg_match(), accepted them too.

    Surrogates are now replaced by the substitute character, as codepoints
    above U+10FFFF already are. The internal mode used by mb_strpos() and
    similar functions still encodes them, so that different surrogates don't
    match each other.

diff --git a/NEWS b/NEWS
index bde051dee66..686eedc4645 100644
--- a/NEWS
+++ b/NEWS
@@ -55,6 +55,8 @@ PHP                                                                        NEWS
 - MBString:
   . Fixed bug GH-23106 (mb_strpos() reads past the end of a haystack ending in
     a truncated UTF-8 sequence). (Lazizbek Ergashev)
+  . Fixed mbstring functions emitting surrogates in UTF-8 output and flagging
+    it as valid UTF-8. (Nicolas Grekas)

 - MySQLi:
   . Fix GH-22854: Fixed failed assertion when accessing mysqli property after
diff --git a/ext/mbstring/libmbfl/filters/mbfilter_utf8.c b/ext/mbstring/libmbfl/filters/mbfilter_utf8.c
index b0a9b6f95a3..e097476178d 100644
--- a/ext/mbstring/libmbfl/filters/mbfilter_utf8.c
+++ b/ext/mbstring/libmbfl/filters/mbfilter_utf8.c
@@ -548,6 +548,12 @@ static void mb_wchar_to_utf8(uint32_t *in, size_t len, mb_convert_buf *buf, bool
 		} else if (w < 0x800) {
 			MB_CONVERT_BUF_ENSURE(buf, out, limit, len + 2);
 			out = mb_convert_buf_add2(out, ((w >> 6) & 0x1F) | 0xC0, (w & 0x3F) | 0x80);
+		} else if (w >= 0xD800 && w <= 0xDFFF && buf->error_mode != MBFL_OUTPUTFILTER_ILLEGAL_MODE_BADUTF8) {
+			/* Surrogate codepoints (which may come from UCS-2, UCS-4 or numeric entities) are not valid in UTF-8.
+			 * BADUTF8 mode is only used internally to search strings; there, encoding them keeps them distinct
+			 * instead of turning them all into the same error marker. */
+			MB_CONVERT_ERROR(buf, out, limit, w, mb_wchar_to_utf8);
+			MB_CONVERT_BUF_ENSURE(buf, out, limit, len);
 		} else if (w < 0x10000) {
 			MB_CONVERT_BUF_ENSURE(buf, out, limit, len + 3);
 			out = mb_convert_buf_add3(out, ((w >> 12) & 0xF) | 0xE0, ((w >> 6) & 0x3F) | 0x80, (w & 0x3F) | 0x80);
diff --git a/ext/mbstring/tests/surrogates_to_utf8.phpt b/ext/mbstring/tests/surrogates_to_utf8.phpt
new file mode 100644
index 00000000000..9707c47f24d
--- /dev/null
+++ b/ext/mbstring/tests/surrogates_to_utf8.phpt
@@ -0,0 +1,47 @@
+--TEST--
+Surrogate codepoints are not encoded as UTF-8
+--EXTENSIONS--
+mbstring
+--FILE--
+<?php
+mb_internal_encoding('UTF-8');
+
+function test(string $desc, string $str) {
+    echo $desc, ': ', bin2hex($str), ' ', var_export(mb_check_encoding($str, 'UTF-8'), true), "\n";
+}
+
+test('UCS-4', mb_convert_encoding("\x00\x00\x00A\x00\x00\xD8\x00\x00\x00\x00B", 'UTF-8', 'UCS-4'));
+test('UCS-4LE', mb_convert_encoding("\xFF\xDF\x00\x00", 'UTF-8', 'UCS-4LE'));
+test('UCS-2', mb_convert_encoding("\xD8\x3D\xDE\x00", 'UTF-8', 'UCS-2'));
+test('UCS-2LE', mb_convert_encoding("\x00\xDC", 'UTF-8', 'UCS-2LE'));
+test('HTML-ENTITIES', mb_convert_encoding('&#xD800;&#57343;', 'UTF-8', 'HTML-ENTITIES'));
+
+$vars = ["\xD8\x00"];
+mb_convert_variables('UTF-8', 'UCS-2', $vars);
+test('mb_convert_variables', $vars[0]);
+
+test('mb_decode_numericentity', mb_decode_numericentity('&#55296;&#xDFFF;', [0, 0x10FFFF, 0, 0x1FFFFF], 'UTF-8'));
+test('mb_decode_mimeheader', mb_decode_mimeheader('=?UCS-2?B?2AA=?='));
+
+mb_substitute_character('long');
+test('long', mb_convert_encoding("\x00\x00\xD8\x00", 'UTF-8', 'UCS-4'));
+
+// Searching still tells surrogates apart
+$haystack = "\xD8\x3D\xDE\x00\x00A\xD8\x3D\xDE\x01";
+var_dump(mb_strpos($haystack, "\xD8\x3D\xDE\x01", 0, 'UCS-2'));
+var_dump(mb_stripos($haystack, "\xD8\x3D\xDE\x01", 0, 'UCS-2'));
+var_dump(mb_substr_count($haystack, "\xDE\x01", 'UCS-2'));
+?>
+--EXPECT--
+UCS-4: 413f42 true
+UCS-4LE: 3f true
+UCS-2: 3f3f true
+UCS-2LE: 3f true
+HTML-ENTITIES: 3f3f true
+mb_convert_variables: 3f true
+mb_decode_numericentity: 3f3f true
+mb_decode_mimeheader: 3f true
+long: 552b44383030 true
+int(3)
+int(3)
+int(1)