Commit 8f43e8884b2 for nodejs
commit 8f43e8884b211361c7e6d88d642afb526f9062ab
Author: James M Snell <jasnell@gmail.com>
Date: Fri Sep 25 18:58:46 2026 +0000
buffer: add isLatin1
Implements a fast check to determine if a string is
a valid byte string (only chars <= 0xff).
Signed-off-by: James M Snell <jasnell@gmail.com>
PR-URL: https://github.com/nodejs/node/pull/66298
Reviewed-By: René <contact.9a5d6388@renegade334.me.uk>
Reviewed-By: Filip Skokan <panva.ip@gmail.com>
Reviewed-By: Anna Henningsen <anna@addaleax.net>
Reviewed-By: Robert Nagy <ronagy@icloud.com>
diff --git a/benchmark/buffers/buffer-islatin1.js b/benchmark/buffers/buffer-islatin1.js
new file mode 100644
index 00000000000..6396873a726
--- /dev/null
+++ b/benchmark/buffers/buffer-islatin1.js
@@ -0,0 +1,57 @@
+'use strict';
+
+const common = require('../common.js');
+const buffer = require('node:buffer');
+const assert = require('node:assert');
+
+const bench = common.createBenchmark(main, {
+ n: [1e7],
+ length: ['short', 'long'],
+ // onebyte: one-byte representation (O(1) check)
+ // twobyte: two-byte representation containing only code units <= 0xFF
+ // invalid: ends with a code unit > 0xFF
+ input: ['onebyte', 'twobyte', 'invalid'],
+ method: ['isLatin1', 'loop', 'regex'],
+});
+
+function loop(str) {
+ for (let i = 0; i < str.length; i++) {
+ if (str.charCodeAt(i) > 255) return false;
+ }
+ return true;
+}
+
+const notLatin1Re = /[\u0100-\uffff]/;
+function regex(str) {
+ return !notLatin1Re.test(str);
+}
+
+const methods = { isLatin1: buffer.isLatin1, loop, regex };
+
+function main({ n, length, input, method }) {
+ const base = length === 'short' ? 'hello w\u00f6rld' : 'hello w\u00f6rld'.repeat(200);
+ let str;
+ switch (input) {
+ case 'onebyte':
+ str = base;
+ break;
+ case 'twobyte':
+ // Slicing a two-byte string keeps the two-byte representation.
+ str = ('\u0100' + base).slice(1);
+ break;
+ case 'invalid':
+ str = base + '\u0100';
+ break;
+ }
+ const expected = input !== 'invalid';
+ const fn = methods[method];
+ assert.strictEqual(fn(str), expected);
+
+ bench.start();
+ let result;
+ for (let i = 0; i < n; ++i) {
+ result = fn(str);
+ }
+ bench.end(n);
+ assert.strictEqual(result, expected);
+}
diff --git a/doc/api/buffer.md b/doc/api/buffer.md
index 7a574e3c4dd..d1eba8b6f43 100644
--- a/doc/api/buffer.md
+++ b/doc/api/buffer.md
@@ -5414,6 +5414,53 @@ including the case in which `input` is empty.
A detached `ArrayBuffer`, or a `TypedArray` backed by one, is treated as empty.
+### `buffer.isLatin1(input)`
+
+<!-- YAML
+added: REPLACEME
+-->
+
+* `input` {string} The string to validate.
+* Returns: {boolean}
+
+This function returns `true` if `input` can be losslessly encoded using the
+Node.js [`'latin1'`][character encodings] encoding, including the case in
+which `input` is empty. That is, it returns `true` if every UTF-16 code unit
+of `input` is in the range `U+0000` to `U+00FF`. Such a string is also a valid
+[WebIDL `ByteString`][].
+
+This check uses the Node.js definition of `'latin1'`, in which each code unit
+from `U+0000` to `U+00FF` maps directly to the byte of the same value. It does
+not use the [WHATWG Encoding Standard][] definition, in which the `'latin1'`
+label is an alias for `windows-1252`. For example, `'\u0080'` is considered
+latin1 by this function, while `'€'` (`U+20AC`, which `windows-1252` encodes as
+`0x80`) is not.
+
+Unlike [`buffer.isAscii()`][] and [`buffer.isUtf8()`][], this function
+validates a string rather than a `Buffer`, `TypedArray`, or `ArrayBuffer`.
+Every byte sequence would trivially be valid `'latin1'`, since every byte maps
+to a code unit less than or equal to `0xFF`.
+
+```mjs
+import { isLatin1 } from 'node:buffer';
+
+isLatin1('hello'); // true
+isLatin1('café'); // true
+isLatin1('\u00ff'); // true
+isLatin1('\u0100'); // false
+isLatin1('€'); // false
+```
+
+```cjs
+const { isLatin1 } = require('node:buffer');
+
+isLatin1('hello'); // true
+isLatin1('café'); // true
+isLatin1('\u00ff'); // true
+isLatin1('\u0100'); // false
+isLatin1('€'); // false
+```
+
### `buffer.isUtf8(input)`
<!-- YAML
@@ -5775,6 +5822,7 @@ or after startup, if the alignment has to hold at run time.
[UTF-16]: https://en.wikipedia.org/wiki/UTF-16
[UTF-8]: https://en.wikipedia.org/wiki/UTF-8
[WHATWG Encoding Standard]: https://encoding.spec.whatwg.org/
+[WebIDL `ByteString`]: https://webidl.spec.whatwg.org/#idl-ByteString
[`--build-snapshot`]: cli.md#--build-snapshot
[`Buffer.alloc()`]: #static-method-bufferallocsize-fill-encoding
[`Buffer.allocUnsafe()`]: #static-method-bufferallocunsafesize-alignment
@@ -5814,10 +5862,13 @@ or after startup, if the alignment has to hold at run time.
[`buf.values()`]: #bufvalues
[`buffer.constants.MAX_LENGTH`]: #bufferconstantsmax_length
[`buffer.constants.MAX_STRING_LENGTH`]: #bufferconstantsmax_string_length
+[`buffer.isAscii()`]: #bufferisasciiinput
+[`buffer.isUtf8()`]: #bufferisutf8input
[`buffer.kMaxLength`]: #bufferkmaxlength
[`util.inspect()`]: util.md#utilinspectobject-options
[`v8.startupSnapshot.setDeserializeMainFunction()`]: v8.md#v8startupsnapshotsetdeserializemainfunctioncallback-data
[`v8::Uint8Array::kMaxLength`]: https://v8.github.io/api/head/classv8_1_1Uint8Array.html#a7677e3d0c9c92e4d40bef7212f5980c6
[base64url]: https://tools.ietf.org/html/rfc4648#section-5
+[character encodings]: #buffers-and-character-encodings
[endianness]: https://en.wikipedia.org/wiki/Endianness
[iterator]: https://developer.mozilla.org/en-US/docs/Web/JavaScript/Reference/Iteration_protocols
diff --git a/lib/buffer.js b/lib/buffer.js
index 8822a9dbca1..d9c64f27cf3 100644
--- a/lib/buffer.js
+++ b/lib/buffer.js
@@ -62,6 +62,7 @@ const {
copy: _copy,
fill: bindingFill,
isAscii: bindingIsAscii,
+ isLatin1: bindingIsLatin1,
isUtf8: bindingIsUtf8,
stringLengthUtf8: bindingStringLengthUtf8,
indexOfBuffer,
@@ -1498,6 +1499,17 @@ function isAscii(input) {
throw new ERR_INVALID_ARG_TYPE('input', ['ArrayBuffer', 'Buffer', 'TypedArray'], input);
}
+/**
+ * Returns true if every UTF-16 code unit of `input` is <= 0xFF, i.e. `input`
+ * can be losslessly encoded as 'latin1'.
+ * @param {string} input
+ * @returns {boolean}
+ */
+function isLatin1(input) {
+ validateString(input, 'input');
+ return bindingIsLatin1(input);
+}
+
function stringLength(input, encoding = 'utf8') {
if (!isTypedArray(input) && !isAnyArrayBuffer(input)) {
throw new ERR_INVALID_ARG_TYPE('input', ['ArrayBuffer', 'Buffer', 'TypedArray'], input);
@@ -1529,6 +1541,7 @@ module.exports = {
transcode,
isUtf8,
isAscii,
+ isLatin1,
// Legacy
kMaxLength,
diff --git a/src/node_buffer.cc b/src/node_buffer.cc
index 7e966e5f393..7ade56fd892 100644
--- a/src/node_buffer.cc
+++ b/src/node_buffer.cc
@@ -1414,6 +1414,28 @@ static bool FastIsAscii(Local<Value> receiver,
static CFunction fast_is_ascii(CFunction::Make(FastIsAscii));
+// Returns true if every UTF-16 code unit of the string is <= 0xFF, i.e. the
+// string can be losslessly encoded using Node.js' 'latin1' encoding (which
+// maps U+0000-U+00FF directly to bytes 0x00-0xFF, unlike the WHATWG
+// 'latin1' label, which is an alias for windows-1252).
+// ContainsOnlyOneByte() is O(1) for strings with a one-byte representation,
+// uses SIMD for flat two-byte strings, and traverses cons strings without
+// flattening (no allocation), which makes it safe to call from a fast API
+// call.
+static void IsLatin1(const FunctionCallbackInfo<Value>& args) {
+ CHECK_EQ(args.Length(), 1);
+ CHECK(args[0]->IsString());
+ args.GetReturnValue().Set(args[0].As<String>()->ContainsOnlyOneByte());
+}
+
+static bool FastIsLatin1(Local<Value> receiver, Local<Value> value) {
+ TRACK_V8_FAST_API_CALL("buffer.isLatin1");
+ CHECK(value->IsString());
+ return value.As<String>()->ContainsOnlyOneByte();
+}
+
+static CFunction fast_is_latin1(CFunction::Make(FastIsLatin1));
+
// Number of UTF-16 code units produced by decoding [p, end) as UTF-8 with
// WHATWG "maximal subpart" U+FFFD replacement, matching the fallback that
// StringBytes::Encode takes for invalid input (v8::String::NewFromUtf8).
@@ -1928,6 +1950,8 @@ void Initialize(Local<Object> target,
SetFastMethodNoSideEffect(context, target, "isUtf8", IsUtf8, &fast_is_utf8);
SetFastMethodNoSideEffect(
context, target, "isAscii", IsAscii, &fast_is_ascii);
+ SetFastMethodNoSideEffect(
+ context, target, "isLatin1", IsLatin1, &fast_is_latin1);
SetFastMethodNoSideEffect(context,
target,
"stringLengthUtf8",
@@ -2008,6 +2032,8 @@ void RegisterExternalReferences(ExternalReferenceRegistry* registry) {
registry->Register(fast_is_utf8);
registry->Register(IsAscii);
registry->Register(fast_is_ascii);
+ registry->Register(IsLatin1);
+ registry->Register(fast_is_latin1);
registry->Register(StringLengthUtf8);
registry->Register(fast_string_length_utf8);
diff --git a/test/parallel/test-buffer-islatin1.js b/test/parallel/test-buffer-islatin1.js
new file mode 100644
index 00000000000..332fa6c0508
--- /dev/null
+++ b/test/parallel/test-buffer-islatin1.js
@@ -0,0 +1,85 @@
+'use strict';
+
+require('../common');
+const assert = require('assert');
+const { isLatin1 } = require('buffer');
+
+function reference(str) {
+ for (let i = 0; i < str.length; i++) {
+ if (str.charCodeAt(i) > 0xFF) return false;
+ }
+ return true;
+}
+
+// Basic cases.
+assert.strictEqual(isLatin1(''), true);
+assert.strictEqual(isLatin1('hello'), true);
+assert.strictEqual(isLatin1('\x00'), true);
+assert.strictEqual(isLatin1('\x7f\x80'), true);
+assert.strictEqual(isLatin1('caf\u00e9'), true);
+assert.strictEqual(isLatin1('\u00ff'), true);
+assert.strictEqual(isLatin1('\u0100'), false);
+assert.strictEqual(isLatin1('\u20ac'), false);
+assert.strictEqual(isLatin1('\uffff'), false);
+// Surrogate pairs and lone surrogates are > 0xFF.
+assert.strictEqual(isLatin1('\ud83d\ude00'), false);
+assert.strictEqual(isLatin1('\ud800'), false);
+assert.strictEqual(isLatin1('\udfff'), false);
+
+// Position of the offending code unit must not matter, and long strings must
+// exercise the vectorized paths.
+for (const len of [1, 7, 8, 15, 16, 31, 32, 33, 63, 64, 65, 1000, 4099]) {
+ const base = 'a\u00ff'.repeat(len).slice(0, len);
+ assert.strictEqual(isLatin1(base), true);
+ for (const pos of [0, len >> 1, len - 1]) {
+ for (const ch of ['\u0100', '\u1234', '\ud800', '\uffff']) {
+ const str = base.slice(0, pos) + ch + base.slice(pos + 1);
+ assert.strictEqual(isLatin1(str), false, `len=${len} pos=${pos}`);
+ }
+ }
+}
+
+// Strings stored with a two-byte representation that only contain code units
+// <= 0xFF must still be reported as latin1.
+{
+ const twoByte = '\u0100' + 'abc\u00e9\u00ff'.repeat(100);
+ const sliced = twoByte.slice(1);
+ assert.strictEqual(isLatin1(twoByte), false);
+ assert.strictEqual(isLatin1(sliced), true);
+ assert.strictEqual(isLatin1(twoByte.substring(1, 20)), true);
+}
+
+// Cons strings (results of concatenation) with mixed representations.
+{
+ let cons = '';
+ for (let i = 0; i < 100; i++) cons += `x${i}\u00e9`;
+ assert.strictEqual(isLatin1(cons), true);
+ assert.strictEqual(isLatin1(cons + '\u0100'), false);
+ assert.strictEqual(isLatin1('\u0100' + cons), false);
+ assert.strictEqual(isLatin1(cons + '\u0100'.slice(1) + cons), true);
+}
+
+// Randomized comparison with the reference implementation.
+for (let i = 0; i < 1000; i++) {
+ const len = Math.floor(Math.random() * 100);
+ const max = Math.random() < 0.5 ? 0x100 : 0x10000;
+ let str = '';
+ for (let j = 0; j < len; j++) {
+ // Keep the probability of producing a code unit > 0xFF low so that
+ // both outcomes are exercised.
+ const code = Math.random() < 0.98 ?
+ Math.floor(Math.random() * 0x100) :
+ Math.floor(Math.random() * max);
+ str += String.fromCharCode(code);
+ }
+ assert.strictEqual(isLatin1(str), reference(str), JSON.stringify(str));
+}
+
+// Invalid argument types.
+[
+ undefined, null, 1, 1n, true, {}, [], Symbol('a'),
+ Buffer.from('a'), new Uint8Array(1), new ArrayBuffer(1),
+ new String('a'),
+].forEach((input) => {
+ assert.throws(() => isLatin1(input), { code: 'ERR_INVALID_ARG_TYPE' });
+});
diff --git a/test/parallel/test-buffer-isutf8-isascii-fast.js b/test/parallel/test-buffer-isutf8-isascii-fast.js
index fb2b57025c5..ad5251095b6 100644
--- a/test/parallel/test-buffer-isutf8-isascii-fast.js
+++ b/test/parallel/test-buffer-isutf8-isascii-fast.js
@@ -3,10 +3,11 @@
const common = require('../common');
const assert = require('assert');
-const { Buffer, isAscii, isUtf8 } = require('buffer');
+const { Buffer, isAscii, isLatin1, isUtf8 } = require('buffer');
const ascii = Buffer.from('hello');
const utf8 = Buffer.from('hello \xc4\x9f');
+const latin1 = 'hello \u00e9';
function testFastIsAscii() {
assert.strictEqual(isAscii(ascii), true);
@@ -16,6 +17,10 @@ function testFastIsUtf8() {
assert.strictEqual(isUtf8(utf8), true);
}
+function testFastIsLatin1() {
+ assert.strictEqual(isLatin1(latin1), true);
+}
+
eval('%PrepareFunctionForOptimization(isAscii)');
testFastIsAscii();
eval('%OptimizeFunctionOnNextCall(isAscii)');
@@ -26,9 +31,15 @@ testFastIsUtf8();
eval('%OptimizeFunctionOnNextCall(isUtf8)');
testFastIsUtf8();
+eval('%PrepareFunctionForOptimization(isLatin1)');
+testFastIsLatin1();
+eval('%OptimizeFunctionOnNextCall(isLatin1)');
+testFastIsLatin1();
+
if (common.isDebug) {
const { internalBinding } = require('internal/test/binding');
const { getV8FastApiCallCount } = internalBinding('debug');
assert.strictEqual(getV8FastApiCallCount('buffer.isAscii'), 1);
assert.strictEqual(getV8FastApiCallCount('buffer.isUtf8'), 1);
+ assert.strictEqual(getV8FastApiCallCount('buffer.isLatin1'), 1);
}
diff --git a/typings/internalBinding/buffer.d.ts b/typings/internalBinding/buffer.d.ts
index a384fb01612..063b1ddd57a 100644
--- a/typings/internalBinding/buffer.d.ts
+++ b/typings/internalBinding/buffer.d.ts
@@ -21,6 +21,7 @@ export interface BufferBinding {
isUtf8(input: ArrayBufferView | ArrayBuffer | SharedArrayBuffer): boolean;
isAscii(input: ArrayBufferView | ArrayBuffer | SharedArrayBuffer): boolean;
+ isLatin1(input: string): boolean;
kMaxLength: number;
kStringMaxLength: number;