Merge pull request #3629 from lightpanda-io/nikneym/text-encoder-encode-into

`TextEncoder`: add `encodeInto`
This commit is contained in:
Karl Seguin authored and GitHub committed 2026-09-28 07:07:17 +08:00
commit 895b967259
3 files changed
+368 -35

No files matched your search

+16 -20
View File
@@ -59,13 +59,9 @@ pub fn toSliceWithAlloc(self: String, allocator: Allocator) ![]u8 {
}
fn _toSlice(self: String, comptime null_terminate: bool, allocator: Allocator) !(if (null_terminate) [:0]u8 else []u8) {
const local = self.local;
const handle = self.handle;
const isolate = local.isolate.handle;
const l = v8.v8__String__Utf8Length(handle, isolate);
const buf = try (if (comptime null_terminate) allocator.allocSentinel(u8, @intCast(l), 0) else allocator.alloc(u8, @intCast(l)));
const n = v8.v8__String__WriteUtf8(handle, isolate, buf.ptr, buf.len, v8.WRITE_REPLACE_INVALID_UTF8, null);
const l = self.len();
const buf = try (if (comptime null_terminate) allocator.allocSentinel(u8, l, 0) else allocator.alloc(u8, l));
const n = self.writeUtf8(buf, null);
if (comptime lp.IS_DEBUG) {
std.debug.assert(n == l);
}
@@ -80,14 +76,11 @@ pub fn toSSO(self: String, comptime global: bool) !(if (global) lp.String.Global
return self.toSSOWithAlloc(self.local.call_arena);
}
pub fn toSSOWithAlloc(self: String, allocator: Allocator) !lp.String {
const handle = self.handle;
const isolate = self.local.isolate.handle;
const l: usize = @intCast(v8.v8__String__Utf8Length(handle, isolate));
const l = self.len();
if (l <= 12) {
var content: [12]u8 = undefined;
const n = v8.v8__String__WriteUtf8(handle, isolate, &content[0], content.len, v8.WRITE_REPLACE_INVALID_UTF8, null);
const n = self.writeUtf8(&content, null);
if (comptime lp.IS_DEBUG) {
std.debug.assert(n == l);
}
@@ -103,7 +96,7 @@ pub fn toSSOWithAlloc(self: String, allocator: Allocator) !lp.String {
}
const buf = try allocator.alloc(u8, l);
const n = v8.v8__String__WriteUtf8(handle, isolate, buf.ptr, buf.len, v8.WRITE_REPLACE_INVALID_UTF8, null);
const n = self.writeUtf8(buf, null);
if (comptime lp.IS_DEBUG) {
std.debug.assert(n == l);
}
@@ -116,15 +109,11 @@ pub fn toSSOWithAlloc(self: String, allocator: Allocator) !lp.String {
}
pub fn format(self: String, writer: *std.Io.Writer) !void {
const local = self.local;
const handle = self.handle;
const isolate = local.isolate.handle;
var small: [1024]u8 = undefined;
const l = v8.v8__String__Utf8Length(handle, isolate);
var buf = if (l < 1024) &small else local.call_arena.alloc(u8, @intCast(l)) catch return error.WriteFailed;
const l = self.len();
const buf = if (l < 1024) &small else self.local.call_arena.alloc(u8, l) catch return error.WriteFailed;
const n = v8.v8__String__WriteUtf8(handle, isolate, buf.ptr, buf.len, v8.WRITE_REPLACE_INVALID_UTF8, null);
const n = self.writeUtf8(buf, null);
return writer.writeAll(buf[0..n]);
}
@@ -153,3 +142,10 @@ pub fn toOneByteSlice(self: String, allocator: Allocator) ![]u8 {
}
return buf;
}
// Encodes into `dest`, stopping before any code point that doesn't fit and
// replacing lone surrogates with U+FFFD. Returns the bytes written;
// `processed` receives the UTF-16 code units consumed.
pub fn writeUtf8(self: String, dest: []u8, processed: ?*usize) usize {
return v8.v8__String__WriteUtf8(self.handle, self.local.isolate.handle, dest.ptr, dest.len, v8.WRITE_REPLACE_INVALID_UTF8, processed);
}
@@ -11,3 +11,312 @@
testing.expectEqual([226, 130, 172], Array.from(encoder.encode('€')));
testing.expectEqual([111,118,101,114,32,57,48,48,48], encoder.encode("over 9000"));
</script>
<script id=encode-conversion>
{
const encoder = new TextEncoder();
testing.expectEqual(0, TextEncoder.prototype.encode.length);
testing.expectEqual('Uint8Array', encoder.encode('a').constructor.name);
testing.expectEqual(0, encoder.encode('').length);
// Astral code points are 4 bytes, lone surrogates become U+FFFD.
testing.expectEqual([240, 159, 152, 128], Array.from(encoder.encode('\u{1F600}')));
testing.expectEqual([97, 239, 191, 189, 98], Array.from(encoder.encode('a\uD800b')));
testing.expectEqual([239, 191, 189], Array.from(encoder.encode('\uDC00')));
// Anything else is stringified.
testing.expectEqual([52, 50], Array.from(encoder.encode(42)));
testing.expectEqual([104, 105], Array.from(encoder.encode({ toString() { return 'hi'; } })));
testing.expectError('Error', () => encoder.encode({ toString() { throw new Error('boom'); } }));
}
</script>
<script id=encodeInto>
{
const encoder = new TextEncoder();
// Plain ASCII fits exactly.
let dest = new Uint8Array(9);
let res = encoder.encodeInto('over 9000', dest);
testing.expectEqual(9, res.read);
testing.expectEqual(9, res.written);
testing.expectEqual([111,118,101,114,32,57,48,48,48], Array.from(dest));
// A roomy destination leaves the trailing bytes untouched.
dest = new Uint8Array(6).fill(255);
res = encoder.encodeInto('ab', dest);
testing.expectEqual(2, res.read);
testing.expectEqual(2, res.written);
testing.expectEqual([97, 98, 255, 255, 255, 255], Array.from(dest));
}
</script>
<script id=encodeInto-multibyte>
{
const encoder = new TextEncoder();
// '€' is 3 bytes but a single UTF-16 code unit.
let dest = new Uint8Array(3);
let res = encoder.encodeInto('€', dest);
testing.expectEqual(1, res.read);
testing.expectEqual(3, res.written);
testing.expectEqual([226, 130, 172], Array.from(dest));
// An astral code point is 4 bytes and a surrogate pair, so read is 2.
dest = new Uint8Array(4);
res = encoder.encodeInto('\u{1F600}', dest);
testing.expectEqual(2, res.read);
testing.expectEqual(4, res.written);
testing.expectEqual([240, 159, 152, 128], Array.from(dest));
// A lone surrogate goes through the USVString conversion and comes out
// as U+FFFD: one code unit read, three bytes written.
dest = new Uint8Array(3);
res = encoder.encodeInto('\uD800', dest);
testing.expectEqual(1, res.read);
testing.expectEqual(3, res.written);
testing.expectEqual([239, 191, 189], Array.from(dest));
}
</script>
<script id=encodeInto-truncation>
{
const encoder = new TextEncoder();
// A partial UTF-8 sequence must never be written: '€' needs 3 bytes, so
// with only 2 left nothing of it is emitted.
let dest = new Uint8Array(2).fill(255);
let res = encoder.encodeInto('€', dest);
testing.expectEqual(0, res.read);
testing.expectEqual(0, res.written);
testing.expectEqual([255, 255], Array.from(dest));
// The ASCII prefix is written, then encoding stops at the '€'.
dest = new Uint8Array(3).fill(255);
res = encoder.encodeInto('ab€', dest);
testing.expectEqual(2, res.read);
testing.expectEqual(2, res.written);
testing.expectEqual([97, 98, 255], Array.from(dest));
// Neither half of a surrogate pair is read when its 4 bytes don't fit.
dest = new Uint8Array(4).fill(255);
res = encoder.encodeInto('a\u{1F600}', dest);
testing.expectEqual(1, res.read);
testing.expectEqual(1, res.written);
testing.expectEqual([97, 255, 255, 255], Array.from(dest));
// An empty destination writes nothing.
dest = new Uint8Array(0);
res = encoder.encodeInto('abc', dest);
testing.expectEqual(0, res.read);
testing.expectEqual(0, res.written);
}
</script>
<script id=encodeInto-required-args>
{
const encoder = new TextEncoder();
// Both arguments are required.
testing.expectEqual(2, TextEncoder.prototype.encodeInto.length);
testing.expectError('TypeError', () => encoder.encodeInto());
testing.expectError('TypeError', () => encoder.encodeInto('abc'));
}
</script>
<script id=encodeInto-destination-type>
{
const encoder = new TextEncoder();
// The destination must be a Uint8Array specifically — no other view type,
// and not a bare ArrayBuffer.
testing.expectError('TypeError', () => encoder.encodeInto('', new Int8Array(8)));
testing.expectError('TypeError', () => encoder.encodeInto('', new Uint8ClampedArray(8)));
testing.expectError('TypeError', () => encoder.encodeInto('', new Uint16Array(8)));
testing.expectError('TypeError', () => encoder.encodeInto('', new Int32Array(8)));
testing.expectError('TypeError', () => encoder.encodeInto('', new Float64Array(8)));
testing.expectError('TypeError', () => encoder.encodeInto('', new DataView(new ArrayBuffer(8))));
testing.expectError('TypeError', () => encoder.encodeInto('', new ArrayBuffer(8)));
// A Uint8Array view over a subrange only sees its own window.
const buf = new ArrayBuffer(8);
const view = new Uint8Array(buf, 2, 3);
const res = encoder.encodeInto('abcd', view);
testing.expectEqual(3, res.read);
testing.expectEqual(3, res.written);
testing.expectEqual([0, 0, 97, 98, 99, 0, 0, 0], Array.from(new Uint8Array(buf)));
}
</script>
<script id=encodeInto-source-stringified>
{
const encoder = new TextEncoder();
// The source is a USVString: anything else is stringified, so a typed
// array is encoded as its toString(), not as its bytes.
let dest = new Uint8Array(8);
let res = encoder.encodeInto(new Uint8Array([104, 105]), dest);
testing.expectEqual(7, res.read);
testing.expectEqual(7, res.written);
testing.expectEqual('104,105', String.fromCharCode(...dest.subarray(0, 7)));
dest = new Uint8Array(9);
res = encoder.encodeInto(undefined, dest);
testing.expectEqual(9, res.written);
testing.expectEqual('undefined', String.fromCharCode(...dest));
// Using the same array as source and destination is fine: the source is
// stringified before anything is written.
const both = new Uint8Array([104, 105]);
res = encoder.encodeInto(both, both);
testing.expectEqual(2, res.read);
testing.expectEqual(2, res.written);
testing.expectEqual([49, 48], Array.from(both));
}
</script>
<script id=encodeInto-detached-and-shared>
{
const encoder = new TextEncoder();
// A detached destination has no room, so nothing is read or written.
const buf = new ArrayBuffer(4);
const detached = new Uint8Array(buf);
buf.transfer();
testing.expectEqual(0, detached.byteLength);
let res = encoder.encodeInto('abc', detached);
testing.expectEqual(0, res.read);
testing.expectEqual(0, res.written);
// [AllowShared]: a SharedArrayBuffer-backed view is written like any other.
const shared = new Uint8Array(new SharedArrayBuffer(4));
res = encoder.encodeInto('ab', shared);
testing.expectEqual(2, res.read);
testing.expectEqual(2, res.written);
testing.expectEqual([97, 98, 0, 0], Array.from(shared));
}
</script>
<script id=encode-output>
{
const encoder = new TextEncoder();
// Every call returns its own Uint8Array, sized exactly to the output.
const a = encoder.encode('a€');
const b = encoder.encode('a€');
testing.expectEqual(true, a.buffer !== b.buffer);
testing.expectEqual(0, a.byteOffset);
testing.expectEqual(4, a.byteLength);
testing.expectEqual(4, a.buffer.byteLength);
// Mixed widths round-trip through TextDecoder, lone surrogates as U+FFFD.
const source = 'aé€\u{1F600}\uD800'.repeat(20000);
const encoded = encoder.encode(source);
testing.expectEqual(20000 * (1 + 2 + 3 + 4 + 3), encoded.length);
testing.expectEqual(source.replaceAll('\uD800', '�'), new TextDecoder().decode(encoded));
}
</script>
<script id=encodeInto-vectors>
{
// The vectors from WPT's encoding/encodeInto.any.js, written at an offset
// into a larger buffer pre-filled with 0x80: bytes around the destination
// window must be left alone.
const vectors = [
{ input: 'Hi', read: 0, destinationLength: 0, written: [] },
{ input: 'A', read: 1, destinationLength: 10, written: [0x41] },
{ input: '\u{1D306}', read: 2, destinationLength: 4, written: [0xF0, 0x9D, 0x8C, 0x86] },
{ input: '\u{1D306}A', read: 0, destinationLength: 3, written: [] },
{ input: '\uD834A\uDF06A¥Hi', read: 5, destinationLength: 10, written: [0xEF, 0xBF, 0xBD, 0x41, 0xEF, 0xBF, 0xBD, 0x41, 0xC2, 0xA5] },
{ input: 'A\uDF06', read: 2, destinationLength: 4, written: [0x41, 0xEF, 0xBF, 0xBD] },
{ input: '¥¥', read: 2, destinationLength: 4, written: [0xC2, 0xA5, 0xC2, 0xA5] },
];
const encoder = new TextEncoder();
for (const v of vectors) {
const offset = 4;
const buffer = new ArrayBuffer(v.destinationLength + 10);
new Uint8Array(buffer).fill(0x80);
const view = new Uint8Array(buffer, offset, v.destinationLength);
const res = encoder.encodeInto(v.input, view);
testing.expectEqual(v.read, res.read);
testing.expectEqual(v.written.length, res.written);
const expected = new Array(buffer.byteLength).fill(0x80);
expected.splice(offset, v.written.length, ...v.written);
testing.expectEqual(expected, Array.from(new Uint8Array(buffer)));
}
}
</script>
<script id=encodeInto-argument-order>
{
const encoder = new TextEncoder();
// Arguments are converted in order: the source is stringified before the
// destination is checked.
let called = false;
const source = { toString() { called = true; return 'a'; } };
testing.expectError('TypeError', () => encoder.encodeInto(source, new Int8Array(4)));
testing.expectEqual(true, called);
// An exception from the stringification propagates, nothing is written.
const dest = new Uint8Array(4).fill(255);
testing.expectError('boom', () => encoder.encodeInto({ toString() { throw new Error('boom'); } }, dest));
testing.expectEqual([255, 255, 255, 255], Array.from(dest));
// The destination is looked at after the source is stringified, so one
// detached from inside toString() is seen as empty.
const buffer = new ArrayBuffer(8);
const view = new Uint8Array(buffer);
const res = encoder.encodeInto({ toString() { buffer.transfer(); return 'abc'; } }, view);
testing.expectEqual(0, res.read);
testing.expectEqual(0, res.written);
// null is stringified, not treated as a missing argument.
const out = new Uint8Array(4);
const nres = encoder.encodeInto(null, out);
testing.expectEqual(4, nres.read);
testing.expectEqual(4, nres.written);
testing.expectEqual([110, 117, 108, 108], Array.from(out));
}
</script>
<script id=encodeInto-resizable>
{
const encoder = new TextEncoder();
// A length-tracking view over a resizable buffer uses its current length.
const growable = new ArrayBuffer(2, { maxByteLength: 8 });
const tracking = new Uint8Array(growable);
growable.resize(6);
let res = encoder.encodeInto('abcdefgh', tracking);
testing.expectEqual(6, res.read);
testing.expectEqual(6, res.written);
testing.expectEqual([97, 98, 99, 100, 101, 102], Array.from(tracking));
// A fixed-length view the buffer shrank out from under is out of bounds,
// so it has no room.
const shrinkable = new ArrayBuffer(8, { maxByteLength: 8 });
const fixed = new Uint8Array(shrinkable, 0, 6);
shrinkable.resize(4);
res = encoder.encodeInto('abc', fixed);
testing.expectEqual(0, res.read);
testing.expectEqual(0, res.written);
testing.expectEqual([0, 0, 0, 0], Array.from(new Uint8Array(shrinkable)));
}
</script>
<script id=encodeInto-large>
{
// A long source is cut at the last code point that fits: 50000 'é's take
// 100000 bytes, the 50001st doesn't fit in the one byte left.
const dest = new Uint8Array(100001);
const res = new TextEncoder().encodeInto('é'.repeat(100001), dest);
testing.expectEqual(50000, res.read);
testing.expectEqual(100000, res.written);
testing.expectEqual(0, dest[100000]);
}
</script>
+43 -15
View File
@@ -16,7 +16,6 @@
// You should have received a copy of the GNU Affero General Public License
// along with this program. If not, see <https://www.gnu.org/licenses/>.
const std = @import("std");
const js = @import("../../js/js.zig");
const TextEncoder = @This();
@@ -26,23 +25,51 @@ pub fn init() TextEncoder {
return .{};
}
pub fn encode(_: *const TextEncoder, v_: ?js.Value) !js.TypedArray(u8) {
const v = v_ orelse return .{ .values = "" };
pub fn encode(_: *const TextEncoder, v_: ?js.Value, exec: *const js.Execution) !js.Value {
const local = exec.js.local.?;
if (v.isUndefined()) {
return .{ .values = "" };
// The input is an optional USVString defaulting to "": undefined is the
// default, anything else (null included) is stringified.
const source = blk: {
const v = v_ orelse break :blk local.newString("");
if (v.isUndefined()) {
break :blk local.newString("");
}
break :blk try v.toString();
};
const array = local.createTypedArray(.uint8, source.len());
const slice = array.slice();
_ = source.writeUtf8(slice, null);
return .{ .local = local, .handle = array.handle };
}
// https://encoding.spec.whatwg.org/#dom-textencoder-encodeinto
// `read` counts UTF-16 code units consumed from the source, `written` counts
// bytes written into the destination.
pub const EncodeIntoResult = struct {
read: usize,
written: usize,
};
pub fn encodeInto(_: *const TextEncoder, source_: js.Value, destination_: js.Value) !EncodeIntoResult {
// The source is a USVString, so anything is stringified, as encode does.
// Binding it as a []const u8 would instead hand us the raw bytes of a
// typed array, which could even alias the destination.
const source = try source_.toString();
if (!destination_.isUint8Array()) {
return error.InvalidArgument;
}
const dest = try destination_.toZig([]u8);
if (v.isNull()) {
return .{ .values = "null" };
}
// V8 encodes straight into the destination, never writing a partial
// sequence, and replaces lone surrogates as the USVString conversion would.
var read: usize = 0;
const written = source.writeUtf8(dest, &read);
const str = try v.toStringSlice();
if (!std.unicode.utf8ValidateSlice(str)) {
return error.InvalidUtf8;
}
return .{ .values = str };
return .{ .read = read, .written = written };
}
pub const JsApi = struct {
@@ -56,7 +83,8 @@ pub const JsApi = struct {
};
pub const constructor = bridge.constructor(TextEncoder.init, .{});
pub const encode = bridge.function(TextEncoder.encode, .{ .as_typed_array = true });
pub const encode = bridge.function(TextEncoder.encode, .{});
pub const encodeInto = bridge.function(TextEncoder.encodeInto, .{});
pub const encoding = bridge.property("utf-8", .{ .template = false });
};