diff --git a/src/browser/js/String.zig b/src/browser/js/String.zig index 0018e5b63..081fd29d8 100644 --- a/src/browser/js/String.zig +++ b/src/browser/js/String.zig @@ -59,13 +59,9 @@ pub fn toSliceWithAlloc(self: String, allocator: Allocator) ![]u8 { } fn _toSlice(self: String, comptime null_terminate: bool, allocator: Allocator) !(if (null_terminate) [:0]u8 else []u8) { - const local = self.local; - const handle = self.handle; - const isolate = local.isolate.handle; - - const l = v8.v8__String__Utf8Length(handle, isolate); - const buf = try (if (comptime null_terminate) allocator.allocSentinel(u8, @intCast(l), 0) else allocator.alloc(u8, @intCast(l))); - const n = v8.v8__String__WriteUtf8(handle, isolate, buf.ptr, buf.len, v8.WRITE_REPLACE_INVALID_UTF8, null); + const l = self.len(); + const buf = try (if (comptime null_terminate) allocator.allocSentinel(u8, l, 0) else allocator.alloc(u8, l)); + const n = self.writeUtf8(buf, null); if (comptime lp.IS_DEBUG) { std.debug.assert(n == l); } @@ -80,14 +76,11 @@ pub fn toSSO(self: String, comptime global: bool) !(if (global) lp.String.Global return self.toSSOWithAlloc(self.local.call_arena); } pub fn toSSOWithAlloc(self: String, allocator: Allocator) !lp.String { - const handle = self.handle; - const isolate = self.local.isolate.handle; - - const l: usize = @intCast(v8.v8__String__Utf8Length(handle, isolate)); + const l = self.len(); if (l <= 12) { var content: [12]u8 = undefined; - const n = v8.v8__String__WriteUtf8(handle, isolate, &content[0], content.len, v8.WRITE_REPLACE_INVALID_UTF8, null); + const n = self.writeUtf8(&content, null); if (comptime lp.IS_DEBUG) { std.debug.assert(n == l); } @@ -103,7 +96,7 @@ pub fn toSSOWithAlloc(self: String, allocator: Allocator) !lp.String { } const buf = try allocator.alloc(u8, l); - const n = v8.v8__String__WriteUtf8(handle, isolate, buf.ptr, buf.len, v8.WRITE_REPLACE_INVALID_UTF8, null); + const n = self.writeUtf8(buf, null); if (comptime lp.IS_DEBUG) { std.debug.assert(n == l); } @@ -116,15 +109,11 @@ pub fn toSSOWithAlloc(self: String, allocator: Allocator) !lp.String { } pub fn format(self: String, writer: *std.Io.Writer) !void { - const local = self.local; - const handle = self.handle; - const isolate = local.isolate.handle; - var small: [1024]u8 = undefined; - const l = v8.v8__String__Utf8Length(handle, isolate); - var buf = if (l < 1024) &small else local.call_arena.alloc(u8, @intCast(l)) catch return error.WriteFailed; + const l = self.len(); + const buf = if (l < 1024) &small else self.local.call_arena.alloc(u8, l) catch return error.WriteFailed; - const n = v8.v8__String__WriteUtf8(handle, isolate, buf.ptr, buf.len, v8.WRITE_REPLACE_INVALID_UTF8, null); + const n = self.writeUtf8(buf, null); return writer.writeAll(buf[0..n]); } @@ -153,3 +142,10 @@ pub fn toOneByteSlice(self: String, allocator: Allocator) ![]u8 { } return buf; } + +// Encodes into `dest`, stopping before any code point that doesn't fit and +// replacing lone surrogates with U+FFFD. Returns the bytes written; +// `processed` receives the UTF-16 code units consumed. +pub fn writeUtf8(self: String, dest: []u8, processed: ?*usize) usize { + return v8.v8__String__WriteUtf8(self.handle, self.local.isolate.handle, dest.ptr, dest.len, v8.WRITE_REPLACE_INVALID_UTF8, processed); +} diff --git a/src/browser/tests/encoding/text_encoder.html b/src/browser/tests/encoding/text_encoder.html index 99fd19592..ec7ecc26d 100644 --- a/src/browser/tests/encoding/text_encoder.html +++ b/src/browser/tests/encoding/text_encoder.html @@ -11,3 +11,312 @@ testing.expectEqual([226, 130, 172], Array.from(encoder.encode('€'))); testing.expectEqual([111,118,101,114,32,57,48,48,48], encoder.encode("over 9000")); + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/src/browser/webapi/encoding/TextEncoder.zig b/src/browser/webapi/encoding/TextEncoder.zig index 112d2e321..5a2e112f8 100644 --- a/src/browser/webapi/encoding/TextEncoder.zig +++ b/src/browser/webapi/encoding/TextEncoder.zig @@ -16,7 +16,6 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . -const std = @import("std"); const js = @import("../../js/js.zig"); const TextEncoder = @This(); @@ -26,23 +25,51 @@ pub fn init() TextEncoder { return .{}; } -pub fn encode(_: *const TextEncoder, v_: ?js.Value) !js.TypedArray(u8) { - const v = v_ orelse return .{ .values = "" }; +pub fn encode(_: *const TextEncoder, v_: ?js.Value, exec: *const js.Execution) !js.Value { + const local = exec.js.local.?; - if (v.isUndefined()) { - return .{ .values = "" }; + // The input is an optional USVString defaulting to "": undefined is the + // default, anything else (null included) is stringified. + const source = blk: { + const v = v_ orelse break :blk local.newString(""); + if (v.isUndefined()) { + break :blk local.newString(""); + } + break :blk try v.toString(); + }; + + const array = local.createTypedArray(.uint8, source.len()); + const slice = array.slice(); + _ = source.writeUtf8(slice, null); + + return .{ .local = local, .handle = array.handle }; +} + +// https://encoding.spec.whatwg.org/#dom-textencoder-encodeinto +// `read` counts UTF-16 code units consumed from the source, `written` counts +// bytes written into the destination. +pub const EncodeIntoResult = struct { + read: usize, + written: usize, +}; + +pub fn encodeInto(_: *const TextEncoder, source_: js.Value, destination_: js.Value) !EncodeIntoResult { + // The source is a USVString, so anything is stringified, as encode does. + // Binding it as a []const u8 would instead hand us the raw bytes of a + // typed array, which could even alias the destination. + const source = try source_.toString(); + + if (!destination_.isUint8Array()) { + return error.InvalidArgument; } + const dest = try destination_.toZig([]u8); - if (v.isNull()) { - return .{ .values = "null" }; - } + // V8 encodes straight into the destination, never writing a partial + // sequence, and replaces lone surrogates as the USVString conversion would. + var read: usize = 0; + const written = source.writeUtf8(dest, &read); - const str = try v.toStringSlice(); - if (!std.unicode.utf8ValidateSlice(str)) { - return error.InvalidUtf8; - } - - return .{ .values = str }; + return .{ .read = read, .written = written }; } pub const JsApi = struct { @@ -56,7 +83,8 @@ pub const JsApi = struct { }; pub const constructor = bridge.constructor(TextEncoder.init, .{}); - pub const encode = bridge.function(TextEncoder.encode, .{ .as_typed_array = true }); + pub const encode = bridge.function(TextEncoder.encode, .{}); + pub const encodeInto = bridge.function(TextEncoder.encodeInto, .{}); pub const encoding = bridge.property("utf-8", .{ .template = false }); };