core/encoding/json: size the unquote buffer for replaced invalid UTF-8

unquote_string replaces each byte that is not valid UTF-8 with U+FFFD,
which is three bytes for one, but sized its buffer as len(s) + 2*UTF_MAX:
slack for a single replacement, not for one per invalid byte. A string
holding several ran the write cursor past the end, an out-of-range slice
under bounds checking and a memory-safety bug without it.

Count the invalid bytes in the remainder up front and size for them. The
escape sequences never grow their input, so they need no allowance.
This commit is contained in:
Jack Mordaunt committed 2026-09-25 14:27:40 -03:00
1 parent 315725197f
commit 641844dae7
2 files changed
+36 -1

No files matched your search

+17 -1
View File
@@ -409,7 +409,23 @@ unquote_string :: proc(token: Token, spec: Specification, allocator := context.a
return clone_string(s, allocator, loc)
}
b := bytes_make(len(s) + 2*utf8.UTF_MAX, 1, allocator) or_return
// A byte that is not valid UTF-8 is replaced by utf8.RUNE_ERROR, which encodes
// wider than the single byte it stands in for, so a string holding several of
// them unquotes to more bytes than it was quoted in. Count them up front so the
// buffer fits. Nothing else in the loop below grows its input: a two-byte
// escape like \n writes one byte, \xXX writes at most two for four, \uXXXX at
// most three for six, and a surrogate pair four for twelve.
extra := 0
replacement_size := utf8.rune_size(utf8.RUNE_ERROR)
for j := i; j < len(s); {
r, size := utf8.decode_rune_in_string(s[j:])
if r == utf8.RUNE_ERROR && size == 1 {
extra += replacement_size - 1
}
j += size
}
b := bytes_make(len(s) + 2*utf8.UTF_MAX + extra, 1, allocator) or_return
w := copy(b, s[0:i])
if len(b) == 0 && allocator.data == nil {
@@ -2,6 +2,7 @@ package test_core_json
import "core:encoding/json"
import "core:testing"
import "core:unicode/utf8"
import "core:mem/virtual"
import "base:runtime"
@@ -428,6 +429,24 @@ utf8_string_of_multibyte_characters :: proc(t: ^testing.T) {
testing.expectf(t, err == nil, "Expected `json.parse` to return nil, got %v", err)
}
@test
invalid_utf8_in_string_is_replaced :: proc(t: ^testing.T) {
// Every one of these bytes is invalid on its own, so each is replaced by
// U+FFFD, which is three bytes: the unquoted string is longer than the quoted
// one, and the buffer has to have been sized for that.
val, err := json.parse_string("\"\xff\xfe\xff\xfe\xff\xfe\xff\xfe\"")
defer json.destroy_value(val)
testing.expectf(t, err == nil, "Expected `json.parse_string` to return nil, got %v", err)
str, ok := val.(json.String)
testing.expect(t, ok, "Expected a string value")
testing.expectf(t, len(str) == 8 * utf8.rune_size(utf8.RUNE_ERROR), "Expected eight replacement characters, got %d bytes", len(str))
testing.expect(t, utf8.valid_string(string(str)), "Expected the unquoted string to be valid UTF-8")
for r in string(str) {
testing.expectf(t, r == utf8.RUNE_ERROR, "Expected every rune to be U+FFFD, got %U", r)
}
}
@test
struct_with_ignore_tags :: proc(t: ^testing.T) {
My_Struct :: struct {