diff --git a/core/encoding/json/parser.odin b/core/encoding/json/parser.odin index f6ca0b3ca..47e0117e8 100644 --- a/core/encoding/json/parser.odin +++ b/core/encoding/json/parser.odin @@ -409,7 +409,23 @@ unquote_string :: proc(token: Token, spec: Specification, allocator := context.a return clone_string(s, allocator, loc) } - b := bytes_make(len(s) + 2*utf8.UTF_MAX, 1, allocator) or_return + // A byte that is not valid UTF-8 is replaced by utf8.RUNE_ERROR, which encodes + // wider than the single byte it stands in for, so a string holding several of + // them unquotes to more bytes than it was quoted in. Count them up front so the + // buffer fits. Nothing else in the loop below grows its input: a two-byte + // escape like \n writes one byte, \xXX writes at most two for four, \uXXXX at + // most three for six, and a surrogate pair four for twelve. + extra := 0 + replacement_size := utf8.rune_size(utf8.RUNE_ERROR) + for j := i; j < len(s); { + r, size := utf8.decode_rune_in_string(s[j:]) + if r == utf8.RUNE_ERROR && size == 1 { + extra += replacement_size - 1 + } + j += size + } + + b := bytes_make(len(s) + 2*utf8.UTF_MAX + extra, 1, allocator) or_return w := copy(b, s[0:i]) if len(b) == 0 && allocator.data == nil { diff --git a/tests/core/encoding/json/test_core_json.odin b/tests/core/encoding/json/test_core_json.odin index 1ef84e07b..a789f3339 100644 --- a/tests/core/encoding/json/test_core_json.odin +++ b/tests/core/encoding/json/test_core_json.odin @@ -2,6 +2,7 @@ package test_core_json import "core:encoding/json" import "core:testing" +import "core:unicode/utf8" import "core:mem/virtual" import "base:runtime" @@ -428,6 +429,24 @@ utf8_string_of_multibyte_characters :: proc(t: ^testing.T) { testing.expectf(t, err == nil, "Expected `json.parse` to return nil, got %v", err) } +@test +invalid_utf8_in_string_is_replaced :: proc(t: ^testing.T) { + // Every one of these bytes is invalid on its own, so each is replaced by + // U+FFFD, which is three bytes: the unquoted string is longer than the quoted + // one, and the buffer has to have been sized for that. + val, err := json.parse_string("\"\xff\xfe\xff\xfe\xff\xfe\xff\xfe\"") + defer json.destroy_value(val) + testing.expectf(t, err == nil, "Expected `json.parse_string` to return nil, got %v", err) + + str, ok := val.(json.String) + testing.expect(t, ok, "Expected a string value") + testing.expectf(t, len(str) == 8 * utf8.rune_size(utf8.RUNE_ERROR), "Expected eight replacement characters, got %d bytes", len(str)) + testing.expect(t, utf8.valid_string(string(str)), "Expected the unquoted string to be valid UTF-8") + for r in string(str) { + testing.expectf(t, r == utf8.RUNE_ERROR, "Expected every rune to be U+FFFD, got %U", r) + } +} + @test struct_with_ignore_tags :: proc(t: ^testing.T) { My_Struct :: struct {