diff --git a/ext/json/ext/parser/parser.c b/ext/json/ext/parser/parser.c index c912a8ee..617e025a 100644 --- a/ext/json/ext/parser/parser.c +++ b/ext/json/ext/parser/parser.c @@ -5,7 +5,7 @@ static VALUE mJSON, eNestingError, eParserError, Encoding_UTF_8; static VALUE CNaN, CInfinity, CMinusInfinity, JSON_empty_string; -static ID i_new, i_try_convert, i_encode, i_at_line, i_at_column, i_at_json_path; +static ID i_new, i_try_convert, i_encode, i_convert, i_finish, i_at_line, i_at_column, i_at_json_path; #ifndef HAVE_RB_STR_TO_INTERNED_STR static ID i_uminus; #endif @@ -2254,6 +2254,8 @@ typedef struct JSON_ResumableParserStruct { rvalue_stack value_stack; json_frame_stack frames; VALUE buffer; + VALUE converter; + int source_encindex; size_t parsed_bytes; size_t incomplete_bytes; bool complete; @@ -2267,6 +2269,7 @@ static void JSON_ResumableParser_mark(void *ptr) rvalue_stack_mark(&parser->value_stack); rvalue_cache_mark(&parser->state.name_cache); rb_gc_mark(parser->buffer); // pin the buffer + rb_gc_mark_movable(parser->converter); rb_gc_mark_movable(parser->state.parser); } @@ -2302,6 +2305,7 @@ static void JSON_ResumableParser_compact(void *ptr) rvalue_stack_compact(&parser->value_stack); rvalue_cache_compact(&parser->state.name_cache); parser->buffer = rb_gc_location(parser->buffer); + parser->converter = rb_gc_location(parser->converter); parser->state.parser = rb_gc_location(parser->state.parser); } @@ -2406,10 +2410,39 @@ static VALUE cResumableParser_initialize(int argc, VALUE *argv, VALUE self) static JSON_ResumableParser *ResumableParser_acquire(VALUE self, bool lock); +static VALUE resumable_convert_encoding(JSON_ResumableParser *parser, VALUE str) +{ + StringValue(str); + if (!RSTRING_LEN(str)) { + return str; + } + int encindex = RB_ENCODING_GET(str); + + if (parser->converter && parser->source_encindex != encindex) { + // Reject an incomplete sequence before switching encodings. + rb_funcall(parser->converter, i_finish, 0); + parser->converter = Qfalse; + } + + if (encindex == utf8_encindex || encindex == binary_encindex) { + return convert_encoding(str); + } + + if (!parser->converter) { + VALUE klass = rb_const_get(rb_cEncoding, rb_intern("Converter")); + VALUE encoding = rb_enc_from_encoding(rb_enc_from_index(encindex)); + parser->converter = rb_funcall(klass, i_new, 2, encoding, Encoding_UTF_8); + parser->source_encindex = encindex; + } + return rb_funcall(parser->converter, i_convert, 1, str); +} + /* * call-seq: self << string -> self * * Appends the given string to the parser's buffer. + * Non-UTF-8 input is converted incrementally, retaining incomplete characters + * between calls. Calling #clear also resets the encoding converter. */ static VALUE cResumableParser_feed(VALUE self, VALUE str) { @@ -2417,7 +2450,7 @@ static VALUE cResumableParser_feed(VALUE self, VALUE str) JSON_ResumableParser *parser = ResumableParser_acquire(self, false); - str = convert_encoding(str); + str = resumable_convert_encoding(parser, str); if (!RSTRING_LEN(str)) { return self; } @@ -2636,6 +2669,7 @@ static VALUE cResumableParser_clear(VALUE self) { JSON_ResumableParser *parser = ResumableParser_acquire(self, false); parser->buffer = 0; + parser->converter = Qfalse; parser->complete = true; parser->parsed_bytes = 0; parser->incomplete_bytes = 0; @@ -2908,6 +2942,8 @@ void Init_parser(void) i_uminus = rb_intern("-@"); #endif i_encode = rb_intern("encode"); + i_convert = rb_intern("convert"); + i_finish = rb_intern("finish"); i_at_line = rb_intern("@line"); i_at_column = rb_intern("@column"); i_at_json_path = rb_intern("@json_path"); diff --git a/test/json/resumable_parser_test.rb b/test/json/resumable_parser_test.rb index a704e9ed..8b1cb9d9 100644 --- a/test/json/resumable_parser_test.rb +++ b/test/json/resumable_parser_test.rb @@ -416,6 +416,58 @@ def test_feed_frozen_multibyte_chunks assert_equal Encoding::UTF_8, value["message"].encoding end + def test_feed_non_utf8_chunks + ["UTF-16", "UTF-16LE", "UTF-16BE", "UTF-32", "UTF-32LE", "UTF-32BE", "Shift_JIS", "ISO-2022-JP"].each do |encoding| + expected = encoding.start_with?("UTF") ? ["日本🌌"] : ["日本"] + json = JSON.generate(expected).encode(encoding).b + chunk_size = 1 + if RUBY_ENGINE == "truffleruby" && encoding.start_with?("UTF") + # TruffleRuby requires UTF-16/32 string sizes to be code-unit aligned. + chunk_size = encoding.start_with?("UTF-32") ? 4 : 2 + end + parser = new_parser + (0...json.bytesize).step(chunk_size) do |offset| + chunk = json.byteslice(offset, chunk_size).force_encoding(encoding).freeze + parser << chunk + parser << '' + assert_equal offset + chunk_size == json.bytesize, parser.parse, encoding + end + assert_equal expected, parser.value, encoding + end + end + + def test_clear_resets_input_encoding + @parser << "\x00\xD8".dup.force_encoding("UTF-16LE") + refute @parser.parse + @parser.clear + @parser << '["語"]'.encode("UTF-16BE") + assert @parser.parse + assert_equal ["語"], @parser.value + end + + def test_feed_different_encodings + @parser << '["日",'.encode("UTF-16LE") + @parser << '"本",'.encode("UTF-16BE") + @parser << '"語"]' + assert @parser.parse + assert_equal ["日", "本", "語"], @parser.value + end + + def test_feed_invalid_non_utf8_sequence + @parser << '["'.encode("UTF-16LE") + @parser << "\x00\xD8".dup.force_encoding("UTF-16LE") + assert_raise(Encoding::InvalidByteSequenceError) do + @parser << 'x'.encode("UTF-16LE") + end + end + + def test_feed_different_encoding_with_incomplete_sequence + @parser << "\x00\xD8".dup.force_encoding("UTF-16LE") + assert_raise(Encoding::InvalidByteSequenceError) do + @parser << '[]' + end + end + def test_eos assert_predicate @parser, :eos?