Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
40 changes: 38 additions & 2 deletions ext/json/ext/parser/parser.c
Original file line number Diff line number Diff line change
Expand Up @@ -5,7 +5,7 @@
static VALUE mJSON, eNestingError, eParserError, Encoding_UTF_8;
static VALUE CNaN, CInfinity, CMinusInfinity, JSON_empty_string;

static ID i_new, i_try_convert, i_encode, i_at_line, i_at_column, i_at_json_path;
static ID i_new, i_try_convert, i_encode, i_convert, i_finish, i_at_line, i_at_column, i_at_json_path;
#ifndef HAVE_RB_STR_TO_INTERNED_STR
static ID i_uminus;
#endif
Expand Down Expand Up @@ -2254,6 +2254,8 @@ typedef struct JSON_ResumableParserStruct {
rvalue_stack value_stack;
json_frame_stack frames;
VALUE buffer;
VALUE converter;
int source_encindex;
size_t parsed_bytes;
size_t incomplete_bytes;
bool complete;
Expand All @@ -2267,6 +2269,7 @@ static void JSON_ResumableParser_mark(void *ptr)
rvalue_stack_mark(&parser->value_stack);
rvalue_cache_mark(&parser->state.name_cache);
rb_gc_mark(parser->buffer); // pin the buffer
rb_gc_mark_movable(parser->converter);
rb_gc_mark_movable(parser->state.parser);
}

Expand Down Expand Up @@ -2302,6 +2305,7 @@ static void JSON_ResumableParser_compact(void *ptr)
rvalue_stack_compact(&parser->value_stack);
rvalue_cache_compact(&parser->state.name_cache);
parser->buffer = rb_gc_location(parser->buffer);
parser->converter = rb_gc_location(parser->converter);
parser->state.parser = rb_gc_location(parser->state.parser);
}

Expand Down Expand Up @@ -2406,18 +2410,47 @@ static VALUE cResumableParser_initialize(int argc, VALUE *argv, VALUE self)

static JSON_ResumableParser *ResumableParser_acquire(VALUE self, bool lock);

static VALUE resumable_convert_encoding(JSON_ResumableParser *parser, VALUE str)
{
StringValue(str);
if (!RSTRING_LEN(str)) {
return str;
}
int encindex = RB_ENCODING_GET(str);

if (parser->converter && parser->source_encindex != encindex) {
// Reject an incomplete sequence before switching encodings.
rb_funcall(parser->converter, i_finish, 0);
parser->converter = Qfalse;
}

if (encindex == utf8_encindex || encindex == binary_encindex) {
return convert_encoding(str);
}

if (!parser->converter) {
VALUE klass = rb_const_get(rb_cEncoding, rb_intern("Converter"));
VALUE encoding = rb_enc_from_encoding(rb_enc_from_index(encindex));
parser->converter = rb_funcall(klass, i_new, 2, encoding, Encoding_UTF_8);
parser->source_encindex = encindex;
}
return rb_funcall(parser->converter, i_convert, 1, str);
}

/*
* call-seq: self << string -> self
*
* Appends the given string to the parser's buffer.
* Non-UTF-8 input is converted incrementally, retaining incomplete characters
* between calls. Calling #clear also resets the encoding converter.
*/
static VALUE cResumableParser_feed(VALUE self, VALUE str)
{
rb_check_frozen(self);

JSON_ResumableParser *parser = ResumableParser_acquire(self, false);

str = convert_encoding(str);
str = resumable_convert_encoding(parser, str);
if (!RSTRING_LEN(str)) {
return self;
}
Expand Down Expand Up @@ -2636,6 +2669,7 @@ static VALUE cResumableParser_clear(VALUE self)
{
JSON_ResumableParser *parser = ResumableParser_acquire(self, false);
parser->buffer = 0;
parser->converter = Qfalse;
parser->complete = true;
parser->parsed_bytes = 0;
parser->incomplete_bytes = 0;
Expand Down Expand Up @@ -2908,6 +2942,8 @@ void Init_parser(void)
i_uminus = rb_intern("-@");
#endif
i_encode = rb_intern("encode");
i_convert = rb_intern("convert");
i_finish = rb_intern("finish");
i_at_line = rb_intern("@line");
i_at_column = rb_intern("@column");
i_at_json_path = rb_intern("@json_path");
Expand Down
52 changes: 52 additions & 0 deletions test/json/resumable_parser_test.rb
Original file line number Diff line number Diff line change
Expand Up @@ -416,6 +416,58 @@ def test_feed_frozen_multibyte_chunks
assert_equal Encoding::UTF_8, value["message"].encoding
end

def test_feed_non_utf8_chunks
["UTF-16", "UTF-16LE", "UTF-16BE", "UTF-32", "UTF-32LE", "UTF-32BE", "Shift_JIS", "ISO-2022-JP"].each do |encoding|
expected = encoding.start_with?("UTF") ? ["日本🌌"] : ["日本"]
json = JSON.generate(expected).encode(encoding).b
chunk_size = 1
if RUBY_ENGINE == "truffleruby" && encoding.start_with?("UTF")
# TruffleRuby requires UTF-16/32 string sizes to be code-unit aligned.
chunk_size = encoding.start_with?("UTF-32") ? 4 : 2
end
parser = new_parser
(0...json.bytesize).step(chunk_size) do |offset|
chunk = json.byteslice(offset, chunk_size).force_encoding(encoding).freeze
parser << chunk
parser << ''
assert_equal offset + chunk_size == json.bytesize, parser.parse, encoding
end
assert_equal expected, parser.value, encoding
end
end

def test_clear_resets_input_encoding
@parser << "\x00\xD8".dup.force_encoding("UTF-16LE")
refute @parser.parse
@parser.clear
@parser << '["語"]'.encode("UTF-16BE")
assert @parser.parse
assert_equal ["語"], @parser.value
end

def test_feed_different_encodings
@parser << '["日",'.encode("UTF-16LE")
@parser << '"本",'.encode("UTF-16BE")
@parser << '"語"]'
assert @parser.parse
assert_equal ["日", "本", "語"], @parser.value
end

def test_feed_invalid_non_utf8_sequence
@parser << '["'.encode("UTF-16LE")
@parser << "\x00\xD8".dup.force_encoding("UTF-16LE")
assert_raise(Encoding::InvalidByteSequenceError) do
@parser << 'x'.encode("UTF-16LE")
end
end

def test_feed_different_encoding_with_incomplete_sequence
@parser << "\x00\xD8".dup.force_encoding("UTF-16LE")
assert_raise(Encoding::InvalidByteSequenceError) do
@parser << '[]'
end
end

def test_eos
assert_predicate @parser, :eos?

Expand Down
Loading