json 2.12.2 → 3.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,48 +1,25 @@
1
- #include "ruby.h"
2
- #include "ruby/encoding.h"
1
+ #include "../json.h"
2
+ #include "../vendor/fast_float_parser.h"
3
+ #include "../simd/simd.h"
3
4
 
4
- /* shims */
5
- /* This is the fallback definition from Ruby 3.4 */
5
+ static VALUE mJSON, eNestingError, eParserError, Encoding_UTF_8;
6
+ static VALUE CNaN, CInfinity, CMinusInfinity, JSON_empty_string;
6
7
 
7
- #ifndef RBIMPL_STDBOOL_H
8
- #if defined(__cplusplus)
9
- # if defined(HAVE_STDBOOL_H) && (__cplusplus >= 201103L)
10
- # include <cstdbool>
11
- # endif
12
- #elif defined(HAVE_STDBOOL_H)
13
- # include <stdbool.h>
14
- #elif !defined(HAVE__BOOL)
15
- typedef unsigned char _Bool;
16
- # define bool _Bool
17
- # define true ((_Bool)+1)
18
- # define false ((_Bool)+0)
19
- # define __bool_true_false_are_defined
20
- #endif
21
- #endif
22
-
23
- #ifndef RB_UNLIKELY
24
- #define RB_UNLIKELY(expr) expr
8
+ static ID i_new, i_try_convert, i_encode, i_at_line, i_at_column, i_at_json_path;
9
+ #ifndef HAVE_RB_STR_TO_INTERNED_STR
10
+ static ID i_uminus;
25
11
  #endif
26
12
 
27
- #ifndef RB_LIKELY
28
- #define RB_LIKELY(expr) expr
29
- #endif
30
-
31
- static VALUE mJSON, eNestingError, Encoding_UTF_8;
32
- static VALUE CNaN, CInfinity, CMinusInfinity;
33
-
34
- static ID i_chr, i_aset, i_aref,
35
- i_leftshift, i_new, i_try_convert, i_uminus, i_encode;
36
-
37
- static VALUE sym_max_nesting, sym_allow_nan, sym_allow_trailing_comma, sym_symbolize_names, sym_freeze,
38
- sym_decimal_class, sym_on_load;
13
+ static VALUE sym_max_nesting, sym_allow_nan, sym_allow_trailing_comma, sym_allow_comments,
14
+ sym_allow_control_characters, sym_allow_invalid_escape, sym_symbolize_names,
15
+ sym_freeze, sym_decimal_class, sym_on_load, sym_allow_duplicate_key;
39
16
 
40
17
  static int binary_encindex;
41
18
  static int utf8_encindex;
42
19
 
43
20
  #ifndef HAVE_RB_HASH_BULK_INSERT
44
21
  // For TruffleRuby
45
- void
22
+ static void
46
23
  rb_hash_bulk_insert(long count, const VALUE *pairs, VALUE hash)
47
24
  {
48
25
  long index = 0;
@@ -59,6 +36,12 @@ rb_hash_bulk_insert(long count, const VALUE *pairs, VALUE hash)
59
36
  #define rb_hash_new_capa(n) rb_hash_new()
60
37
  #endif
61
38
 
39
+ #ifndef HAVE_RB_STR_TO_INTERNED_STR
40
+ static VALUE rb_str_to_interned_str(VALUE str)
41
+ {
42
+ return rb_funcall(rb_str_freeze(str), i_uminus, 0);
43
+ }
44
+ #endif
62
45
 
63
46
  /* name cache */
64
47
 
@@ -78,6 +61,20 @@ typedef struct rvalue_cache_struct {
78
61
  VALUE entries[JSON_RVALUE_CACHE_CAPA];
79
62
  } rvalue_cache;
80
63
 
64
+ static void rvalue_cache_mark(rvalue_cache *cache)
65
+ {
66
+ for (int index = 0; index < cache->length; index++) {
67
+ rb_gc_mark_movable(cache->entries[index]);
68
+ }
69
+ }
70
+
71
+ static void rvalue_cache_compact(rvalue_cache *cache)
72
+ {
73
+ for (int index = 0; index < cache->length; index++) {
74
+ cache->entries[index] = rb_gc_location(cache->entries[index]);
75
+ }
76
+ }
77
+
81
78
  static rb_encoding *enc_utf8;
82
79
 
83
80
  #define JSON_RVALUE_CACHE_MAX_ENTRY_LENGTH 55
@@ -104,116 +101,104 @@ static void rvalue_cache_insert_at(rvalue_cache *cache, int index, VALUE rstring
104
101
  cache->entries[index] = rstring;
105
102
  }
106
103
 
107
- static inline int rstring_cache_cmp(const char *str, const long length, VALUE rstring)
104
+ #define rstring_cache_memcmp memcmp
105
+
106
+ #if JSON_CPU_LITTLE_ENDIAN_64BITS
107
+ #if __has_builtin(__builtin_bswap64)
108
+ #undef rstring_cache_memcmp
109
+ ALWAYS_INLINE(static) int rstring_cache_memcmp(const char *str, const char *rptr, const long length)
108
110
  {
109
- long rstring_length = RSTRING_LEN(rstring);
110
- if (length == rstring_length) {
111
- return memcmp(str, RSTRING_PTR(rstring), length);
112
- } else {
113
- return (int)(length - rstring_length);
111
+ // The libc memcmp has numerous complex optimizations, but in this particular case,
112
+ // we know the string is small (JSON_RVALUE_CACHE_MAX_ENTRY_LENGTH), so being able to
113
+ // inline a simpler memcmp outperforms calling the libc version.
114
+ long i = 0;
115
+
116
+ for (; i + 8 <= length; i += 8) {
117
+ uint64_t a, b;
118
+ memcpy(&a, str + i, 8);
119
+ memcpy(&b, rptr + i, 8);
120
+ if (a != b) {
121
+ a = __builtin_bswap64(a);
122
+ b = __builtin_bswap64(b);
123
+ return (a < b) ? -1 : 1;
124
+ }
125
+ }
126
+
127
+ for (; i < length; i++) {
128
+ if (str[i] != rptr[i]) {
129
+ return (str[i] < rptr[i]) ? -1 : 1;
130
+ }
114
131
  }
132
+
133
+ return 0;
115
134
  }
135
+ #endif
136
+ #endif
116
137
 
117
- static VALUE rstring_cache_fetch(rvalue_cache *cache, const char *str, const long length)
138
+ ALWAYS_INLINE(static) int rstring_cache_cmp(const char *str, const long length, VALUE rstring)
118
139
  {
119
- if (RB_UNLIKELY(length > JSON_RVALUE_CACHE_MAX_ENTRY_LENGTH)) {
120
- // Common names aren't likely to be very long. So we just don't
121
- // cache names above an arbitrary threshold.
122
- return Qfalse;
123
- }
140
+ const char *rstring_ptr;
141
+ long rstring_length;
124
142
 
125
- if (RB_UNLIKELY(!isalpha((unsigned char)str[0]))) {
126
- // Simple heuristic, if the first character isn't a letter,
127
- // we're much less likely to see this string again.
128
- // We mostly want to cache strings that are likely to be repeated.
129
- return Qfalse;
143
+ RSTRING_GETMEM(rstring, rstring_ptr, rstring_length);
144
+
145
+ if (length == rstring_length) {
146
+ return rstring_cache_memcmp(str, rstring_ptr, length);
147
+ } else {
148
+ return (int)(length - rstring_length);
130
149
  }
150
+ }
131
151
 
152
+ ALWAYS_INLINE(static) VALUE rstring_cache_fetch(rvalue_cache *cache, const char *str, const long length)
153
+ {
132
154
  int low = 0;
133
155
  int high = cache->length - 1;
134
- int mid = 0;
135
- int last_cmp = 0;
136
156
 
137
157
  while (low <= high) {
138
- mid = (high + low) >> 1;
158
+ int mid = (high + low) >> 1;
139
159
  VALUE entry = cache->entries[mid];
140
- last_cmp = rstring_cache_cmp(str, length, entry);
160
+ int cmp = rstring_cache_cmp(str, length, entry);
141
161
 
142
- if (last_cmp == 0) {
162
+ if (cmp == 0) {
143
163
  return entry;
144
- } else if (last_cmp > 0) {
164
+ } else if (cmp > 0) {
145
165
  low = mid + 1;
146
166
  } else {
147
167
  high = mid - 1;
148
168
  }
149
169
  }
150
170
 
151
- if (RB_UNLIKELY(memchr(str, '\\', length))) {
152
- // We assume the overwhelming majority of names don't need to be escaped.
153
- // But if they do, we have to fallback to the slow path.
154
- return Qfalse;
155
- }
156
-
157
171
  VALUE rstring = build_interned_string(str, length);
158
172
 
159
173
  if (cache->length < JSON_RVALUE_CACHE_CAPA) {
160
- if (last_cmp > 0) {
161
- mid += 1;
162
- }
163
-
164
- rvalue_cache_insert_at(cache, mid, rstring);
174
+ rvalue_cache_insert_at(cache, low, rstring);
165
175
  }
166
176
  return rstring;
167
177
  }
168
178
 
169
179
  static VALUE rsymbol_cache_fetch(rvalue_cache *cache, const char *str, const long length)
170
180
  {
171
- if (RB_UNLIKELY(length > JSON_RVALUE_CACHE_MAX_ENTRY_LENGTH)) {
172
- // Common names aren't likely to be very long. So we just don't
173
- // cache names above an arbitrary threshold.
174
- return Qfalse;
175
- }
176
-
177
- if (RB_UNLIKELY(!isalpha((unsigned char)str[0]))) {
178
- // Simple heuristic, if the first character isn't a letter,
179
- // we're much less likely to see this string again.
180
- // We mostly want to cache strings that are likely to be repeated.
181
- return Qfalse;
182
- }
183
-
184
181
  int low = 0;
185
182
  int high = cache->length - 1;
186
- int mid = 0;
187
- int last_cmp = 0;
188
183
 
189
184
  while (low <= high) {
190
- mid = (high + low) >> 1;
185
+ int mid = (high + low) >> 1;
191
186
  VALUE entry = cache->entries[mid];
192
- last_cmp = rstring_cache_cmp(str, length, rb_sym2str(entry));
187
+ int cmp = rstring_cache_cmp(str, length, rb_sym2str(entry));
193
188
 
194
- if (last_cmp == 0) {
189
+ if (cmp == 0) {
195
190
  return entry;
196
- } else if (last_cmp > 0) {
191
+ } else if (cmp > 0) {
197
192
  low = mid + 1;
198
193
  } else {
199
194
  high = mid - 1;
200
195
  }
201
196
  }
202
197
 
203
- if (RB_UNLIKELY(memchr(str, '\\', length))) {
204
- // We assume the overwhelming majority of names don't need to be escaped.
205
- // But if they do, we have to fallback to the slow path.
206
- return Qfalse;
207
- }
208
-
209
198
  VALUE rsymbol = build_symbol(str, length);
210
199
 
211
200
  if (cache->length < JSON_RVALUE_CACHE_CAPA) {
212
- if (last_cmp > 0) {
213
- mid += 1;
214
- }
215
-
216
- rvalue_cache_insert_at(cache, mid, rsymbol);
201
+ rvalue_cache_insert_at(cache, low, rsymbol);
217
202
  }
218
203
  return rsymbol;
219
204
  }
@@ -238,12 +223,12 @@ static rvalue_stack *rvalue_stack_spill(rvalue_stack *old_stack, VALUE *handle,
238
223
 
239
224
  static rvalue_stack *rvalue_stack_grow(rvalue_stack *stack, VALUE *handle, rvalue_stack **stack_ref)
240
225
  {
241
- long required = stack->capa * 2;
226
+ long required = stack->capa ? stack->capa * 2 : RVALUE_STACK_INITIAL_CAPA;
242
227
 
243
228
  if (stack->type == RVALUE_STACK_STACK_ALLOCATED) {
244
229
  stack = rvalue_stack_spill(stack, handle, stack_ref);
245
230
  } else {
246
- REALLOC_N(stack->ptr, VALUE, required);
231
+ JSON_SIZED_REALLOC_N(stack->ptr, VALUE, required, stack->capa);
247
232
  stack->capa = required;
248
233
  }
249
234
  return stack;
@@ -251,11 +236,15 @@ static rvalue_stack *rvalue_stack_grow(rvalue_stack *stack, VALUE *handle, rvalu
251
236
 
252
237
  static VALUE rvalue_stack_push(rvalue_stack *stack, VALUE value, VALUE *handle, rvalue_stack **stack_ref)
253
238
  {
239
+ JSON_ASSERT(stack->type != RVALUE_STACK_STACK_ALLOCATED || handle);
240
+
254
241
  if (RB_UNLIKELY(stack->head >= stack->capa)) {
255
242
  stack = rvalue_stack_grow(stack, handle, stack_ref);
256
243
  }
244
+
257
245
  stack->ptr[stack->head] = value;
258
246
  stack->head++;
247
+
259
248
  return value;
260
249
  }
261
250
 
@@ -273,35 +262,62 @@ static void rvalue_stack_mark(void *ptr)
273
262
  {
274
263
  rvalue_stack *stack = (rvalue_stack *)ptr;
275
264
  long index;
276
- for (index = 0; index < stack->head; index++) {
277
- rb_gc_mark(stack->ptr[index]);
265
+ if (stack && stack->ptr) {
266
+ for (index = 0; index < stack->head; index++) {
267
+ rb_gc_mark_movable(stack->ptr[index]);
268
+ }
278
269
  }
279
270
  }
280
271
 
272
+ static void rvalue_stack_free_buffer(rvalue_stack *stack)
273
+ {
274
+ JSON_SIZED_FREE_N(stack->ptr, stack->capa);
275
+ stack->ptr = NULL;
276
+ }
277
+
281
278
  static void rvalue_stack_free(void *ptr)
282
279
  {
283
280
  rvalue_stack *stack = (rvalue_stack *)ptr;
284
281
  if (stack) {
285
- ruby_xfree(stack->ptr);
286
- ruby_xfree(stack);
282
+ rvalue_stack_free_buffer(stack);
283
+ #ifndef HAVE_RUBY_TYPED_EMBEDDABLE
284
+ JSON_SIZED_FREE(stack);
285
+ #endif
287
286
  }
288
287
  }
289
288
 
290
289
  static size_t rvalue_stack_memsize(const void *ptr)
291
290
  {
292
291
  const rvalue_stack *stack = (const rvalue_stack *)ptr;
293
- return sizeof(rvalue_stack) + sizeof(VALUE) * stack->capa;
292
+ size_t memsize = sizeof(VALUE) * stack->capa;
293
+ #ifndef HAVE_RUBY_TYPED_EMBEDDABLE
294
+ memsize += sizeof(rvalue_stack);
295
+ #endif
296
+ return memsize;
297
+ }
298
+
299
+ static void rvalue_stack_compact(void *ptr)
300
+ {
301
+ rvalue_stack *stack = (rvalue_stack *)ptr;
302
+ long index;
303
+ if (stack && stack->ptr) {
304
+ for (index = 0; index < stack->head; index++) {
305
+ stack->ptr[index] = rb_gc_location(stack->ptr[index]);
306
+ }
307
+ }
294
308
  }
295
309
 
296
310
  static const rb_data_type_t JSON_Parser_rvalue_stack_type = {
297
- "JSON::Ext::Parser/rvalue_stack",
298
- {
311
+ .wrap_struct_name = "JSON::Ext::Parser/rvalue_stack",
312
+ .function = {
299
313
  .dmark = rvalue_stack_mark,
300
314
  .dfree = rvalue_stack_free,
301
315
  .dsize = rvalue_stack_memsize,
316
+ .dcompact = rvalue_stack_compact,
302
317
  },
303
- 0, 0,
304
- RUBY_TYPED_FREE_IMMEDIATELY,
318
+ // We deliberately don't declare rvalue_stack as RUBY_TYPED_WB_PROTECTED
319
+ // because it churns a lot of values so trigering write barriers every time is very costly.
320
+ .flags = RUBY_TYPED_THREAD_SAFE_FREE | RUBY_TYPED_EMBEDDABLE,
305
321
  };
306
322
 
307
323
  static rvalue_stack *rvalue_stack_spill(rvalue_stack *old_stack, VALUE *handle, rvalue_stack **stack_ref)
@@ -323,19 +339,204 @@ static void rvalue_stack_eagerly_release(VALUE handle)
323
339
  if (handle) {
324
340
  rvalue_stack *stack;
325
341
  TypedData_Get_Struct(handle, rvalue_stack, &JSON_Parser_rvalue_stack_type, stack);
326
- RTYPEDDATA_DATA(handle) = NULL;
342
+ #ifdef HAVE_RUBY_TYPED_EMBEDDABLE
343
+ rvalue_stack_free_buffer(stack);
344
+ #else
327
345
  rvalue_stack_free(stack);
346
+ RTYPEDDATA_DATA(handle) = NULL;
347
+ #endif
348
+ }
349
+ }
350
+
351
+ /* frame stack */
352
+
353
+ // Iterative (non-recursive) parsing keeps an explicit stack of the containers
354
+ // currently being built, instead of relying on the C call stack. Each frame
355
+ // only needs enough bookkeeping to close its container: which kind it is, the
356
+ // rvalue_stack position where its children start (so we know how many to pop),
357
+ // and the cursor at its opening brace (used to rewind for duplicate key
358
+ // errors). Frames hold no VALUEs, so this stack needs no GC marking; it reuses
359
+ // the same stack-allocated-with-heap-spill strategy as the rvalue_stack so that
360
+ // it's freed even if parsing raises.
361
+ //
362
+ // The lifecycle helpers below (grow/push/peek/pop/spill/free/eagerly_release
363
+ // and the rb_data_type_t) deliberately mirror their rvalue_stack counterparts
364
+ // -- the element type and the absence of a mark function are the only real
365
+ // differences. Keep the two in sync: a fix to the spill/release or
366
+ // HAVE_RUBY_TYPED_EMBEDDABLE handling in one almost certainly belongs in the
367
+ // other.
368
+ #define JSON_FRAME_STACK_INITIAL_CAPA 32
369
+
370
+ enum json_frame_type {
371
+ JSON_FRAME_ROOT, // == JSON_PHASE_DONE
372
+ JSON_FRAME_ARRAY, // == JSON_PHASE_ARRAY_COMMA
373
+ JSON_FRAME_OBJECT, // = JSON_PHASE_OBJECT_COMMA
374
+ };
375
+
376
+ // Where a frame is within its container's grammar. This is the entirety of the
377
+ // parser's "what to do next" state: json_parse_any dispatches on the top
378
+ // frame's phase and holds no resume state in C locals, so a parse can stop at
379
+ // any value boundary and be resumed purely from the (persistable) frame stack.
380
+ //
381
+ // The first three phases are deliberately equal to the corresponding json_frame_type
382
+ // to simplify the transition of phase in json_value_completed.
383
+ enum json_frame_phase {
384
+ JSON_PHASE_DONE = JSON_FRAME_ROOT, // root only: the document value has been parsed
385
+ JSON_PHASE_ARRAY_COMMA = JSON_FRAME_ARRAY, // after a value: expecting ',' or the closing ']'
386
+ JSON_PHASE_OBJECT_COMMA = JSON_FRAME_OBJECT, // after a value: expecting ',' or the closing '}'
387
+ JSON_PHASE_VALUE, // expecting a value (document root, array element, or object value after ':')
388
+ JSON_PHASE_OBJECT_KEY, // expecting a '"' key (after '{' or ',')
389
+ JSON_PHASE_OBJECT_COLON, // object only: after a key, expecting ':'
390
+ };
391
+
392
+ typedef struct json_frame_struct {
393
+ enum json_frame_type type;
394
+ enum json_frame_phase phase;
395
+ long value_stack_head; // rvalue_stack->head when this container opened
396
+ size_t start_offset; // object frames only (the '{'); NULL otherwise
397
+ } json_frame;
398
+
399
+ typedef struct json_frame_stack_struct {
400
+ enum rvalue_stack_type type; // shared with rvalue_stack: is ptr stack- or heap-allocated
401
+ long capa;
402
+ long head;
403
+ json_frame *ptr;
404
+ } json_frame_stack;
405
+
406
+ typedef struct JSON_ParserStruct {
407
+ VALUE on_load_proc;
408
+ VALUE decimal_class;
409
+ ID decimal_method_id;
410
+ int max_nesting;
411
+ bool allow_comments;
412
+ bool allow_duplicate_key;
413
+ bool allow_nan;
414
+ bool allow_trailing_comma;
415
+ bool allow_control_characters;
416
+ bool allow_invalid_escape;
417
+ bool symbolize_names;
418
+ bool freeze;
419
+ } JSON_ParserConfig;
420
+
421
+ typedef struct JSON_ParserStateStruct {
422
+ VALUE *value_stack_handle;
423
+ VALUE *frame_stack_handle;
424
+ const char *start;
425
+ const char *cursor;
426
+ const char *end;
427
+ rvalue_stack *value_stack;
428
+ json_frame_stack *frames;
429
+ rvalue_cache name_cache;
430
+ int in_array;
431
+ int current_nesting;
432
+ VALUE parser;
433
+ } JSON_ParserState;
434
+
435
+ static json_frame_stack *json_frame_stack_spill(json_frame_stack *old_stack, VALUE *handle, json_frame_stack **stack_ref);
436
+
437
+ static json_frame_stack *json_frame_stack_grow(json_frame_stack *stack, VALUE *handle, json_frame_stack **stack_ref)
438
+ {
439
+ long required = stack->capa ? stack->capa * 2 : JSON_FRAME_STACK_INITIAL_CAPA;
440
+
441
+ if (stack->type == RVALUE_STACK_STACK_ALLOCATED) {
442
+ stack = json_frame_stack_spill(stack, handle, stack_ref);
443
+ } else {
444
+ JSON_SIZED_REALLOC_N(stack->ptr, json_frame, required, stack->capa);
445
+ stack->capa = required;
446
+ }
447
+ return stack;
448
+ }
449
+
450
+ static json_frame *json_frame_stack_push(JSON_ParserState *state, json_frame frame)
451
+ {
452
+ json_frame_stack *stack = state->frames;
453
+
454
+ JSON_ASSERT(stack->type != RVALUE_STACK_STACK_ALLOCATED || state->frame_stack_handle);
455
+
456
+ if (RB_UNLIKELY(stack->head >= stack->capa)) {
457
+ stack = json_frame_stack_grow(stack, state->frame_stack_handle, &state->frames);
458
+ }
459
+
460
+ json_frame *frame_ptr = &stack->ptr[stack->head++];
461
+ *frame_ptr = frame;
462
+ return frame_ptr;
463
+ }
464
+
465
+ static inline json_frame *json_frame_stack_peek(json_frame_stack *stack)
466
+ {
467
+ return &stack->ptr[stack->head - 1];
468
+ }
469
+
470
+ static inline void json_frame_stack_pop(json_frame_stack *stack)
471
+ {
472
+ stack->head--;
473
+ }
474
+
475
+ static void json_frame_stack_free_buffer(json_frame_stack *stack)
476
+ {
477
+ JSON_SIZED_FREE_N(stack->ptr, stack->capa);
478
+ stack->ptr = NULL;
479
+ }
480
+
481
+ static void json_frame_stack_free(void *ptr)
482
+ {
483
+ json_frame_stack *stack = (json_frame_stack *)ptr;
484
+ if (stack) {
485
+ json_frame_stack_free_buffer(stack);
486
+ #ifndef HAVE_RUBY_TYPED_EMBEDDABLE
487
+ JSON_SIZED_FREE(stack);
488
+ #endif
328
489
  }
329
490
  }
330
491
 
492
+ static size_t json_frame_stack_memsize(const void *ptr)
493
+ {
494
+ const json_frame_stack *stack = (const json_frame_stack *)ptr;
495
+
496
+ size_t memsize = sizeof(json_frame) * stack->capa;
497
+ #ifndef HAVE_RUBY_TYPED_EMBEDDABLE
498
+ memsize += sizeof(json_frame_stack);
499
+ #endif
500
+ return memsize;
501
+ }
331
502
 
332
- #ifndef HAVE_STRNLEN
333
- static size_t strnlen(const char *s, size_t maxlen)
503
+ static const rb_data_type_t JSON_Parser_frame_stack_type = {
504
+ .wrap_struct_name = "JSON::Ext::Parser/frame_stack",
505
+ .function = {
506
+ .dmark = NULL,
507
+ .dfree = json_frame_stack_free,
508
+ .dsize = json_frame_stack_memsize,
509
+ },
510
+ .flags = RUBY_TYPED_THREAD_SAFE_FREE | RUBY_TYPED_WB_PROTECTED | RUBY_TYPED_EMBEDDABLE,
511
+ };
512
+
513
+ static json_frame_stack *json_frame_stack_spill(json_frame_stack *old_stack, VALUE *handle, json_frame_stack **stack_ref)
334
514
  {
335
- char *p;
336
- return ((p = memchr(s, '\0', maxlen)) ? p - s : maxlen);
515
+ json_frame_stack *stack;
516
+ *handle = TypedData_Make_Struct(0, json_frame_stack, &JSON_Parser_frame_stack_type, stack);
517
+ *stack_ref = stack;
518
+ MEMCPY(stack, old_stack, json_frame_stack, 1);
519
+
520
+ stack->capa = old_stack->capa << 1;
521
+ stack->ptr = ALLOC_N(json_frame, stack->capa);
522
+ stack->type = RVALUE_STACK_HEAP_ALLOCATED;
523
+ MEMCPY(stack->ptr, old_stack->ptr, json_frame, old_stack->head);
524
+ return stack;
337
525
  }
526
+
527
+ static void json_frame_stack_eagerly_release(VALUE handle)
528
+ {
529
+ if (handle) {
530
+ json_frame_stack *stack;
531
+ TypedData_Get_Struct(handle, json_frame_stack, &JSON_Parser_frame_stack_type, stack);
532
+ #ifdef HAVE_RUBY_TYPED_EMBEDDABLE
533
+ json_frame_stack_free_buffer(stack);
534
+ #else
535
+ json_frame_stack_free(stack);
536
+ RTYPEDDATA_DATA(handle) = NULL;
338
537
  #endif
538
+ }
539
+ }
339
540
 
340
541
  static int convert_UTF32_to_UTF8(char *buf, uint32_t ch)
341
542
  {
@@ -363,37 +564,32 @@ static int convert_UTF32_to_UTF8(char *buf, uint32_t ch)
363
564
  return len;
364
565
  }
365
566
 
366
- typedef struct JSON_ParserStruct {
367
- VALUE on_load_proc;
368
- VALUE decimal_class;
369
- ID decimal_method_id;
370
- int max_nesting;
371
- bool allow_nan;
372
- bool allow_trailing_comma;
373
- bool parsing_name;
374
- bool symbolize_names;
375
- bool freeze;
376
- } JSON_ParserConfig;
567
+ static inline size_t rest(JSON_ParserState *state) {
568
+ return state->end - state->cursor;
569
+ }
377
570
 
378
- typedef struct JSON_ParserStateStruct {
379
- VALUE stack_handle;
380
- const char *start;
381
- const char *cursor;
382
- const char *end;
383
- rvalue_stack *stack;
384
- rvalue_cache name_cache;
385
- int in_array;
386
- int current_nesting;
387
- } JSON_ParserState;
571
+ static inline bool eos(JSON_ParserState *state) {
572
+ return state->cursor >= state->end;
573
+ }
388
574
 
575
+ static inline char peek(JSON_ParserState *state)
576
+ {
577
+ if (RB_UNLIKELY(eos(state))) {
578
+ return 0;
579
+ }
580
+ return *state->cursor;
581
+ }
389
582
 
390
- #define PARSE_ERROR_FRAGMENT_LEN 32
391
- #ifdef RBIMPL_ATTR_NORETURN
392
- RBIMPL_ATTR_NORETURN()
393
- #endif
394
- static void raise_parse_error(const char *format, JSON_ParserState *state)
583
+ static void cursor_position(JSON_ParserState *state, long *line_out, long *column_out)
395
584
  {
396
- unsigned char buffer[PARSE_ERROR_FRAGMENT_LEN + 3];
585
+ JSON_ASSERT(!state->parser);
586
+ JSON_ASSERT(state->cursor);
587
+ JSON_ASSERT(state->cursor <= state->end);
588
+
589
+ // Redundant but helpful for hardening
590
+ if (RB_UNLIKELY(state->cursor > state->end)) {
591
+ state->cursor = state->end;
592
+ }
397
593
 
398
594
  const char *cursor = state->cursor;
399
595
  long column = 0;
@@ -401,6 +597,7 @@ static void raise_parse_error(const char *format, JSON_ParserState *state)
401
597
 
402
598
  while (cursor >= state->start) {
403
599
  if (*cursor-- == '\n') {
600
+ line++;
404
601
  break;
405
602
  }
406
603
  column++;
@@ -411,6 +608,15 @@ static void raise_parse_error(const char *format, JSON_ParserState *state)
411
608
  line++;
412
609
  }
413
610
  }
611
+ *line_out = line;
612
+ *column_out = column;
613
+ }
614
+
615
+ #define PARSE_ERROR_FRAGMENT_LEN 32
616
+
617
+ static VALUE build_parse_error_message(const char *format, JSON_ParserState *state)
618
+ {
619
+ unsigned char buffer[PARSE_ERROR_FRAGMENT_LEN + 3];
414
620
 
415
621
  const char *ptr = "EOF";
416
622
  if (state->cursor && state->cursor < state->end) {
@@ -442,23 +648,92 @@ static void raise_parse_error(const char *format, JSON_ParserState *state)
442
648
  }
443
649
  }
444
650
 
445
- VALUE msg = rb_sprintf(format, ptr);
446
- VALUE message = rb_enc_sprintf(enc_utf8, "%s at line %ld column %ld", RSTRING_PTR(msg), line, column);
447
- RB_GC_GUARD(msg);
651
+ return rb_enc_sprintf(enc_utf8, format, ptr);
652
+ }
448
653
 
449
- VALUE exc = rb_exc_new_str(rb_path2class("JSON::ParserError"), message);
450
- rb_ivar_set(exc, rb_intern("@line"), LONG2NUM(line));
451
- rb_ivar_set(exc, rb_intern("@column"), LONG2NUM(column));
452
- rb_exc_raise(exc);
654
+ static VALUE json_path_new(JSON_ParserState *state, VALUE duplicate_key)
655
+ {
656
+ VALUE path = rb_ary_new_capa(state->current_nesting);
657
+
658
+ json_frame_stack *frames = state->frames;
659
+ rvalue_stack *values = state->value_stack;
660
+
661
+ for (long depth = 1; depth < frames->head; depth++) {
662
+ json_frame *frame = &frames->ptr[depth];
663
+
664
+ bool innermost = depth == frames->head - 1;
665
+ long child_head = innermost ? values->head : frames->ptr[depth + 1].value_stack_head;
666
+ long count = child_head - frame->value_stack_head;
667
+
668
+ if (frame->type == JSON_FRAME_ARRAY) {
669
+ rb_ary_push(path, LONG2NUM(frame->phase == JSON_PHASE_ARRAY_COMMA ? count - 1 : count));
670
+ } else if (innermost && !UNDEF_P(duplicate_key)) {
671
+ rb_ary_push(path, duplicate_key);
672
+ } else if (count & 1) {
673
+ rb_ary_push(path, values->ptr[child_head - 1]);
674
+ } else if (frame->phase == JSON_PHASE_OBJECT_COMMA && count >= 2) {
675
+ rb_ary_push(path, values->ptr[child_head - 2]);
676
+ } else {
677
+ break;
678
+ }
679
+ }
680
+
681
+ return path;
453
682
  }
454
683
 
455
- #ifdef RBIMPL_ATTR_NORETURN
456
- RBIMPL_ATTR_NORETURN()
457
- #endif
458
- static void raise_parse_error_at(const char *format, JSON_ParserState *state, const char *at)
684
+ static VALUE parse_error_new(JSON_ParserState *state, VALUE message, long line, long column, bool eos)
685
+ {
686
+ VALUE exc = rb_exc_new_str(eParserError, message);
687
+ rb_ivar_set(exc, i_at_line, LONG2NUM(line));
688
+ rb_ivar_set(exc, i_at_column, LONG2NUM(column));
689
+ rb_ivar_set(exc, i_at_json_path, json_path_new(state, Qundef));
690
+ return exc;
691
+ }
692
+
693
+ NORETURN(static) void raise_parse_error(const char *format, JSON_ParserState *state, bool eos)
694
+ {
695
+ if (state->parser) {
696
+ if (eos) {
697
+ // the error will be swallowed by ResumableParser#parse, so no
698
+ // point building a message or backtrace.
699
+ rb_throw_obj(state->parser, state->parser);
700
+ } else {
701
+ // line and columns can't be accurate in resumable
702
+ rb_exc_raise(parse_error_new(state, build_parse_error_message(format, state), 0, 0, eos));
703
+ }
704
+ } else {
705
+ VALUE message = build_parse_error_message(format, state);
706
+ long line, column;
707
+ cursor_position(state, &line, &column);
708
+ rb_str_catf(message, " at line %ld column %ld", line, column);
709
+ rb_exc_raise(parse_error_new(state, message, line, column, eos));
710
+ }
711
+ }
712
+
713
+ NORETURN(static) void raise_eos_error(const char *format, JSON_ParserState *state)
714
+ {
715
+ raise_parse_error(format, state, true);
716
+ }
717
+
718
+ NORETURN(static) void raise_syntax_error(const char *format, JSON_ParserState *state)
719
+ {
720
+ raise_parse_error(format, state, false);
721
+ }
722
+
723
+ NORETURN(static) void raise_parse_error_at(const char *format, JSON_ParserState *state, const char *at, bool eos)
459
724
  {
460
725
  state->cursor = at;
461
- raise_parse_error(format, state);
726
+ raise_parse_error(format, state, eos);
727
+ }
728
+
729
+ NORETURN(static) void raise_eos_error_at(const char *format, JSON_ParserState *state, const char *at)
730
+ {
731
+ raise_parse_error_at(format, state, at, true);
732
+ }
733
+
734
+ NORETURN(static) void raise_syntax_error_at(const char *format, JSON_ParserState *state, const char *at)
735
+ {
736
+ raise_parse_error_at(format, state, at, false);
462
737
  }
463
738
 
464
739
  /* unicode */
@@ -480,23 +755,24 @@ static const signed char digit_values[256] = {
480
755
  -1, -1, -1, -1, -1, -1, -1
481
756
  };
482
757
 
483
- static uint32_t unescape_unicode(JSON_ParserState *state, const unsigned char *p)
484
- {
485
- signed char b;
486
- uint32_t result = 0;
487
- b = digit_values[p[0]];
488
- if (b < 0) raise_parse_error_at("incomplete unicode character escape sequence at %s", state, (char *)p - 2);
489
- result = (result << 4) | (unsigned char)b;
490
- b = digit_values[p[1]];
491
- if (b < 0) raise_parse_error_at("incomplete unicode character escape sequence at %s", state, (char *)p - 2);
492
- result = (result << 4) | (unsigned char)b;
493
- b = digit_values[p[2]];
494
- if (b < 0) raise_parse_error_at("incomplete unicode character escape sequence at %s", state, (char *)p - 2);
495
- result = (result << 4) | (unsigned char)b;
496
- b = digit_values[p[3]];
497
- if (b < 0) raise_parse_error_at("incomplete unicode character escape sequence at %s", state, (char *)p - 2);
498
- result = (result << 4) | (unsigned char)b;
499
- return result;
758
+ static uint32_t unescape_unicode(JSON_ParserState *state, const char *sp, const char *spe)
759
+ {
760
+ if (RB_UNLIKELY(sp > spe - 4)) {
761
+ raise_eos_error_at("incomplete unicode character escape sequence at %s", state, sp - 2);
762
+ }
763
+
764
+ const unsigned char *p = (const unsigned char *)sp;
765
+
766
+ const signed char b0 = digit_values[p[0]];
767
+ const signed char b1 = digit_values[p[1]];
768
+ const signed char b2 = digit_values[p[2]];
769
+ const signed char b3 = digit_values[p[3]];
770
+
771
+ if (RB_UNLIKELY((signed char)(b0 | b1 | b2 | b3) < 0)) {
772
+ raise_syntax_error_at("incomplete unicode character escape sequence at %s", state, sp - 2);
773
+ }
774
+
775
+ return ((uint32_t)b0 << 12) | ((uint32_t)b1 << 8) | ((uint32_t)b2 << 4) | (uint32_t)b3;
500
776
  }
501
777
 
502
778
  #define GET_PARSER_CONFIG \
@@ -505,65 +781,112 @@ static uint32_t unescape_unicode(JSON_ParserState *state, const unsigned char *p
505
781
 
506
782
  static const rb_data_type_t JSON_ParserConfig_type;
507
783
 
508
- static const bool whitespace[256] = {
509
- [' '] = 1,
510
- ['\t'] = 1,
511
- ['\n'] = 1,
512
- ['\r'] = 1,
513
- ['/'] = 1,
514
- };
784
+ NOINLINE(static) void
785
+ json_eat_comments(JSON_ParserState *state, JSON_ParserConfig *config, const char *resume_pos)
786
+ {
787
+ if (!config->allow_comments) {
788
+ raise_syntax_error("unexpected token %s", state);
789
+ }
515
790
 
516
- static void
517
- json_eat_comments(JSON_ParserState *state)
518
- {
519
- if (state->cursor + 1 < state->end) {
520
- switch(state->cursor[1]) {
521
- case '/': {
522
- state->cursor = memchr(state->cursor, '\n', state->end - state->cursor);
523
- if (!state->cursor) {
524
- state->cursor = state->end;
525
- } else {
526
- state->cursor++;
791
+ const char *start = state->cursor;
792
+ // An incomplete comment suspends a resumable parse by rewinding the cursor
793
+ // and throwing. Callers that already consumed a token not yet committed to
794
+ // the frame stack pass resume_pos so the rewind re-reads that token too.
795
+ // Non-resumable error positions keep pointing at the comment either way.
796
+ const char *rewind_pos = (state->parser && resume_pos) ? resume_pos : start;
797
+ state->cursor++;
798
+
799
+ switch (peek(state)) {
800
+ case '/': {
801
+ const char *newline = memchr(state->cursor, '\n', state->end - state->cursor);
802
+ if (!newline) {
803
+ // state->parser marks resumable mode, where the buffer end is only a
804
+ // chunk boundary: the terminating newline may still arrive, so leave
805
+ // the comment unterminated instead of consuming to end as a one-shot
806
+ // parse would.
807
+ if (state->parser) {
808
+ raise_eos_error_at("unterminated comment, expected end of line", state, rewind_pos);
527
809
  }
528
- break;
810
+ state->cursor = state->end;
811
+ } else {
812
+ state->cursor = newline + 1;
529
813
  }
530
- case '*': {
531
- state->cursor += 2;
532
- while (true) {
533
- state->cursor = memchr(state->cursor, '*', state->end - state->cursor);
534
- if (!state->cursor) {
535
- raise_parse_error_at("unexpected end of input, expected closing '*/'", state, state->end);
536
- } else {
537
- state->cursor++;
538
- if (state->cursor < state->end && *state->cursor == '/') {
539
- state->cursor++;
540
- break;
541
- }
542
- }
814
+ break;
815
+ }
816
+ case '*': {
817
+ state->cursor++;
818
+
819
+ while (true) {
820
+ const char *next_match = memchr(state->cursor, '*', state->end - state->cursor);
821
+ if (!next_match) {
822
+ raise_eos_error_at("unterminated comment, expected closing '*/'", state, rewind_pos);
823
+ }
824
+
825
+ state->cursor = next_match + 1;
826
+ if (peek(state) == '/') {
827
+ state->cursor++;
828
+ break;
543
829
  }
544
- break;
545
830
  }
546
- default:
547
- raise_parse_error("unexpected token %s", state);
548
- break;
831
+ break;
549
832
  }
550
- } else {
551
- raise_parse_error("unexpected token %s", state);
833
+ default:
834
+ raise_parse_error_at("unexpected token %s", state, eos(state) ? rewind_pos : start, eos(state));
835
+ break;
552
836
  }
553
837
  }
554
838
 
555
- static inline void
556
- json_eat_whitespace(JSON_ParserState *state)
839
+ ALWAYS_INLINE(static) void
840
+ json_eat_whitespace_resume_at(JSON_ParserState *state, JSON_ParserConfig *config, bool include_comments, const char *resume_pos)
557
841
  {
558
- while (state->cursor < state->end && RB_UNLIKELY(whitespace[(unsigned char)*state->cursor])) {
559
- if (RB_LIKELY(*state->cursor != '/')) {
560
- state->cursor++;
561
- } else {
562
- json_eat_comments(state);
842
+ while (true) {
843
+ switch (peek(state)) {
844
+ case ' ':
845
+ state->cursor++;
846
+ break;
847
+ case '\n':
848
+ state->cursor++;
849
+
850
+ // Heuristic: if we see a newline, there is likely consecutive spaces after it.
851
+ #if JSON_CPU_LITTLE_ENDIAN_64BITS
852
+ while (rest(state) > 8) {
853
+ uint64_t chunk;
854
+ memcpy(&chunk, state->cursor, sizeof(uint64_t));
855
+ if (chunk == 0x2020202020202020) {
856
+ state->cursor += 8;
857
+ continue;
858
+ }
859
+
860
+ uint32_t consecutive_spaces = trailing_zeros64(chunk ^ 0x2020202020202020) / CHAR_BIT;
861
+ state->cursor += consecutive_spaces;
862
+ break;
863
+ }
864
+ #endif
865
+ break;
866
+ case '\t':
867
+ case '\r':
868
+ state->cursor++;
869
+ break;
870
+ case '/':
871
+ if (!include_comments) {
872
+ return;
873
+ }
874
+
875
+ json_eat_comments(state, config, resume_pos);
876
+ break;
877
+
878
+ default:
879
+ return;
563
880
  }
564
881
  }
565
882
  }
566
883
 
884
+ ALWAYS_INLINE(static) void
885
+ json_eat_whitespace(JSON_ParserState *state, JSON_ParserConfig *config, bool include_comments)
886
+ {
887
+ json_eat_whitespace_resume_at(state, config, include_comments, NULL);
888
+ }
889
+
567
890
  static inline VALUE build_string(const char *start, const char *end, bool intern, bool symbolize)
568
891
  {
569
892
  if (symbolize) {
@@ -590,11 +913,22 @@ static inline VALUE build_string(const char *start, const char *end, bool intern
590
913
  return result;
591
914
  }
592
915
 
593
- static inline VALUE json_string_fastpath(JSON_ParserState *state, const char *string, const char *stringEnd, bool is_name, bool intern, bool symbolize)
916
+ static inline bool json_string_cacheable_p(const char *string, size_t length)
917
+ {
918
+ // We mostly want to cache strings that are likely to be repeated.
919
+ // Simple heuristics:
920
+ // - Common names aren't likely to be very long. So we just don't cache names above an arbitrary threshold.
921
+ // - If the first character isn't a letter, we're much less likely to see this string again.
922
+ return length <= JSON_RVALUE_CACHE_MAX_ENTRY_LENGTH && rb_isalpha(string[0]);
923
+ }
924
+
925
+ static inline VALUE json_string_fastpath(JSON_ParserState *state, JSON_ParserConfig *config, const char *string, const char *stringEnd, bool is_name)
594
926
  {
927
+ bool intern = is_name || config->freeze;
928
+ bool symbolize = is_name && config->symbolize_names;
595
929
  size_t bufferSize = stringEnd - string;
596
930
 
597
- if (is_name && state->in_array) {
931
+ if (is_name && state->in_array && RB_LIKELY(json_string_cacheable_p(string, bufferSize))) {
598
932
  VALUE cached_key;
599
933
  if (RB_UNLIKELY(symbolize)) {
600
934
  cached_key = rsymbol_cache_fetch(&state->name_cache, string, bufferSize);
@@ -610,104 +944,134 @@ static inline VALUE json_string_fastpath(JSON_ParserState *state, const char *st
610
944
  return build_string(string, stringEnd, intern, symbolize);
611
945
  }
612
946
 
613
- static VALUE json_string_unescape(JSON_ParserState *state, const char *string, const char *stringEnd, bool is_name, bool intern, bool symbolize)
614
- {
615
- size_t bufferSize = stringEnd - string;
616
- const char *p = string, *pe = string, *unescape, *bufferStart;
617
- char *buffer;
618
- int unescape_len;
619
- char buf[4];
947
+ #define JSON_MAX_UNESCAPE_POSITIONS 16
948
+ typedef struct _json_unescape_positions {
949
+ long size;
950
+ const char **positions;
951
+ unsigned long additional_backslashes;
952
+ } JSON_UnescapePositions;
620
953
 
621
- if (is_name && state->in_array) {
622
- VALUE cached_key;
623
- if (RB_UNLIKELY(symbolize)) {
624
- cached_key = rsymbol_cache_fetch(&state->name_cache, string, bufferSize);
625
- } else {
626
- cached_key = rstring_cache_fetch(&state->name_cache, string, bufferSize);
954
+ static inline const char *json_next_backslash(const char *pe, const char *stringEnd, JSON_UnescapePositions *positions)
955
+ {
956
+ while (positions->size) {
957
+ positions->size--;
958
+ const char *next_position = positions->positions[0];
959
+ positions->positions++;
960
+ if (next_position >= pe) {
961
+ return next_position;
627
962
  }
963
+ }
628
964
 
629
- if (RB_LIKELY(cached_key)) {
630
- return cached_key;
631
- }
965
+ if (positions->additional_backslashes) {
966
+ positions->additional_backslashes--;
967
+ return memchr(pe, '\\', stringEnd - pe);
632
968
  }
633
969
 
970
+ return NULL;
971
+ }
972
+
973
+ NOINLINE(static) VALUE json_string_unescape(JSON_ParserState *state, JSON_ParserConfig *config, const char *string, const char *stringEnd, bool is_name, JSON_UnescapePositions *positions)
974
+ {
975
+ bool intern = is_name || config->freeze;
976
+ bool symbolize = is_name && config->symbolize_names;
977
+ size_t bufferSize = stringEnd - string;
978
+ const char *p = string, *pe = string, *bufferStart;
979
+ char *buffer;
980
+
634
981
  VALUE result = rb_str_buf_new(bufferSize);
635
982
  rb_enc_associate_index(result, utf8_encindex);
636
983
  buffer = RSTRING_PTR(result);
637
984
  bufferStart = buffer;
638
985
 
639
- while (pe < stringEnd && (pe = memchr(pe, '\\', stringEnd - pe))) {
640
- unescape = (char *) "?";
641
- unescape_len = 1;
986
+ #define APPEND_CHAR(chr) *buffer++ = chr; p = ++pe;
987
+
988
+ while (pe < stringEnd && (pe = json_next_backslash(pe, stringEnd, positions))) {
642
989
  if (pe > p) {
643
990
  MEMCPY(buffer, p, char, pe - p);
644
991
  buffer += pe - p;
645
992
  }
646
993
  switch (*++pe) {
994
+ case '"':
995
+ case '/':
996
+ p = pe; // nothing to unescape just need to skip the backslash
997
+ break;
998
+ case '\\':
999
+ APPEND_CHAR('\\');
1000
+ break;
647
1001
  case 'n':
648
- unescape = (char *) "\n";
1002
+ APPEND_CHAR('\n');
649
1003
  break;
650
1004
  case 'r':
651
- unescape = (char *) "\r";
1005
+ APPEND_CHAR('\r');
652
1006
  break;
653
1007
  case 't':
654
- unescape = (char *) "\t";
655
- break;
656
- case '"':
657
- unescape = (char *) "\"";
658
- break;
659
- case '\\':
660
- unescape = (char *) "\\";
1008
+ APPEND_CHAR('\t');
661
1009
  break;
662
1010
  case 'b':
663
- unescape = (char *) "\b";
1011
+ APPEND_CHAR('\b');
664
1012
  break;
665
1013
  case 'f':
666
- unescape = (char *) "\f";
1014
+ APPEND_CHAR('\f');
667
1015
  break;
668
- case 'u':
669
- if (pe > stringEnd - 5) {
670
- raise_parse_error_at("incomplete unicode character escape sequence at %s", state, p);
671
- } else {
672
- uint32_t ch = unescape_unicode(state, (unsigned char *) ++pe);
673
- pe += 3;
674
- /* To handle values above U+FFFF, we take a sequence of
675
- * \uXXXX escapes in the U+D800..U+DBFF then
676
- * U+DC00..U+DFFF ranges, take the low 10 bits from each
677
- * to make a 20-bit number, then add 0x10000 to get the
678
- * final codepoint.
679
- *
680
- * See Unicode 15: 3.8 "Surrogates", 5.3 "Handling
681
- * Surrogate Pairs in UTF-16", and 23.6 "Surrogates
682
- * Area".
683
- */
684
- if ((ch & 0xFC00) == 0xD800) {
685
- pe++;
686
- if (pe > stringEnd - 6) {
687
- raise_parse_error_at("incomplete surrogate pair at %s", state, p);
688
- }
689
- if (pe[0] == '\\' && pe[1] == 'u') {
690
- uint32_t sur = unescape_unicode(state, (unsigned char *) pe + 2);
691
- ch = (((ch & 0x3F) << 10) | ((((ch >> 6) & 0xF) + 1) << 16)
692
- | (sur & 0x3FF));
693
- pe += 5;
694
- } else {
695
- unescape = (char *) "?";
696
- break;
1016
+ case 'u': {
1017
+ uint32_t ch = unescape_unicode(state, ++pe, stringEnd);
1018
+ pe += 3;
1019
+ /* To handle values above U+FFFF, we take a sequence of
1020
+ * \uXXXX escapes in the U+D800..U+DBFF then
1021
+ * U+DC00..U+DFFF ranges, take the low 10 bits from each
1022
+ * to make a 20-bit number, then add 0x10000 to get the
1023
+ * final codepoint.
1024
+ *
1025
+ * See Unicode 15: 3.8 "Surrogates", 5.3 "Handling
1026
+ * Surrogate Pairs in UTF-16", and 23.6 "Surrogates
1027
+ * Area".
1028
+ */
1029
+ if ((ch & 0xFC00) == 0xD800) {
1030
+ pe++;
1031
+ if (RB_LIKELY((pe <= stringEnd - 6) && memcmp(pe, "\\u", 2) == 0)) {
1032
+ uint32_t sur = unescape_unicode(state, pe + 2, stringEnd);
1033
+
1034
+ if (RB_UNLIKELY((sur & 0xFC00) != 0xDC00)) {
1035
+ raise_syntax_error_at("invalid surrogate pair at %s", state, p);
697
1036
  }
1037
+
1038
+ ch = (((ch & 0x3F) << 10) | ((((ch >> 6) & 0xF) + 1) << 16) | (sur & 0x3FF));
1039
+ pe += 5;
1040
+ } else {
1041
+ raise_syntax_error_at("incomplete surrogate pair at %s", state, p);
1042
+ break;
698
1043
  }
699
- unescape_len = convert_UTF32_to_UTF8(buf, ch);
700
- unescape = buf;
1044
+ } else if ((ch & 0xFC00) == 0xDC00) {
1045
+ raise_syntax_error_at("unpaired trailing surrogate at %s", state, p);
1046
+ break;
701
1047
  }
1048
+
1049
+ int unescape_len = convert_UTF32_to_UTF8(buffer, ch);
1050
+ buffer += unescape_len;
1051
+ p = ++pe;
702
1052
  break;
1053
+ }
1054
+ case 0:
1055
+ return Qundef;
703
1056
  default:
704
- p = pe;
705
- continue;
1057
+ if ((unsigned char)*pe < 0x20) {
1058
+ if (!config->allow_control_characters) {
1059
+ if (*pe == '\n') {
1060
+ raise_syntax_error_at("Invalid unescaped newline character (\\n) in string: %s", state, pe - 1);
1061
+ }
1062
+ raise_syntax_error_at("invalid ASCII control character in string: %s", state, pe - 1);
1063
+ }
1064
+ }
1065
+
1066
+ if (config->allow_invalid_escape) {
1067
+ APPEND_CHAR(*pe);
1068
+ } else {
1069
+ raise_syntax_error_at("invalid escape character in string: %s", state, pe - 1);
1070
+ }
1071
+ break;
706
1072
  }
707
- MEMCPY(buffer, unescape, char, unescape_len);
708
- buffer += unescape_len;
709
- p = ++pe;
710
1073
  }
1074
+ #undef APPEND_CHAR
711
1075
 
712
1076
  if (stringEnd > p) {
713
1077
  MEMCPY(buffer, p, char, stringEnd - p);
@@ -718,87 +1082,124 @@ static VALUE json_string_unescape(JSON_ParserState *state, const char *string, c
718
1082
  if (symbolize) {
719
1083
  result = rb_str_intern(result);
720
1084
  } else if (intern) {
721
- result = rb_funcall(rb_str_freeze(result), i_uminus, 0);
1085
+ result = rb_str_to_interned_str(result);
722
1086
  }
723
1087
 
724
1088
  return result;
725
1089
  }
726
1090
 
727
- #define MAX_FAST_INTEGER_SIZE 18
728
- static inline VALUE fast_decode_integer(const char *p, const char *pe)
729
- {
730
- bool negative = false;
731
- if (*p == '-') {
732
- negative = true;
733
- p++;
734
- }
1091
+ #define MAX_FAST_INTEGER_SIZE 19
1092
+ #define MAX_FAST_UINT64_SIZE 20
1093
+ #define MAX_NUMBER_STACK_BUFFER 128
735
1094
 
736
- long long memo = 0;
737
- while (p < pe) {
738
- memo *= 10;
739
- memo += *p - '0';
740
- p++;
741
- }
1095
+ typedef VALUE (*json_number_decode_func_t)(const char *ptr);
742
1096
 
743
- if (negative) {
744
- memo = -memo;
1097
+ static inline VALUE json_decode_large_number(const char *start, long len, json_number_decode_func_t func)
1098
+ {
1099
+ if (RB_LIKELY(len < MAX_NUMBER_STACK_BUFFER)) {
1100
+ char buffer[MAX_NUMBER_STACK_BUFFER];
1101
+ MEMCPY(buffer, start, char, len);
1102
+ buffer[len] = '\0';
1103
+ return func(buffer);
1104
+ } else {
1105
+ VALUE buffer_v = rb_str_tmp_new(len);
1106
+ char *buffer = RSTRING_PTR(buffer_v);
1107
+ MEMCPY(buffer, start, char, len);
1108
+ buffer[len] = '\0';
1109
+ VALUE number = func(buffer);
1110
+ RB_GC_GUARD(buffer_v);
1111
+ return number;
745
1112
  }
746
- return LL2NUM(memo);
747
1113
  }
748
1114
 
749
- static VALUE json_decode_large_integer(const char *start, long len)
1115
+ static VALUE json_decode_inum(const char *buffer)
1116
+ {
1117
+ return rb_cstr2inum(buffer, 10);
1118
+ }
1119
+
1120
+ NOINLINE(static) VALUE json_decode_large_integer(const char *start, long len)
750
1121
  {
751
- VALUE buffer_v;
752
- char *buffer = RB_ALLOCV_N(char, buffer_v, len + 1);
753
- MEMCPY(buffer, start, char, len);
754
- buffer[len] = '\0';
755
- VALUE number = rb_cstr2inum(buffer, 10);
756
- RB_ALLOCV_END(buffer_v);
757
- return number;
1122
+ return json_decode_large_number(start, len, json_decode_inum);
758
1123
  }
759
1124
 
760
- static inline VALUE
761
- json_decode_integer(const char *start, const char *end)
1125
+ static inline VALUE json_decode_integer(uint64_t mantissa, int mantissa_digits, bool negative, const char *start, const char *end)
762
1126
  {
763
- long len = end - start;
764
- if (RB_LIKELY(len < MAX_FAST_INTEGER_SIZE)) {
765
- return fast_decode_integer(start, end);
1127
+ if (RB_LIKELY(mantissa_digits <= MAX_FAST_INTEGER_SIZE)) {
1128
+ if (RB_LIKELY(!negative)) {
1129
+ return UINT64T2NUM(mantissa);
766
1130
  }
767
- return json_decode_large_integer(start, len);
1131
+
1132
+ // For a negative number 19 digits in length, we only get half of the range,
1133
+ // so ensure this negative number is less than INT64_MAX.
1134
+ //
1135
+ // Note: This does miss INT64_MIN as it's value is one past INT64_MAX
1136
+ // when converted to a uint64_t. It will still be parsed correctly by
1137
+ // falling through to json_decode_large_integer.
1138
+ if (RB_LIKELY(mantissa <= (uint64_t)INT64_MAX)) {
1139
+ return INT64T2NUM(-((int64_t)mantissa));
1140
+ }
1141
+ }
1142
+
1143
+ if (!negative && mantissa_digits == MAX_FAST_UINT64_SIZE) {
1144
+ // Not all 20 digit integers can be safely represented by a uint64_t but
1145
+ // some can. The memcmp with uint64_max is safe as we've rejected leading
1146
+ // zeros and we have guaranteed we're comparing it with a 20 digit number.
1147
+ static const char uint64_max[] = "18446744073709551615";
1148
+ if (memcmp(end - MAX_FAST_UINT64_SIZE, uint64_max, MAX_FAST_UINT64_SIZE) <= 0) {
1149
+ return UINT64T2NUM(mantissa);
1150
+ }
1151
+ }
1152
+
1153
+ return json_decode_large_integer(start, end - start);
768
1154
  }
769
1155
 
770
- static VALUE json_decode_large_float(const char *start, long len)
1156
+ static VALUE json_decode_dnum(const char *buffer)
771
1157
  {
772
- VALUE buffer_v;
773
- char *buffer = RB_ALLOCV_N(char, buffer_v, len + 1);
774
- MEMCPY(buffer, start, char, len);
775
- buffer[len] = '\0';
776
- VALUE number = DBL2NUM(rb_cstr_to_dbl(buffer, 1));
777
- RB_ALLOCV_END(buffer_v);
778
- return number;
1158
+ return DBL2NUM(rb_cstr_to_dbl(buffer, 1));
779
1159
  }
780
1160
 
781
- static VALUE json_decode_float(JSON_ParserConfig *config, const char *start, const char *end)
1161
+ NOINLINE(static) VALUE json_decode_large_float(const char *start, long len)
782
1162
  {
783
- long len = end - start;
1163
+ return json_decode_large_number(start, len, json_decode_dnum);
1164
+ }
784
1165
 
1166
+ /* Ruby JSON optimized float decoder using vendored Ryu algorithm
1167
+ * Accepts pre-extracted mantissa and exponent from first-pass validation
1168
+ */
1169
+ static inline VALUE json_decode_float(JSON_ParserConfig *config, uint64_t mantissa, int mantissa_digits, int64_t exponent, bool negative,
1170
+ const char *start, const char *end)
1171
+ {
785
1172
  if (RB_UNLIKELY(config->decimal_class)) {
786
- VALUE text = rb_str_new(start, len);
1173
+ VALUE text = rb_str_new(start, end - start);
787
1174
  return rb_funcallv(config->decimal_class, config->decimal_method_id, 1, &text);
788
- } else if (RB_LIKELY(len < 64)) {
789
- char buffer[64];
790
- MEMCPY(buffer, start, char, len);
791
- buffer[len] = '\0';
792
- return DBL2NUM(rb_cstr_to_dbl(buffer, 1));
793
- } else {
794
- return json_decode_large_float(start, len);
795
1175
  }
1176
+
1177
+ if (RB_UNLIKELY(exponent > INT32_MAX)) {
1178
+ return negative ? CMinusInfinity : CInfinity;
1179
+ }
1180
+
1181
+ if (RB_UNLIKELY(exponent < INT32_MIN)) {
1182
+ return rb_float_new(negative ? -0.0 : 0.0);
1183
+ }
1184
+
1185
+ if (RB_UNLIKELY(mantissa_digits > 18 || mantissa_digits + exponent < -307)) {
1186
+ // If the value is so small that it definitely underflows to 0.0, return early
1187
+ // to avoid triggering a "Float out of range" warning from rb_cstr_to_dbl.
1188
+ // When mantissa_digits + exponent < -324, value < 10^(-324) < DBL_TRUE_MIN/2,
1189
+ // so it rounds to 0 in IEEE 754 round-to-nearest.
1190
+ if (RB_UNLIKELY(mantissa_digits + exponent < -324)) {
1191
+ return rb_float_new(negative ? -0.0 : 0.0);
1192
+ }
1193
+ return json_decode_large_float(start, end - start);
1194
+ }
1195
+
1196
+ return DBL2NUM(ffp_s2d(exponent, mantissa, negative));
796
1197
  }
797
1198
 
798
1199
  static inline VALUE json_decode_array(JSON_ParserState *state, JSON_ParserConfig *config, long count)
799
1200
  {
800
- VALUE array = rb_ary_new_from_values(count, rvalue_stack_peek(state->stack, count));
801
- rvalue_stack_pop(state->stack, count);
1201
+ VALUE array = rb_ary_new_from_values(count, rvalue_stack_peek(state->value_stack, count));
1202
+ rvalue_stack_pop(state->value_stack, count);
802
1203
 
803
1204
  if (config->freeze) {
804
1205
  RB_OBJ_FREEZE(array);
@@ -807,32 +1208,69 @@ static inline VALUE json_decode_array(JSON_ParserState *state, JSON_ParserConfig
807
1208
  return array;
808
1209
  }
809
1210
 
810
- static inline VALUE json_decode_object(JSON_ParserState *state, JSON_ParserConfig *config, long count)
1211
+ static VALUE json_find_duplicated_key(size_t count, const VALUE *pairs)
811
1212
  {
812
- VALUE object = rb_hash_new_capa(count);
813
- rb_hash_bulk_insert(count, rvalue_stack_peek(state->stack, count), object);
814
-
815
- rvalue_stack_pop(state->stack, count);
1213
+ VALUE set = rb_hash_new_capa(count / 2);
1214
+ for (size_t index = 0; index < count; index += 2) {
1215
+ size_t before = RHASH_SIZE(set);
1216
+ VALUE key = pairs[index];
1217
+ rb_hash_aset(set, key, Qtrue);
1218
+ if (RHASH_SIZE(set) == before) {
1219
+ if (RB_SYMBOL_P(key)) {
1220
+ return rb_sym2str(key);
1221
+ }
1222
+ return key;
1223
+ }
1224
+ }
1225
+ return Qfalse;
1226
+ }
816
1227
 
817
- if (config->freeze) {
818
- RB_OBJ_FREEZE(object);
1228
+ NORETURN(static) void raise_duplicate_key_error(JSON_ParserState *state, VALUE duplicate_key)
1229
+ {
1230
+ VALUE message = rb_sprintf(
1231
+ "duplicate key %"PRIsVALUE,
1232
+ rb_inspect(duplicate_key)
1233
+ );
1234
+
1235
+ rb_str_concat(message, build_parse_error_message("", state));
1236
+ VALUE exc;
1237
+ if (state->parser) { // line and columns can't be accurate in resumable
1238
+ exc = parse_error_new(state, message, 0, 0, false);
1239
+ } else {
1240
+ long line, column;
1241
+ cursor_position(state, &line, &column);
1242
+ rb_str_catf(message, " at line %ld column %ld", line, column);
1243
+ exc = parse_error_new(state, message, line, column, false);
819
1244
  }
1245
+ rb_ivar_set(exc, i_at_json_path, json_path_new(state, duplicate_key));
1246
+ rb_exc_raise(exc);
1247
+ }
820
1248
 
821
- return object;
1249
+ NOINLINE(static) void json_on_duplicate_key(JSON_ParserState *state, JSON_ParserConfig *config, size_t count, const VALUE *pairs)
1250
+ {
1251
+ if (!config->allow_duplicate_key) {
1252
+ raise_duplicate_key_error(state, json_find_duplicated_key(count, pairs));
1253
+ }
822
1254
  }
823
1255
 
824
- static inline VALUE json_decode_string(JSON_ParserState *state, JSON_ParserConfig *config, const char *start, const char *end, bool escaped, bool is_name)
1256
+ static inline VALUE json_decode_object(JSON_ParserState *state, JSON_ParserConfig *config, size_t count)
825
1257
  {
826
- VALUE string;
827
- bool intern = is_name || config->freeze;
828
- bool symbolize = is_name && config->symbolize_names;
829
- if (escaped) {
830
- string = json_string_unescape(state, start, end, is_name, intern, symbolize);
831
- } else {
832
- string = json_string_fastpath(state, start, end, is_name, intern, symbolize);
1258
+ size_t entries_count = count / 2;
1259
+ VALUE object = rb_hash_new_capa(entries_count);
1260
+ const VALUE *pairs = rvalue_stack_peek(state->value_stack, count);
1261
+ rb_hash_bulk_insert(count, pairs, object);
1262
+
1263
+ if (RB_UNLIKELY(RHASH_SIZE(object) < entries_count)) {
1264
+ json_on_duplicate_key(state, config, count, pairs);
833
1265
  }
834
1266
 
835
- return string;
1267
+ rvalue_stack_pop(state->value_stack, count);
1268
+
1269
+ if (config->freeze) {
1270
+ RB_OBJ_FREEZE(object);
1271
+ }
1272
+
1273
+ return object;
836
1274
  }
837
1275
 
838
1276
  static inline VALUE json_push_value(JSON_ParserState *state, JSON_ParserConfig *config, VALUE value)
@@ -840,11 +1278,11 @@ static inline VALUE json_push_value(JSON_ParserState *state, JSON_ParserConfig *
840
1278
  if (RB_UNLIKELY(config->on_load_proc)) {
841
1279
  value = rb_proc_call_with_block(config->on_load_proc, 1, &value, Qnil);
842
1280
  }
843
- rvalue_stack_push(state->stack, value, &state->stack_handle, &state->stack);
1281
+ rvalue_stack_push(state->value_stack, value, state->value_stack_handle, &state->value_stack);
844
1282
  return value;
845
1283
  }
846
1284
 
847
- static const bool string_scan[256] = {
1285
+ static const bool string_scan_table[256] = {
848
1286
  // ASCII Control Characters
849
1287
  1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1,
850
1288
  1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1,
@@ -857,292 +1295,671 @@ static const bool string_scan[256] = {
857
1295
  0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
858
1296
  };
859
1297
 
860
- static inline VALUE json_parse_string(JSON_ParserState *state, JSON_ParserConfig *config, bool is_name)
1298
+ #ifdef HAVE_SIMD
1299
+ static SIMD_Implementation simd_impl = SIMD_NONE;
1300
+ #endif /* HAVE_SIMD */
1301
+
1302
+ ALWAYS_INLINE(static) bool string_scan(JSON_ParserState *state)
861
1303
  {
862
- state->cursor++;
863
- const char *start = state->cursor;
864
- bool escaped = false;
1304
+ #ifdef HAVE_SIMD
1305
+ #if defined(HAVE_SIMD_NEON)
865
1306
 
866
- while (state->cursor < state->end) {
867
- if (RB_UNLIKELY(string_scan[(unsigned char)*state->cursor])) {
868
- switch (*state->cursor) {
869
- case '"': {
870
- VALUE string = json_decode_string(state, config, start, state->cursor, escaped, is_name);
871
- state->cursor++;
872
- return json_push_value(state, config, string);
873
- }
874
- case '\\': {
875
- state->cursor++;
876
- escaped = true;
877
- if ((unsigned char)*state->cursor < 0x20) {
878
- raise_parse_error("invalid ASCII control character in string: %s", state);
879
- }
880
- break;
881
- }
882
- default:
883
- raise_parse_error("invalid ASCII control character in string: %s", state);
884
- break;
885
- }
1307
+ uint64_t mask = 0;
1308
+ if (string_scan_simd_neon(&state->cursor, state->end, &mask)) {
1309
+ state->cursor += trailing_zeros64(mask) >> 2;
1310
+ return true;
1311
+ }
1312
+
1313
+ #elif defined(HAVE_SIMD_SSE2)
1314
+ if (simd_impl == SIMD_SSE2) {
1315
+ int mask = 0;
1316
+ if (string_scan_simd_sse2(&state->cursor, state->end, &mask)) {
1317
+ state->cursor += trailing_zeros(mask);
1318
+ return true;
886
1319
  }
1320
+ }
1321
+ #endif /* HAVE_SIMD_NEON or HAVE_SIMD_SSE2 */
1322
+ #endif /* HAVE_SIMD */
887
1323
 
1324
+ while (!eos(state)) {
1325
+ if (RB_UNLIKELY(string_scan_table[(unsigned char)*state->cursor])) {
1326
+ return true;
1327
+ }
888
1328
  state->cursor++;
889
1329
  }
890
1330
 
891
- raise_parse_error("unexpected end of input, expected closing \"", state);
892
- return Qfalse;
893
- }
894
-
895
- static VALUE json_parse_any(JSON_ParserState *state, JSON_ParserConfig *config)
896
- {
897
- json_eat_whitespace(state);
898
- if (state->cursor >= state->end) {
899
- raise_parse_error("unexpected end of input", state);
1331
+ // If the string ended with an unterminated escape sequence, we might
1332
+ // have gone past the end.
1333
+ if (RB_UNLIKELY(state->cursor > state->end)) {
1334
+ state->cursor = state->end;
900
1335
  }
901
1336
 
902
- switch (*state->cursor) {
903
- case 'n':
904
- if ((state->end - state->cursor >= 4) && (memcmp(state->cursor, "null", 4) == 0)) {
905
- state->cursor += 4;
906
- return json_push_value(state, config, Qnil);
907
- }
908
-
909
- raise_parse_error("unexpected token %s", state);
910
- break;
911
- case 't':
912
- if ((state->end - state->cursor >= 4) && (memcmp(state->cursor, "true", 4) == 0)) {
913
- state->cursor += 4;
914
- return json_push_value(state, config, Qtrue);
915
- }
916
-
917
- raise_parse_error("unexpected token %s", state);
918
- break;
919
- case 'f':
920
- // Note: memcmp with a small power of two compile to an integer comparison
921
- if ((state->end - state->cursor >= 5) && (memcmp(state->cursor + 1, "alse", 4) == 0)) {
922
- state->cursor += 5;
923
- return json_push_value(state, config, Qfalse);
924
- }
1337
+ return false;
1338
+ }
925
1339
 
926
- raise_parse_error("unexpected token %s", state);
927
- break;
928
- case 'N':
929
- // Note: memcmp with a small power of two compile to an integer comparison
930
- if (config->allow_nan && (state->end - state->cursor >= 3) && (memcmp(state->cursor + 1, "aN", 2) == 0)) {
931
- state->cursor += 3;
932
- return json_push_value(state, config, CNaN);
933
- }
1340
+ static VALUE json_parse_escaped_string(JSON_ParserState *state, JSON_ParserConfig *config, bool is_name, const char *start)
1341
+ {
1342
+ const char *backslashes[JSON_MAX_UNESCAPE_POSITIONS];
1343
+ JSON_UnescapePositions positions = {
1344
+ .size = 0,
1345
+ .positions = backslashes,
1346
+ .additional_backslashes = 0,
1347
+ };
934
1348
 
935
- raise_parse_error("unexpected token %s", state);
936
- break;
937
- case 'I':
938
- if (config->allow_nan && (state->end - state->cursor >= 8) && (memcmp(state->cursor, "Infinity", 8) == 0)) {
939
- state->cursor += 8;
940
- return json_push_value(state, config, CInfinity);
1349
+ do {
1350
+ switch (*state->cursor) {
1351
+ case '"': {
1352
+ VALUE string = json_string_unescape(state, config, start, state->cursor, is_name, &positions);
1353
+ state->cursor++;
1354
+ return string;
941
1355
  }
942
-
943
- raise_parse_error("unexpected token %s", state);
944
- break;
945
- case '-':
946
- // Note: memcmp with a small power of two compile to an integer comparison
947
- if ((state->end - state->cursor >= 9) && (memcmp(state->cursor + 1, "Infinity", 8) == 0)) {
948
- if (config->allow_nan) {
949
- state->cursor += 9;
950
- return json_push_value(state, config, CMinusInfinity);
1356
+ case '\\': {
1357
+ if (RB_LIKELY(positions.size < JSON_MAX_UNESCAPE_POSITIONS)) {
1358
+ backslashes[positions.size] = state->cursor;
1359
+ positions.size++;
951
1360
  } else {
952
- raise_parse_error("unexpected token %s", state);
1361
+ positions.additional_backslashes++;
953
1362
  }
954
- }
955
- // Fallthrough
956
- case '0': case '1': case '2': case '3': case '4': case '5': case '6': case '7': case '8': case '9': {
957
- bool integer = true;
958
-
959
- // /\A-?(0|[1-9]\d*)(\.\d+)?([Ee][-+]?\d+)?/
960
- const char *start = state->cursor;
961
- state->cursor++;
962
-
963
- while ((state->cursor < state->end) && (*state->cursor >= '0') && (*state->cursor <= '9')) {
964
1363
  state->cursor++;
1364
+ break;
965
1365
  }
1366
+ default:
1367
+ if (!config->allow_control_characters) {
1368
+ raise_syntax_error("invalid ASCII control character in string: %s", state);
1369
+ }
1370
+ break;
1371
+ }
966
1372
 
967
- long integer_length = state->cursor - start;
1373
+ state->cursor++;
1374
+ } while (string_scan(state));
968
1375
 
969
- if (RB_UNLIKELY(start[0] == '0' && integer_length > 1)) {
970
- raise_parse_error_at("invalid number: %s", state, start);
971
- } else if (RB_UNLIKELY(integer_length > 2 && start[0] == '-' && start[1] == '0')) {
972
- raise_parse_error_at("invalid number: %s", state, start);
973
- } else if (RB_UNLIKELY(integer_length == 1 && start[0] == '-')) {
974
- raise_parse_error_at("invalid number: %s", state, start);
975
- }
1376
+ return Qundef;
1377
+ }
976
1378
 
977
- if ((state->cursor < state->end) && (*state->cursor == '.')) {
978
- integer = false;
979
- state->cursor++;
1379
+ ALWAYS_INLINE(static) VALUE json_parse_string(JSON_ParserState *state, JSON_ParserConfig *config, bool is_name)
1380
+ {
1381
+ state->cursor++;
1382
+ const char *start = state->cursor;
980
1383
 
981
- if (state->cursor == state->end || *state->cursor < '0' || *state->cursor > '9') {
982
- raise_parse_error("invalid number: %s", state);
983
- }
1384
+ if (RB_UNLIKELY(!string_scan(state))) {
1385
+ return Qundef;
1386
+ }
984
1387
 
985
- while ((state->cursor < state->end) && (*state->cursor >= '0') && (*state->cursor <= '9')) {
986
- state->cursor++;
987
- }
988
- }
1388
+ VALUE string;
1389
+ if (RB_LIKELY(*state->cursor == '"')) {
1390
+ string = json_string_fastpath(state, config, start, state->cursor, is_name);
1391
+ state->cursor++;
1392
+ }
1393
+ else {
1394
+ string = json_parse_escaped_string(state, config, is_name, start);
1395
+ }
989
1396
 
990
- if ((state->cursor < state->end) && ((*state->cursor == 'e') || (*state->cursor == 'E'))) {
991
- integer = false;
992
- state->cursor++;
993
- if ((state->cursor < state->end) && ((*state->cursor == '+') || (*state->cursor == '-'))) {
994
- state->cursor++;
995
- }
1397
+ return string;
1398
+ }
996
1399
 
997
- if (state->cursor == state->end || *state->cursor < '0' || *state->cursor > '9') {
998
- raise_parse_error("invalid number: %s", state);
999
- }
1400
+ #if JSON_CPU_LITTLE_ENDIAN_64BITS
1401
+ // From: https://lemire.me/blog/2022/01/21/swar-explained-parsing-eight-digits/
1402
+ // Additional References:
1403
+ // https://johnnylee-sde.github.io/Fast-numeric-string-to-int/
1404
+ // http://0x80.pl/notesen/2014-10-12-parsing-decimal-numbers-part-1-swar.html
1405
+ static inline uint64_t decode_8digits_unrolled(uint64_t val) {
1406
+ const uint64_t mask = 0x000000FF000000FF;
1407
+ const uint64_t mul1 = 0x000F424000000064; // 100 + (1000000ULL << 32)
1408
+ const uint64_t mul2 = 0x0000271000000001; // 1 + (10000ULL << 32)
1409
+ val -= 0x3030303030303030;
1410
+ val = (val * 10) + (val >> 8); // val = (val * 2561) >> 8;
1411
+ val = (((val & mask) * mul1) + (((val >> 16) & mask) * mul2)) >> 32;
1412
+ return val;
1413
+ }
1000
1414
 
1001
- while ((state->cursor < state->end) && (*state->cursor >= '0') && (*state->cursor <= '9')) {
1002
- state->cursor++;
1003
- }
1004
- }
1415
+ static inline uint64_t decode_4digits_unrolled(uint32_t val) {
1416
+ const uint32_t mask = 0x000000FF;
1417
+ const uint32_t mul1 = 100;
1418
+ val -= 0x30303030;
1419
+ val = (val * 10) + (val >> 8); // val = (val * 2561) >> 8;
1420
+ val = ((val & mask) * mul1) + (((val >> 16) & mask));
1421
+ return val;
1422
+ }
1423
+ #endif
1005
1424
 
1006
- if (integer) {
1007
- return json_push_value(state, config, json_decode_integer(start, state->cursor));
1008
- }
1009
- return json_push_value(state, config, json_decode_float(config, start, state->cursor));
1010
- }
1011
- case '"': {
1012
- // %r{\A"[^"\\\t\n\x00]*(?:\\[bfnrtu\\/"][^"\\]*)*"}
1013
- return json_parse_string(state, config, false);
1014
- break;
1015
- }
1016
- case '[': {
1017
- state->cursor++;
1018
- json_eat_whitespace(state);
1019
- long stack_head = state->stack->head;
1425
+ static inline int json_parse_digits(JSON_ParserState *state, uint64_t *accumulator)
1426
+ {
1427
+ const char *start = state->cursor;
1020
1428
 
1021
- if ((state->cursor < state->end) && (*state->cursor == ']')) {
1022
- state->cursor++;
1023
- return json_push_value(state, config, json_decode_array(state, config, 0));
1024
- } else {
1025
- state->current_nesting++;
1026
- if (RB_UNLIKELY(config->max_nesting && (config->max_nesting < state->current_nesting))) {
1027
- rb_raise(eNestingError, "nesting of %d is too deep", state->current_nesting);
1028
- }
1029
- state->in_array++;
1030
- json_parse_any(state, config);
1031
- }
1429
+ #if JSON_CPU_LITTLE_ENDIAN_64BITS
1430
+ while (rest(state) >= sizeof(uint64_t)) {
1431
+ uint64_t next_8bytes;
1432
+ memcpy(&next_8bytes, state->cursor, sizeof(uint64_t));
1032
1433
 
1033
- while (true) {
1034
- json_eat_whitespace(state);
1035
-
1036
- if (state->cursor < state->end) {
1037
- if (*state->cursor == ']') {
1038
- state->cursor++;
1039
- long count = state->stack->head - stack_head;
1040
- state->current_nesting--;
1041
- state->in_array--;
1042
- return json_push_value(state, config, json_decode_array(state, config, count));
1043
- }
1434
+ // From: https://github.com/simdjson/simdjson/blob/32b301893c13d058095a07d9868edaaa42ee07aa/include/simdjson/generic/numberparsing.h#L333
1435
+ // Branchless version of: http://0x80.pl/articles/swar-digits-validate.html
1436
+ uint64_t match = (next_8bytes & 0xF0F0F0F0F0F0F0F0) | (((next_8bytes + 0x0606060606060606) & 0xF0F0F0F0F0F0F0F0) >> 4);
1044
1437
 
1045
- if (*state->cursor == ',') {
1046
- state->cursor++;
1047
- if (config->allow_trailing_comma) {
1048
- json_eat_whitespace(state);
1049
- if ((state->cursor < state->end) && (*state->cursor == ']')) {
1050
- continue;
1051
- }
1052
- }
1053
- json_parse_any(state, config);
1054
- continue;
1055
- }
1056
- }
1438
+ if (match == 0x3333333333333333) { // 8 consecutive digits
1439
+ *accumulator = (*accumulator * 100000000) + decode_8digits_unrolled(next_8bytes);
1440
+ state->cursor += 8;
1441
+ continue;
1442
+ }
1057
1443
 
1058
- raise_parse_error("expected ',' or ']' after array value", state);
1059
- }
1060
- break;
1444
+ uint32_t consecutive_digits = trailing_zeros64(match ^ 0x3333333333333333) / CHAR_BIT;
1445
+
1446
+ if (consecutive_digits >= 4) {
1447
+ *accumulator = (*accumulator * 10000) + decode_4digits_unrolled((uint32_t)next_8bytes);
1448
+ state->cursor += 4;
1449
+ consecutive_digits -= 4;
1061
1450
  }
1062
- case '{': {
1451
+
1452
+ while (consecutive_digits) {
1453
+ *accumulator = *accumulator * 10 + (*state->cursor - '0');
1454
+ consecutive_digits--;
1063
1455
  state->cursor++;
1064
- json_eat_whitespace(state);
1065
- long stack_head = state->stack->head;
1456
+ }
1066
1457
 
1067
- if ((state->cursor < state->end) && (*state->cursor == '}')) {
1068
- state->cursor++;
1069
- return json_push_value(state, config, json_decode_object(state, config, 0));
1070
- } else {
1071
- state->current_nesting++;
1072
- if (RB_UNLIKELY(config->max_nesting && (config->max_nesting < state->current_nesting))) {
1073
- rb_raise(eNestingError, "nesting of %d is too deep", state->current_nesting);
1074
- }
1458
+ return (int)(state->cursor - start);
1459
+ }
1460
+ #endif
1075
1461
 
1076
- if (*state->cursor != '"') {
1077
- raise_parse_error("expected object key, got %s", state);
1078
- }
1079
- json_parse_string(state, config, true);
1462
+ char next_char;
1463
+ while (rb_isdigit(next_char = peek(state))) {
1464
+ *accumulator = *accumulator * 10 + (next_char - '0');
1465
+ state->cursor++;
1466
+ }
1467
+ return (int)(state->cursor - start);
1468
+ }
1080
1469
 
1081
- json_eat_whitespace(state);
1082
- if ((state->cursor >= state->end) || (*state->cursor != ':')) {
1083
- raise_parse_error("expected ':' after object key", state);
1084
- }
1085
- state->cursor++;
1470
+ static inline VALUE json_parse_number(JSON_ParserState *state, JSON_ParserConfig *config, bool negative, const char *start, bool resumable)
1471
+ {
1472
+ bool integer = true;
1473
+ const char first_digit = *state->cursor;
1086
1474
 
1087
- json_parse_any(state, config);
1088
- }
1475
+ // Variables for Ryu optimization - extract digits during parsing
1476
+ int64_t exponent = 0;
1477
+ int decimal_point_pos = -1;
1478
+ uint64_t mantissa = 0;
1089
1479
 
1090
- while (true) {
1091
- json_eat_whitespace(state);
1092
-
1093
- if (state->cursor < state->end) {
1094
- if (*state->cursor == '}') {
1095
- state->cursor++;
1096
- state->current_nesting--;
1097
- long count = state->stack->head - stack_head;
1098
- return json_push_value(state, config, json_decode_object(state, config, count));
1099
- }
1480
+ // Parse integer part and extract mantissa digits
1481
+ int mantissa_digits = json_parse_digits(state, &mantissa);
1100
1482
 
1101
- if (*state->cursor == ',') {
1102
- state->cursor++;
1103
- json_eat_whitespace(state);
1483
+ if (RB_UNLIKELY((first_digit == '0' && mantissa_digits > 1) || (negative && mantissa_digits == 0))) {
1484
+ return Qundef;
1485
+ }
1104
1486
 
1105
- if (config->allow_trailing_comma) {
1106
- if ((state->cursor < state->end) && (*state->cursor == '}')) {
1107
- continue;
1108
- }
1109
- }
1487
+ // Parse fractional part
1488
+ if (peek(state) == '.') {
1489
+ integer = false;
1490
+ decimal_point_pos = mantissa_digits; // Remember position of decimal point
1491
+ state->cursor++;
1110
1492
 
1111
- if (*state->cursor != '"') {
1112
- raise_parse_error("expected object key, got: %s", state);
1113
- }
1114
- json_parse_string(state, config, true);
1493
+ int fractional_digits = json_parse_digits(state, &mantissa);
1494
+ mantissa_digits += fractional_digits;
1115
1495
 
1116
- json_eat_whitespace(state);
1117
- if ((state->cursor >= state->end) || (*state->cursor != ':')) {
1118
- raise_parse_error("expected ':' after object key, got: %s", state);
1119
- }
1120
- state->cursor++;
1496
+ if (RB_UNLIKELY(!fractional_digits)) {
1497
+ return Qundef;
1498
+ }
1499
+ }
1121
1500
 
1122
- json_parse_any(state, config);
1501
+ // Parse exponent
1502
+ if (rb_tolower(peek(state)) == 'e') {
1503
+ integer = false;
1504
+ state->cursor++;
1123
1505
 
1124
- continue;
1506
+ bool negative_exponent = false;
1507
+ const char next_char = peek(state);
1508
+ if (next_char == '-' || next_char == '+') {
1509
+ negative_exponent = next_char == '-';
1510
+ state->cursor++;
1511
+ }
1512
+
1513
+ uint64_t abs_exponent = 0;
1514
+ int exponent_digits = json_parse_digits(state, &abs_exponent);
1515
+
1516
+ if (RB_UNLIKELY(!exponent_digits)) {
1517
+ return Qundef;
1518
+ }
1519
+
1520
+ if (RB_UNLIKELY(exponent_digits >= 20 || abs_exponent > (uint64_t)INT64_MAX)) {
1521
+ exponent = negative_exponent ? INT64_MIN : INT64_MAX;
1522
+ } else {
1523
+ exponent = negative_exponent ? -(int64_t)abs_exponent : (int64_t)abs_exponent;
1524
+ }
1525
+ }
1526
+
1527
+ // A number touching the end of the buffer may still grow in a later chunk,
1528
+ // so the caller will rewind and wait. Decoding it now would build a value
1529
+ // -- for a long run of digits, an expensive bignum -- only to discard it,
1530
+ // and repeating that on every resumed chunk is quadratic in the number's
1531
+ // length. The digit scan above already advanced the cursor, which is all
1532
+ // the caller needs to detect the incomplete number.
1533
+ if (RB_UNLIKELY(resumable && eos(state))) {
1534
+ return Qundef;
1535
+ }
1536
+
1537
+ if (integer) {
1538
+ return json_decode_integer(mantissa, mantissa_digits, negative, start, state->cursor);
1539
+ }
1540
+
1541
+ // Adjust exponent based on decimal point position
1542
+ if (decimal_point_pos >= 0) {
1543
+ exponent -= (mantissa_digits - decimal_point_pos);
1544
+ }
1545
+
1546
+ return json_decode_float(config, mantissa, mantissa_digits, exponent, negative, start, state->cursor);
1547
+ }
1548
+
1549
+ // How many values (array elements, or interleaved object keys+values) have been
1550
+ // pushed onto the rvalue stack since this container opened. Used to size the
1551
+ // bulk decode on close, and to tell the first key/colon from later ones.
1552
+ static inline long json_frame_entry_count(const json_frame *frame, const rvalue_stack *value_stack)
1553
+ {
1554
+ return value_stack->head - frame->value_stack_head;
1555
+ }
1556
+
1557
+ // A complete value now sits on top of the rvalue stack. Advance the frame that
1558
+ // was waiting for it: the root document is done, or the enclosing container
1559
+ // moves on to expecting a ',' or its closing bracket. The caller passes the
1560
+ // frame it already has in hand -- the one that was expecting the value -- which
1561
+ // after a container close is the freshly re-exposed parent.
1562
+ static inline enum json_frame_phase json_value_completed(json_frame *frame)
1563
+ {
1564
+ JSON_ASSERT((int)JSON_PHASE_DONE == (int)JSON_FRAME_ROOT);
1565
+ JSON_ASSERT((int)JSON_PHASE_ARRAY_COMMA == (int)JSON_FRAME_ARRAY);
1566
+ JSON_ASSERT((int)JSON_PHASE_OBJECT_COMMA == (int)JSON_FRAME_OBJECT);
1567
+
1568
+ return frame->phase = (enum json_frame_phase) frame->type;
1569
+ }
1570
+
1571
+ ALWAYS_INLINE(static) void json_match_keyword(JSON_ParserState *state, const char *keyword, size_t offset)
1572
+ {
1573
+ // It is assumed that since `keyword` is always a literal, the compiler is able to constantize this
1574
+ // `strlen` and several other computations in that routine.
1575
+
1576
+ size_t len = strlen(keyword);
1577
+
1578
+ // Note: memcmp with a small power of two and a literal string compile to an integer comparison /
1579
+ // That's why we sometime compare starting from the first byte and sometimes from the second.
1580
+ if (rest(state) >= len && (memcmp(state->cursor + offset, keyword + offset, len - offset) == 0)) {
1581
+ state->cursor += len;
1582
+ return;
1583
+ }
1584
+
1585
+ bool eos = rest(state) < len && memcmp(state->cursor, keyword, rest(state)) == 0;
1586
+ raise_parse_error("unexpected token %s", state, eos);
1587
+ }
1588
+
1589
+ // Parse an arbitrary JSON value iteratively. This is a state machine driven
1590
+ // entirely by the top frame's phase so it can stop at any value boundary and
1591
+ // resume purely from the frame stack. A JSON_FRAME_ROOT frame sits at the
1592
+ // bottom of the stack, so the stack is never empty mid-parse and the document
1593
+ // itself is just another frame whose value, once parsed, leaves its phase DONE.
1594
+ // When invoked in resumable mode, it returns true after parsing a complete document.
1595
+ // If reaching EOS without having parsed a complete document, either returns false
1596
+ // of raise a JSON::ParserError tagged with `@eos=true`.
1597
+ ALWAYS_INLINE(static) bool json_parse_any(JSON_ParserState *state, JSON_ParserConfig *config, bool resumable)
1598
+ {
1599
+ json_frame *frame = json_frame_stack_peek(state->frames);
1600
+
1601
+ switch (frame->phase) {
1602
+ case JSON_PHASE_DONE: JSON_UNREACHABLE_RETURN(false);
1603
+ case JSON_PHASE_ARRAY_COMMA: goto JSON_PHASE_ARRAY_COMMA;
1604
+ case JSON_PHASE_OBJECT_COMMA: goto JSON_PHASE_OBJECT_COMMA;
1605
+ case JSON_PHASE_VALUE: goto JSON_PHASE_VALUE;
1606
+ case JSON_PHASE_OBJECT_KEY: goto JSON_PHASE_OBJECT_KEY;
1607
+ case JSON_PHASE_OBJECT_COLON: goto JSON_PHASE_OBJECT_COLON;
1608
+ }
1609
+ JSON_UNREACHABLE_RETURN(false);
1610
+
1611
+ JSON_PHASE_VALUE: {
1612
+ json_eat_whitespace(state, config, true);
1613
+
1614
+ // A trailing comma lands us here expecting an element but finding the
1615
+ // closing bracket; hand off to ARRAY_COMMA to close. An empty array
1616
+ // closes inline at '[', so this position is only reached after a ','.
1617
+ if (config->allow_trailing_comma && frame->type == JSON_FRAME_ARRAY && peek(state) == ']') {
1618
+ goto JSON_PHASE_ARRAY_COMMA;
1619
+ }
1620
+
1621
+ VALUE value;
1622
+ const char *value_start = state->cursor;
1623
+
1624
+ switch (peek(state)) {
1625
+ case 'n':
1626
+ json_match_keyword(state, "null", 0);
1627
+ value = Qnil;
1628
+ break;
1629
+
1630
+ case 't':
1631
+ json_match_keyword(state, "true", 0);
1632
+ value = Qtrue;
1633
+ break;
1634
+
1635
+ case 'f':
1636
+ json_match_keyword(state, "false", 1);
1637
+ value = Qfalse;
1638
+ break;
1639
+
1640
+ case 'N':
1641
+ if (!config->allow_nan) {
1642
+ raise_syntax_error("unexpected token %s", state);
1643
+ }
1644
+
1645
+ json_match_keyword(state, "NaN", 1);
1646
+ value = CNaN;
1647
+ break;
1648
+
1649
+ case 'I':
1650
+ if (!config->allow_nan) {
1651
+ raise_syntax_error("unexpected token %s", state);
1652
+ }
1653
+
1654
+ json_match_keyword(state, "Infinity", 0);
1655
+ value = CInfinity;
1656
+ break;
1657
+
1658
+ case '-': {
1659
+ state->cursor++;
1660
+
1661
+ value = json_parse_number(state, config, true, value_start, resumable);
1662
+
1663
+ if (RB_UNLIKELY(UNDEF_P(value) && config->allow_nan && peek(state) == 'I')) {
1664
+ state->cursor = value_start;
1665
+ json_match_keyword(state, "-Infinity", 1);
1666
+ value = CMinusInfinity;
1667
+ break;
1668
+ }
1669
+
1670
+ // Top level numbers are ambiguous when parsing streams, we can't
1671
+ // know if we parsed all the digits if we hit EOS.
1672
+ if (RB_UNLIKELY(resumable && eos(state))) {
1673
+ state->cursor = value_start;
1674
+ return false;
1675
+ }
1676
+
1677
+ if (RB_UNLIKELY(UNDEF_P(value))) {
1678
+ raise_syntax_error_at("invalid number: %s", state, value_start);
1679
+ }
1680
+ break;
1681
+ }
1682
+
1683
+ case '0': case '1': case '2': case '3': case '4': case '5': case '6': case '7': case '8': case '9': {
1684
+ value = json_parse_number(state, config, false, value_start, resumable);
1685
+
1686
+ // Top level numbers are ambiguous when parsing streams, we can't
1687
+ // know if we parsed all the digits if we hit EOS.
1688
+ if (RB_UNLIKELY(resumable && eos(state))) {
1689
+ state->cursor = value_start;
1690
+ return false;
1691
+ }
1692
+
1693
+ if (RB_UNLIKELY(UNDEF_P(value))) {
1694
+ raise_syntax_error_at("invalid number: %s", state, value_start);
1695
+ }
1696
+ break;
1697
+ }
1698
+
1699
+ case '"': {
1700
+ // %r{\A"[^"\\\t\n\x00]*(?:\\[bfnrtu\\/"][^"\\]*)*"}
1701
+ value = json_parse_string(state, config, false);
1702
+
1703
+ if (RB_UNLIKELY(UNDEF_P(value))) {
1704
+ bool is_eos = eos(state);
1705
+ if (resumable && is_eos) {
1706
+ state->cursor = value_start;
1707
+ return false;
1125
1708
  }
1709
+ raise_parse_error("unexpected end of input, expected closing \"", state, is_eos);
1710
+ }
1711
+ break;
1712
+ }
1713
+
1714
+ case '[': {
1715
+ state->cursor++;
1716
+ // The '[' is consumed but its frame is only pushed below, so a
1717
+ // comment suspending here must resume from the bracket.
1718
+ json_eat_whitespace_resume_at(state, config, true, value_start);
1719
+
1720
+ const char next = peek(state);
1721
+ if (next == ']') {
1722
+ state->cursor++;
1723
+ value = json_decode_array(state, config, 0);
1724
+ break;
1725
+ } else if (resumable && eos(state)) {
1726
+ state->cursor = value_start;
1727
+ return false;
1728
+ }
1729
+
1730
+ state->current_nesting++;
1731
+ if (RB_UNLIKELY(config->max_nesting && (config->max_nesting < state->current_nesting))) {
1732
+ rb_raise(eNestingError, "nesting of %d is too deep", state->current_nesting);
1126
1733
  }
1734
+ state->in_array++;
1127
1735
 
1128
- raise_parse_error("expected ',' or '}' after object value, got: %s", state);
1736
+ // Phase stays VALUE: the next iteration reads the first element.
1737
+ frame = json_frame_stack_push(state, (json_frame){
1738
+ .type = JSON_FRAME_ARRAY,
1739
+ .phase = JSON_PHASE_VALUE,
1740
+ .value_stack_head = state->value_stack->head,
1741
+ });
1742
+ goto JSON_PHASE_VALUE;
1129
1743
  }
1130
- break;
1744
+
1745
+ case '{': {
1746
+ state->cursor++;
1747
+ // Same as '[': the frame is only pushed below.
1748
+ json_eat_whitespace_resume_at(state, config, true, value_start);
1749
+
1750
+ if (peek(state) == '}') {
1751
+ state->cursor++;
1752
+ value = json_decode_object(state, config, 0);
1753
+ break;
1754
+ } else if (resumable && eos(state)) {
1755
+ state->cursor = value_start;
1756
+ return false;
1757
+ }
1758
+
1759
+ state->current_nesting++;
1760
+ if (RB_UNLIKELY(config->max_nesting && (config->max_nesting < state->current_nesting))) {
1761
+ rb_raise(eNestingError, "nesting of %d is too deep", state->current_nesting);
1762
+ }
1763
+
1764
+ // Phase KEY: the next iteration reads the first key.
1765
+ frame = json_frame_stack_push(state, (json_frame){
1766
+ .type = JSON_FRAME_OBJECT,
1767
+ .phase = JSON_PHASE_OBJECT_KEY,
1768
+ .value_stack_head = state->value_stack->head,
1769
+ .start_offset = value_start - state->start,
1770
+ });
1771
+ goto JSON_PHASE_OBJECT_KEY;
1772
+ }
1773
+
1774
+ case 0:
1775
+ // peek() returns 0 both at end-of-stream and for a literal NUL byte in the
1776
+ // buffer. Only a genuine EOS means "feed me more"; a NUL byte that is not at
1777
+ // EOS is just an invalid character.
1778
+ if (eos(state)) {
1779
+ return false;
1780
+ } else {
1781
+ raise_syntax_error("unexpected NULL byte: %s", state);
1782
+ }
1783
+ default:
1784
+ raise_syntax_error("unexpected character: %s", state);
1131
1785
  }
1132
1786
 
1133
- default:
1134
- raise_parse_error("unexpected character: %s", state);
1135
- break;
1787
+ json_push_value(state, config, value);
1788
+ json_value_completed(frame);
1789
+
1790
+ switch (frame->phase) {
1791
+ case JSON_PHASE_DONE: return true;
1792
+ case JSON_PHASE_ARRAY_COMMA: goto JSON_PHASE_ARRAY_COMMA;
1793
+ case JSON_PHASE_OBJECT_COMMA: goto JSON_PHASE_OBJECT_COMMA;
1794
+ case JSON_PHASE_VALUE: goto JSON_PHASE_VALUE;
1795
+ case JSON_PHASE_OBJECT_KEY: JSON_UNREACHABLE_RETURN(false);
1796
+ case JSON_PHASE_OBJECT_COLON: goto JSON_PHASE_OBJECT_COLON;
1797
+ }
1798
+ JSON_UNREACHABLE_RETURN(false);
1799
+ }
1800
+
1801
+ JSON_PHASE_OBJECT_KEY: {
1802
+ JSON_ASSERT(frame->type == JSON_FRAME_OBJECT);
1803
+
1804
+ json_eat_whitespace(state, config, true);
1805
+
1806
+ // A trailing comma lands us here expecting a key but finding the closing
1807
+ // brace; hand off to OBJECT_COMMA to close. An empty object closes inline
1808
+ // at '{', so this position is only reached after a ','.
1809
+ if (config->allow_trailing_comma && peek(state) == '}') {
1810
+ goto JSON_PHASE_OBJECT_COMMA;
1811
+ }
1812
+
1813
+ const char *start = state->cursor;
1814
+
1815
+ if (RB_LIKELY(peek(state) == '"')) {
1816
+ VALUE string = json_parse_string(state, config, true);
1817
+ if (UNDEF_P(string)) {
1818
+ if (resumable) {
1819
+ state->cursor = start;
1820
+ return false;
1821
+ } else {
1822
+ raise_syntax_error("unexpected end of input, expected closing \"", state);
1823
+ }
1824
+ }
1825
+ json_push_value(state, config, string);
1826
+ frame->phase = JSON_PHASE_OBJECT_COLON;
1827
+ goto JSON_PHASE_OBJECT_COLON;
1828
+ } else if (resumable && eos(state)) {
1829
+ return false;
1830
+ } else {
1831
+ // The message differs for the first key vs. a key after a
1832
+ // ',': the first is the only one reached with nothing pushed
1833
+ // for this object yet.
1834
+ if (json_frame_entry_count(frame, state->value_stack) == 0) {
1835
+ raise_syntax_error("expected object key, got %s", state);
1836
+ } else {
1837
+ raise_syntax_error("expected object key, got: %s", state);
1838
+ }
1839
+ }
1840
+ JSON_UNREACHABLE_RETURN(false);
1136
1841
  }
1137
1842
 
1138
- raise_parse_error("unreacheable: %s", state);
1843
+ JSON_PHASE_OBJECT_COLON: {
1844
+ JSON_ASSERT(frame->type == JSON_FRAME_OBJECT);
1845
+
1846
+ json_eat_whitespace(state, config, true);
1847
+
1848
+ if (RB_LIKELY(peek(state) == ':')) {
1849
+ state->cursor++;
1850
+ frame->phase = JSON_PHASE_VALUE;
1851
+ goto JSON_PHASE_VALUE;
1852
+ } else if (resumable && eos(state)) {
1853
+ return false;
1854
+ } else {
1855
+ // First colon (only the first pair's key is pushed, nothing
1856
+ // else) vs. a later one.
1857
+ if (json_frame_entry_count(frame, state->value_stack) == 1) {
1858
+ raise_syntax_error("expected ':' after object key", state);
1859
+ } else {
1860
+ raise_syntax_error("expected ':' after object key, got: %s", state);
1861
+ }
1862
+ }
1863
+ JSON_UNREACHABLE_RETURN(false);
1864
+ }
1865
+
1866
+ JSON_PHASE_ARRAY_COMMA: {
1867
+ JSON_ASSERT(frame->type == JSON_FRAME_ARRAY);
1868
+
1869
+ json_eat_whitespace(state, config, true);
1870
+
1871
+ const char next_char = peek(state);
1872
+
1873
+ if (RB_LIKELY(next_char == ',')) {
1874
+ state->cursor++;
1875
+ // Commit the phase before eating the whitespace that follows: an
1876
+ // incomplete comment there would suspend the parse, and a phase not
1877
+ // yet advanced past the ',' would drop it on resume. A trailing comma
1878
+ // is recognized in JSON_PHASE_VALUE once the ']' is in the buffer.
1879
+ frame->phase = JSON_PHASE_VALUE;
1880
+ goto JSON_PHASE_VALUE;
1881
+ } else if (next_char == ']') {
1882
+ state->cursor++;
1883
+ long count = json_frame_entry_count(frame, state->value_stack);
1884
+ state->current_nesting--;
1885
+ state->in_array--;
1886
+
1887
+ json_push_value(state, config, json_decode_array(state, config, count));
1888
+ json_frame_stack_pop(state->frames);
1889
+ frame = json_frame_stack_peek(state->frames);
1890
+
1891
+ json_value_completed(frame);
1892
+
1893
+ switch (frame->phase) {
1894
+ case JSON_PHASE_DONE: return true;
1895
+ case JSON_PHASE_ARRAY_COMMA: goto JSON_PHASE_ARRAY_COMMA;
1896
+ case JSON_PHASE_OBJECT_COMMA: goto JSON_PHASE_OBJECT_COMMA;
1897
+ case JSON_PHASE_VALUE: goto JSON_PHASE_VALUE;
1898
+ case JSON_PHASE_OBJECT_KEY: JSON_UNREACHABLE_RETURN(false);
1899
+ case JSON_PHASE_OBJECT_COLON: goto JSON_PHASE_OBJECT_COLON;
1900
+ }
1901
+ } else if (resumable && eos(state)) {
1902
+ return false;
1903
+ } else {
1904
+ raise_syntax_error("expected ',' or ']' after array value", state);
1905
+ }
1906
+ JSON_UNREACHABLE_RETURN(false);
1907
+ }
1908
+
1909
+ JSON_PHASE_OBJECT_COMMA: {
1910
+ JSON_ASSERT(frame->type == JSON_FRAME_OBJECT);
1911
+
1912
+ json_eat_whitespace(state, config, true);
1913
+ const char next_char = peek(state);
1914
+
1915
+ if (RB_LIKELY(next_char == ',')) {
1916
+ state->cursor++;
1917
+ // Commit the phase before eating the whitespace that follows: an
1918
+ // incomplete comment there would suspend the parse, and a phase not
1919
+ // yet advanced past the ',' would drop it on resume. A trailing comma
1920
+ // is recognized in JSON_PHASE_OBJECT_KEY once the '}' is in the buffer.
1921
+ frame->phase = JSON_PHASE_OBJECT_KEY;
1922
+ goto JSON_PHASE_OBJECT_KEY;
1923
+ } else if (next_char == '}') {
1924
+ state->cursor++;
1925
+ state->current_nesting--;
1926
+ size_t count = json_frame_entry_count(frame, state->value_stack);
1927
+
1928
+ // Temporary rewind cursor in case an error is raised
1929
+ const char *final_cursor = state->cursor;
1930
+ state->cursor = state->start + frame->start_offset;
1931
+ VALUE object = json_decode_object(state, config, count);
1932
+ state->cursor = final_cursor;
1933
+
1934
+ json_push_value(state, config, object);
1935
+ json_frame_stack_pop(state->frames);
1936
+ frame = json_frame_stack_peek(state->frames);
1937
+ json_value_completed(frame);
1938
+
1939
+ switch (frame->phase) {
1940
+ case JSON_PHASE_DONE: return true;
1941
+ case JSON_PHASE_ARRAY_COMMA: goto JSON_PHASE_ARRAY_COMMA;
1942
+ case JSON_PHASE_OBJECT_COMMA: goto JSON_PHASE_OBJECT_COMMA;
1943
+ case JSON_PHASE_VALUE: goto JSON_PHASE_VALUE;
1944
+ case JSON_PHASE_OBJECT_KEY: JSON_UNREACHABLE_RETURN(false);
1945
+ case JSON_PHASE_OBJECT_COLON: goto JSON_PHASE_OBJECT_COLON;
1946
+ }
1947
+ } else if (resumable && eos(state)) {
1948
+ return false;
1949
+ } else {
1950
+ raise_syntax_error("expected ',' or '}' after object value, got: %s", state);
1951
+ }
1952
+ JSON_UNREACHABLE_RETURN(false);
1953
+ }
1954
+
1955
+ JSON_UNREACHABLE_RETURN(false);
1139
1956
  }
1140
1957
 
1141
- static void json_ensure_eof(JSON_ParserState *state)
1958
+ static void json_ensure_eof(JSON_ParserState *state, JSON_ParserConfig *config)
1142
1959
  {
1143
- json_eat_whitespace(state);
1144
- if (state->cursor != state->end) {
1145
- raise_parse_error("unexpected token at end of stream %s", state);
1960
+ json_eat_whitespace(state, config, true);
1961
+ if (!eos(state)) {
1962
+ raise_syntax_error("unexpected token at end of stream %s", state);
1146
1963
  }
1147
1964
  }
1148
1965
 
@@ -1160,37 +1977,59 @@ static void json_ensure_eof(JSON_ParserState *state)
1160
1977
 
1161
1978
  static VALUE convert_encoding(VALUE source)
1162
1979
  {
1163
- int encindex = RB_ENCODING_GET(source);
1980
+ StringValue(source);
1981
+ int encindex = RB_ENCODING_GET(source);
1982
+
1983
+ if (RB_LIKELY(encindex == utf8_encindex)) {
1984
+ return source;
1985
+ }
1164
1986
 
1165
- if (RB_LIKELY(encindex == utf8_encindex)) {
1987
+ if (encindex == binary_encindex) {
1988
+ // For historical reason, we silently reinterpret binary strings as UTF-8
1989
+ return rb_enc_associate_index(rb_str_dup(source), utf8_encindex);
1990
+ }
1991
+
1992
+ source = rb_funcall(source, i_encode, 1, Encoding_UTF_8);
1993
+ StringValue(source);
1166
1994
  return source;
1167
- }
1995
+ }
1168
1996
 
1169
- if (encindex == binary_encindex) {
1170
- // For historical reason, we silently reinterpret binary strings as UTF-8
1171
- return rb_enc_associate_index(rb_str_dup(source), utf8_encindex);
1172
- }
1997
+ struct parser_config_init_args {
1998
+ JSON_ParserConfig *config;
1999
+ VALUE self;
2000
+ VALUE unknown_keywords;
2001
+ bool strict;
2002
+ };
1173
2003
 
1174
- return rb_funcall(source, i_encode, 1, Encoding_UTF_8);
2004
+ static void parser_config_wb_write(VALUE self, VALUE *dest, VALUE val)
2005
+ {
2006
+ *dest = val;
2007
+ if (self) RB_OBJ_WRITTEN(self, Qundef, val);
1175
2008
  }
1176
2009
 
1177
2010
  static int parser_config_init_i(VALUE key, VALUE val, VALUE data)
1178
2011
  {
1179
- JSON_ParserConfig *config = (JSON_ParserConfig *)data;
1180
-
1181
- if (key == sym_max_nesting) { config->max_nesting = RTEST(val) ? FIX2INT(val) : 0; }
1182
- else if (key == sym_allow_nan) { config->allow_nan = RTEST(val); }
1183
- else if (key == sym_allow_trailing_comma) { config->allow_trailing_comma = RTEST(val); }
1184
- else if (key == sym_symbolize_names) { config->symbolize_names = RTEST(val); }
1185
- else if (key == sym_freeze) { config->freeze = RTEST(val); }
1186
- else if (key == sym_on_load) { config->on_load_proc = RTEST(val) ? val : Qfalse; }
1187
- else if (key == sym_decimal_class) {
2012
+ struct parser_config_init_args *args = (struct parser_config_init_args *)data;
2013
+ JSON_ParserConfig *config = args->config;
2014
+ VALUE self = args->self;
2015
+
2016
+ if (key == sym_max_nesting) { config->max_nesting = RTEST(val) ? FIX2INT(val) : 0; }
2017
+ else if (key == sym_allow_nan) { config->allow_nan = RTEST(val); }
2018
+ else if (key == sym_allow_trailing_comma) { config->allow_trailing_comma = RTEST(val); }
2019
+ else if (key == sym_allow_comments) { config->allow_comments = RTEST(val); }
2020
+ else if (key == sym_allow_control_characters) { config->allow_control_characters = RTEST(val); }
2021
+ else if (key == sym_allow_invalid_escape) { config->allow_invalid_escape = RTEST(val); }
2022
+ else if (key == sym_symbolize_names) { config->symbolize_names = RTEST(val); }
2023
+ else if (key == sym_freeze) { config->freeze = RTEST(val); }
2024
+ else if (key == sym_on_load) { parser_config_wb_write(self, &config->on_load_proc, RTEST(val) ? val : Qfalse); }
2025
+ else if (key == sym_allow_duplicate_key) { config->allow_duplicate_key = RTEST(val); }
2026
+ else if (key == sym_decimal_class) {
1188
2027
  if (RTEST(val)) {
1189
2028
  if (rb_respond_to(val, i_try_convert)) {
1190
- config->decimal_class = val;
2029
+ parser_config_wb_write(self, &config->decimal_class, val);
1191
2030
  config->decimal_method_id = i_try_convert;
1192
2031
  } else if (rb_respond_to(val, i_new)) {
1193
- config->decimal_class = val;
2032
+ parser_config_wb_write(self, &config->decimal_class, val);
1194
2033
  config->decimal_method_id = i_new;
1195
2034
  } else if (RB_TYPE_P(val, T_CLASS)) {
1196
2035
  VALUE name = rb_class_name(val);
@@ -1199,7 +2038,7 @@ static int parser_config_init_i(VALUE key, VALUE val, VALUE data)
1199
2038
  if (last_colon) {
1200
2039
  const char *mod_path_end = last_colon - 1;
1201
2040
  VALUE mod_path = rb_str_substr(name, 0, mod_path_end - name_cstr);
1202
- config->decimal_class = rb_path_to_class(mod_path);
2041
+ parser_config_wb_write(self, &config->decimal_class, rb_path_to_class(mod_path));
1203
2042
 
1204
2043
  const char *method_name_beg = last_colon + 1;
1205
2044
  long before_len = method_name_beg - name_cstr;
@@ -1207,29 +2046,41 @@ static int parser_config_init_i(VALUE key, VALUE val, VALUE data)
1207
2046
  VALUE method_name = rb_str_substr(name, before_len, len);
1208
2047
  config->decimal_method_id = SYM2ID(rb_str_intern(method_name));
1209
2048
  } else {
1210
- config->decimal_class = rb_mKernel;
2049
+ parser_config_wb_write(self, &config->decimal_class, rb_mKernel);
1211
2050
  config->decimal_method_id = SYM2ID(rb_str_intern(name));
1212
2051
  }
1213
2052
  }
1214
2053
  }
1215
2054
  }
2055
+ else {
2056
+ if (!args->unknown_keywords) {
2057
+ args->unknown_keywords = rb_obj_hide(rb_ary_new());
2058
+ }
2059
+ rb_ary_push(args->unknown_keywords, key);
2060
+ }
1216
2061
 
1217
2062
  return ST_CONTINUE;
1218
2063
  }
1219
2064
 
1220
- static void parser_config_init(JSON_ParserConfig *config, VALUE opts)
2065
+ static void parser_config_init(JSON_ParserConfig *config, VALUE opts, VALUE self, bool strict)
1221
2066
  {
1222
2067
  config->max_nesting = 100;
1223
2068
 
1224
- if (!NIL_P(opts)) {
1225
- Check_Type(opts, T_HASH);
1226
- if (RHASH_SIZE(opts) > 0) {
1227
- // We assume in most cases few keys are set so it's faster to go over
1228
- // the provided keys than to check all possible keys.
1229
- rb_hash_foreach(opts, parser_config_init_i, (VALUE)config);
1230
- }
2069
+ struct parser_config_init_args args = {
2070
+ .config = config,
2071
+ .self = self,
2072
+ .strict = strict,
2073
+ };
1231
2074
 
1232
- }
2075
+ if (NIL_P(opts)) return;
2076
+ Check_Type(opts, T_HASH);
2077
+ if (RHASH_SIZE(opts) == 0) return;
2078
+
2079
+ // We assume in most cases few keys are set so it's faster to go over
2080
+ // the provided keys than to check all possible keys.
2081
+ rb_hash_foreach(opts, parser_config_init_i, (VALUE)&args);
2082
+
2083
+ raise_argument_error_on_unknown_keywords(args.unknown_keywords);
1233
2084
  }
1234
2085
 
1235
2086
  /*
@@ -1237,66 +2088,90 @@ static void parser_config_init(JSON_ParserConfig *config, VALUE opts)
1237
2088
  *
1238
2089
  * Creates a new JSON::Ext::ParserConfig instance.
1239
2090
  *
1240
- * It will be configured by the _opts_ hash. _opts_ can have the following
1241
- * keys:
2091
+ * Argument +opts+, if given, contains a \Hash of options for the parsing.
2092
+ * See {Parsing Options}[#module-JSON-label-Parsing+Options].
1242
2093
  *
1243
- * _opts_ can have the following keys:
1244
- * * *max_nesting*: The maximum depth of nesting allowed in the parsed data
1245
- * structures. Disable depth checking with :max_nesting => false|nil|0, it
1246
- * defaults to 100.
1247
- * * *allow_nan*: If set to true, allow NaN, Infinity and -Infinity in
1248
- * defiance of RFC 4627 to be parsed by the Parser. This option defaults to
1249
- * false.
1250
- * * *symbolize_names*: If set to true, returns symbols for the names
1251
- * (keys) in a JSON object. Otherwise strings are returned, which is
1252
- * also the default. It's not possible to use this option in
1253
- * conjunction with the *create_additions* option.
1254
- * * *decimal_class*: Specifies which class to use instead of the default
1255
- * (Float) when parsing decimal numbers. This class must accept a single
1256
- * string argument in its constructor.
1257
2094
  */
1258
2095
  static VALUE cParserConfig_initialize(VALUE self, VALUE opts)
1259
2096
  {
2097
+ rb_check_frozen(self);
1260
2098
  GET_PARSER_CONFIG;
1261
2099
 
1262
- parser_config_init(config, opts);
1263
-
1264
- RB_OBJ_WRITTEN(self, Qundef, config->decimal_class);
2100
+ parser_config_init(config, opts, self, false);
1265
2101
 
1266
2102
  return self;
1267
2103
  }
1268
2104
 
1269
- static VALUE cParser_parse(JSON_ParserConfig *config, VALUE Vsource)
2105
+ static VALUE cParser_parse(JSON_ParserConfig *config, VALUE src)
1270
2106
  {
1271
- Vsource = convert_encoding(StringValue(Vsource));
1272
- StringValue(Vsource);
2107
+ VALUE Vsource = convert_encoding(src);
2108
+
2109
+ // Ensure the string isn't mutated under us.
2110
+ // The classic API to use is `rb_str_locktmp`, but then we'd
2111
+ // need to use `rb_protect` to make sure we always unlock.
2112
+ if (Vsource == src) {
2113
+ Vsource = rb_str_new_frozen(Vsource);
2114
+ }
1273
2115
 
1274
2116
  VALUE rvalue_stack_buffer[RVALUE_STACK_INITIAL_CAPA];
1275
- rvalue_stack stack = {
2117
+ rvalue_stack value_stack = {
1276
2118
  .type = RVALUE_STACK_STACK_ALLOCATED,
1277
2119
  .ptr = rvalue_stack_buffer,
1278
2120
  .capa = RVALUE_STACK_INITIAL_CAPA,
1279
2121
  };
1280
2122
 
2123
+ // Seed the frame stack with the root frame, establishing the invariant that
2124
+ // json_parse_any always has a top frame to dispatch on (so the stack is never
2125
+ // empty mid-parse).
2126
+ json_frame frame_stack_buffer[JSON_FRAME_STACK_INITIAL_CAPA];
2127
+ frame_stack_buffer[0] = (json_frame){
2128
+ .type = JSON_FRAME_ROOT,
2129
+ .phase = JSON_PHASE_VALUE,
2130
+ };
2131
+ json_frame_stack frames = {
2132
+ .type = RVALUE_STACK_STACK_ALLOCATED,
2133
+ .ptr = frame_stack_buffer,
2134
+ .capa = JSON_FRAME_STACK_INITIAL_CAPA,
2135
+ .head = 1,
2136
+ };
2137
+
1281
2138
  long len;
1282
2139
  const char *start;
2140
+
1283
2141
  RSTRING_GETMEM(Vsource, start, len);
1284
2142
 
2143
+ VALUE value_stack_handle = 0;
2144
+ VALUE frame_stack_handle = 0;
1285
2145
  JSON_ParserState _state = {
1286
2146
  .start = start,
1287
2147
  .cursor = start,
1288
2148
  .end = start + len,
1289
- .stack = &stack,
2149
+ .value_stack = &value_stack,
2150
+ .value_stack_handle = &value_stack_handle,
2151
+ .frames = &frames,
2152
+ .frame_stack_handle = &frame_stack_handle,
1290
2153
  };
1291
2154
  JSON_ParserState *state = &_state;
1292
2155
 
1293
- VALUE result = json_parse_any(state, config);
2156
+ bool complete = json_parse_any(state, config, false);
2157
+
2158
+ // The root document value is parsed; it is the lone survivor on
2159
+ // the rvalue stack.
2160
+ VALUE result = complete ? *rvalue_stack_peek(state->value_stack, 1) : Qundef;
2161
+
2162
+ if (complete) {
2163
+ json_ensure_eof(state, config);
2164
+ } else {
2165
+ raise_eos_error("unexpected end of input", state);
2166
+ }
1294
2167
 
1295
2168
  // This may be skipped in case of exception, but
1296
2169
  // it won't cause a leak.
1297
- rvalue_stack_eagerly_release(state->stack_handle);
1298
-
1299
- json_ensure_eof(state);
2170
+ rvalue_stack_eagerly_release(value_stack_handle);
2171
+ json_frame_stack_eagerly_release(frame_stack_handle);
2172
+ RB_GC_GUARD(value_stack_handle);
2173
+ RB_GC_GUARD(frame_stack_handle);
2174
+ RB_GC_GUARD(Vsource);
1300
2175
 
1301
2176
  return result;
1302
2177
  }
@@ -1316,12 +2191,9 @@ static VALUE cParserConfig_parse(VALUE self, VALUE Vsource)
1316
2191
 
1317
2192
  static VALUE cParser_m_parse(VALUE klass, VALUE Vsource, VALUE opts)
1318
2193
  {
1319
- Vsource = convert_encoding(StringValue(Vsource));
1320
- StringValue(Vsource);
1321
-
1322
2194
  JSON_ParserConfig _config = {0};
1323
2195
  JSON_ParserConfig *config = &_config;
1324
- parser_config_init(config, opts);
2196
+ parser_config_init(config, opts, Qfalse, false);
1325
2197
 
1326
2198
  return cParser_parse(config, Vsource);
1327
2199
  }
@@ -1329,30 +2201,35 @@ static VALUE cParser_m_parse(VALUE klass, VALUE Vsource, VALUE opts)
1329
2201
  static void JSON_ParserConfig_mark(void *ptr)
1330
2202
  {
1331
2203
  JSON_ParserConfig *config = ptr;
1332
- rb_gc_mark(config->on_load_proc);
1333
- rb_gc_mark(config->decimal_class);
2204
+ rb_gc_mark_movable(config->on_load_proc);
2205
+ rb_gc_mark_movable(config->decimal_class);
1334
2206
  }
1335
2207
 
1336
- static void JSON_ParserConfig_free(void *ptr)
2208
+ static size_t JSON_ParserConfig_memsize(const void *ptr)
1337
2209
  {
1338
- JSON_ParserConfig *config = ptr;
1339
- ruby_xfree(config);
2210
+ #ifdef HAVE_RUBY_TYPED_EMBEDDABLE
2211
+ return 0;
2212
+ #else
2213
+ return sizeof(JSON_ParserConfig);
2214
+ #endif
1340
2215
  }
1341
2216
 
1342
- static size_t JSON_ParserConfig_memsize(const void *ptr)
2217
+ static void JSON_ParserConfig_compact(void *ptr)
1343
2218
  {
1344
- return sizeof(JSON_ParserConfig);
2219
+ JSON_ParserConfig *config = ptr;
2220
+ config->on_load_proc = rb_gc_location(config->on_load_proc);
2221
+ config->decimal_class = rb_gc_location(config->decimal_class);
1345
2222
  }
1346
2223
 
1347
2224
  static const rb_data_type_t JSON_ParserConfig_type = {
1348
- "JSON::Ext::Parser/ParserConfig",
1349
- {
1350
- JSON_ParserConfig_mark,
1351
- JSON_ParserConfig_free,
1352
- JSON_ParserConfig_memsize,
2225
+ .wrap_struct_name = "JSON::Ext::Parser/ParserConfig",
2226
+ .function = {
2227
+ .dmark = JSON_ParserConfig_mark,
2228
+ .dfree = RUBY_DEFAULT_FREE,
2229
+ .dsize = JSON_ParserConfig_memsize,
2230
+ .dcompact = JSON_ParserConfig_compact,
1353
2231
  },
1354
- 0, 0,
1355
- RUBY_TYPED_FREE_IMMEDIATELY | RUBY_TYPED_WB_PROTECTED,
2232
+ .flags = RUBY_TYPED_THREAD_SAFE_FREE | RUBY_TYPED_WB_PROTECTED | RUBY_TYPED_FROZEN_SHAREABLE | RUBY_TYPED_EMBEDDABLE,
1356
2233
  };
1357
2234
 
1358
2235
  static VALUE cJSON_parser_s_allocate(VALUE klass)
@@ -1361,6 +2238,602 @@ static VALUE cJSON_parser_s_allocate(VALUE klass)
1361
2238
  return TypedData_Make_Struct(klass, JSON_ParserConfig, &JSON_ParserConfig_type, config);
1362
2239
  }
1363
2240
 
2241
+ static void json_str_clear(VALUE str)
2242
+ {
2243
+ if (RB_OBJ_FROZEN_RAW(str)) {
2244
+ return;
2245
+ }
2246
+ rb_str_replace(str, JSON_empty_string);
2247
+ }
2248
+
2249
+ typedef struct JSON_ResumableParserStruct {
2250
+ JSON_ParserConfig config;
2251
+ JSON_ParserState state;
2252
+ rvalue_stack value_stack;
2253
+ json_frame_stack frames;
2254
+ VALUE buffer;
2255
+ size_t parsed_bytes;
2256
+ size_t incomplete_bytes;
2257
+ bool complete;
2258
+ bool in_use;
2259
+ } JSON_ResumableParser;
2260
+
2261
+ static void JSON_ResumableParser_mark(void *ptr)
2262
+ {
2263
+ JSON_ResumableParser *parser = (JSON_ResumableParser *)ptr;
2264
+ JSON_ParserConfig_mark(&parser->config);
2265
+ rvalue_stack_mark(&parser->value_stack);
2266
+ rvalue_cache_mark(&parser->state.name_cache);
2267
+ rb_gc_mark(parser->buffer); // pin the buffer
2268
+ rb_gc_mark_movable(parser->state.parser);
2269
+ }
2270
+
2271
+ static void JSON_ResumableParser_free(void *ptr)
2272
+ {
2273
+ JSON_ResumableParser *parser = (JSON_ResumableParser *)ptr;
2274
+ rvalue_stack_free_buffer(&parser->value_stack);
2275
+ json_frame_stack_free_buffer(&parser->frames);
2276
+ }
2277
+
2278
+ static size_t JSON_ResumableParser_memsize(const void *ptr)
2279
+ {
2280
+ const JSON_ResumableParser *parser = (const JSON_ResumableParser *)ptr;
2281
+ size_t memsize = JSON_ParserConfig_memsize(&parser->config);
2282
+ memsize += rvalue_stack_memsize(&parser->value_stack);
2283
+ memsize += json_frame_stack_memsize(&parser->frames);
2284
+ #ifndef HAVE_RUBY_TYPED_EMBEDDABLE
2285
+ memsize += (
2286
+ sizeof(JSON_ResumableParser)
2287
+ - sizeof(JSON_ParserState)
2288
+ - sizeof(JSON_ParserConfig)
2289
+ - sizeof(rvalue_stack)
2290
+ - sizeof(json_frame_stack)
2291
+ );
2292
+ #endif
2293
+ return memsize;
2294
+ }
2295
+
2296
+ static void JSON_ResumableParser_compact(void *ptr)
2297
+ {
2298
+ JSON_ResumableParser *parser = (JSON_ResumableParser *)ptr;
2299
+ JSON_ParserConfig_compact(&parser->config);
2300
+ rvalue_stack_compact(&parser->value_stack);
2301
+ rvalue_cache_compact(&parser->state.name_cache);
2302
+ parser->buffer = rb_gc_location(parser->buffer);
2303
+ parser->state.parser = rb_gc_location(parser->state.parser);
2304
+ }
2305
+
2306
+ static const rb_data_type_t JSON_ResumableParser_type = {
2307
+ .wrap_struct_name = "JSON::Ext::ResumableParser",
2308
+ .function = {
2309
+ JSON_ResumableParser_mark,
2310
+ JSON_ResumableParser_free,
2311
+ JSON_ResumableParser_memsize,
2312
+ JSON_ResumableParser_compact,
2313
+ },
2314
+ // RUBY_TYPED_WB_PROTECTED is deliberately not declared because
2315
+ // this is a superset of JSON_Parser_rvalue_stack_type, so we'd need
2316
+ // to trigger a lot of write barriers.
2317
+ .flags = RUBY_TYPED_THREAD_SAFE_FREE | RUBY_TYPED_EMBEDDABLE,
2318
+ };
2319
+
2320
+ static VALUE cResumableParser_allocate(VALUE klass)
2321
+ {
2322
+ JSON_ResumableParser *parser;
2323
+ VALUE obj = TypedData_Make_Struct(klass, JSON_ResumableParser, &JSON_ResumableParser_type, parser);
2324
+ parser->state.in_array++;
2325
+ parser->state.parser = obj;
2326
+ return obj;
2327
+ }
2328
+
2329
+ static inline JSON_ResumableParser *cResumableParser_get(VALUE self)
2330
+ {
2331
+ JSON_ResumableParser *parser;
2332
+ TypedData_Get_Struct(self, JSON_ResumableParser, &JSON_ResumableParser_type, parser);
2333
+ return parser;
2334
+ }
2335
+
2336
+ /*
2337
+ * call-seq: new(opts => {})
2338
+ *
2339
+ * Creates a new JSON::ResumableParser instance.
2340
+ *
2341
+ * Argument +opts+, if given, contains a \Hash of options for the parsing.
2342
+ * See {Parsing Options}[#module-JSON-label-Parsing+Options].
2343
+ *
2344
+ * A ResumableParser is able to parse partial documents and resume parsing later
2345
+ * when more of the document is provided:
2346
+ *
2347
+ * parser = JSON::ResumableParser.new
2348
+ * parser << '{"user": "george", "role": "ad'
2349
+ * parser.parse # => false
2350
+ * parser.eos? # => true
2351
+ * parser.partial_value # => { "user" => "george", "role" => nil }
2352
+ * parser.rest # => '"ad'
2353
+ *
2354
+ * parser << 'min" }[1, 2, 3]'
2355
+ * parser.parse # => true
2356
+ * parser.value # => { "user" => "george", "role" => "admin" }
2357
+ *
2358
+ * parser.parse # => true
2359
+ * parser.value # => [1, 2, 3]
2360
+ *
2361
+ * === Limitations
2362
+ *
2363
+ * While ResumableParser is able to parse streams of documents without any
2364
+ * explicit separators between them, it is highly recommended to separate documents
2365
+ * by either spaces or newlines, as otherwise the \JSON syntax for numbers may be ambiguous.
2366
+ * When parsing a number, ResumableParser will not consider the number complete until something follows:
2367
+ *
2368
+ * parser << '123'
2369
+ * parser.parse # => false
2370
+ * parser << ' '
2371
+ * parser.parse # => true
2372
+ * parser.value # => 123
2373
+ *
2374
+ * === Security
2375
+ *
2376
+ * An incomplete document is buffered in full and there is no size limit, so when reading
2377
+ * from an untrusted source the caller is responsible for bounding how much data is fed.
2378
+ * For example:
2379
+ *
2380
+ * loop do
2381
+ * if parser.parsed_bytes > DOCUMENT_MAX_SIZE
2382
+ * raise "document too large"
2383
+ * end
2384
+ *
2385
+ * parser << read_chunk
2386
+ * while parser.parse
2387
+ * process(parser.value)
2388
+ * end
2389
+ * end
2390
+ */
2391
+ static VALUE cResumableParser_initialize(int argc, VALUE *argv, VALUE self)
2392
+ {
2393
+ rb_check_frozen(self);
2394
+
2395
+ VALUE opts = Qfalse;
2396
+ rb_scan_args_kw(RB_SCAN_ARGS_LAST_HASH_KEYWORDS, argc, argv, "0:", &opts);
2397
+ JSON_ResumableParser *parser = cResumableParser_get(self);
2398
+
2399
+ opts = argc > 0 ? argv[0] : Qnil;
2400
+ parser_config_init(&parser->config, opts, self, true);
2401
+
2402
+ return self;
2403
+ }
2404
+
2405
+ static JSON_ResumableParser *ResumableParser_acquire(VALUE self, bool lock);
2406
+
2407
+ /*
2408
+ * call-seq: self << string -> self
2409
+ *
2410
+ * Appends the given string to the parser's buffer.
2411
+ */
2412
+ static VALUE cResumableParser_feed(VALUE self, VALUE str)
2413
+ {
2414
+ rb_check_frozen(self);
2415
+
2416
+ JSON_ResumableParser *parser = ResumableParser_acquire(self, false);
2417
+
2418
+ str = convert_encoding(str);
2419
+ if (!RSTRING_LEN(str)) {
2420
+ return self;
2421
+ }
2422
+
2423
+ size_t offset = parser->state.cursor - parser->state.start;
2424
+ const size_t remaining = parser->state.end - parser->state.cursor;
2425
+
2426
+ if (!remaining) {
2427
+ if (parser->buffer) {
2428
+ json_str_clear(parser->buffer);
2429
+ }
2430
+ parser->buffer = RB_OBJ_FROZEN_RAW(str) ? str : rb_obj_hide(rb_str_new_shared(str));
2431
+ offset = 0;
2432
+ } else {
2433
+ JSON_ASSERT(parser->buffer);
2434
+
2435
+ const size_t size = parser->state.end - parser->state.start;
2436
+ const size_t consumed = size - remaining;
2437
+
2438
+ if (RB_OBJ_FROZEN_RAW(parser->buffer)) {
2439
+ VALUE new_buffer = rb_obj_hide(rb_str_buf_new(remaining + RSTRING_LEN(str)));
2440
+ rb_enc_associate_index(new_buffer, utf8_encindex);
2441
+
2442
+ char *old_ptr = RSTRING_PTR(parser->buffer);
2443
+ memcpy(RSTRING_PTR(new_buffer), old_ptr + consumed, remaining);
2444
+ rb_str_set_len(new_buffer, remaining);
2445
+ offset = 0;
2446
+ parser->buffer = new_buffer;
2447
+ } else if (consumed > (size / 2) && size >= 512) {
2448
+ rb_str_modify(parser->buffer);
2449
+ char *old_ptr = RSTRING_PTR(parser->buffer);
2450
+ memmove(old_ptr, old_ptr + consumed, remaining);
2451
+ rb_str_set_len(parser->buffer, remaining);
2452
+ offset = 0;
2453
+ }
2454
+ rb_str_append(parser->buffer, str);
2455
+ }
2456
+
2457
+ long len;
2458
+ const char *start;
2459
+ RSTRING_GETMEM(parser->buffer, start, len);
2460
+ parser->state.start = start;
2461
+ parser->state.end = start + len;
2462
+ parser->state.cursor = parser->state.start + offset;
2463
+
2464
+ return self;
2465
+ }
2466
+
2467
+ struct json_parse_any_args {
2468
+ JSON_ParserState *state;
2469
+ JSON_ParserConfig *config;
2470
+ VALUE parser;
2471
+ };
2472
+
2473
+ static VALUE json_parse_any_resumable_safe0(RB_BLOCK_CALL_FUNC_ARGLIST(yielded_arg, _args))
2474
+ {
2475
+ struct json_parse_any_args *args = (struct json_parse_any_args *)_args;
2476
+ return (VALUE)json_parse_any(args->state, args->config, true);
2477
+ }
2478
+
2479
+ static VALUE json_parse_any_resumable_safe(VALUE _args)
2480
+ {
2481
+ struct json_parse_any_args *args = (struct json_parse_any_args *)_args;
2482
+ VALUE result = rb_catch_obj(args->parser, json_parse_any_resumable_safe0, _args);
2483
+ return result == args->parser ? Qfalse : result;
2484
+ }
2485
+
2486
+ static JSON_ResumableParser *ResumableParser_acquire(VALUE self, bool lock)
2487
+ {
2488
+ JSON_ResumableParser *parser = cResumableParser_get(self);
2489
+
2490
+ if (parser->in_use) {
2491
+ rb_raise(rb_eArgError, "ResumableParser can't be used recursively");
2492
+ }
2493
+
2494
+ if (lock) {
2495
+ parser->in_use = true;
2496
+ }
2497
+
2498
+ // self may have moved, so we need to update all pointers
2499
+ // Investigate: We might be better off keeping JSON_ParserState on the stack
2500
+ // and only persist what we need.
2501
+ parser->state.value_stack = &parser->value_stack;
2502
+ parser->state.frames = &parser->frames;
2503
+
2504
+ return parser;
2505
+ }
2506
+
2507
+ /*
2508
+ * call-seq: parse -> true or false
2509
+ *
2510
+ * Attemps to parse a JSON document from the internal buffer.
2511
+ * Returns whether a complete document could be parsed.
2512
+ *
2513
+ * It does raise +JSON::ParserError+ when encountering invalid \JSON syntax.
2514
+ *
2515
+ * The parsed object can be retrieved by calling #value
2516
+ */
2517
+ static VALUE cResumableParser_parse(VALUE self)
2518
+ {
2519
+ JSON_ResumableParser *parser = ResumableParser_acquire(self, true);
2520
+
2521
+ if (parser->complete) {
2522
+ parser->parsed_bytes = 0;
2523
+ parser->incomplete_bytes = 0;
2524
+ parser->complete = false;
2525
+ }
2526
+
2527
+ if (!parser->buffer) {
2528
+ parser->in_use = false;
2529
+ return Qfalse;
2530
+ }
2531
+
2532
+ if (parser->frames.head == 0) {
2533
+ json_frame_stack_push(&parser->state, (json_frame){
2534
+ .type = JSON_FRAME_ROOT,
2535
+ .phase = JSON_PHASE_VALUE,
2536
+ });
2537
+ }
2538
+
2539
+ VALUE Vsource = parser->buffer; // Prevent compaction
2540
+
2541
+ json_frame *frame = json_frame_stack_peek(&parser->frames);
2542
+
2543
+ if (frame->phase == JSON_PHASE_DONE) {
2544
+ JSON_ASSERT(parser->value_stack.head == 1);
2545
+ JSON_ASSERT(parser->frames.head == 1);
2546
+
2547
+ frame->phase = JSON_PHASE_VALUE;
2548
+ rvalue_stack_pop(parser->state.value_stack, 1);
2549
+ }
2550
+
2551
+ struct json_parse_any_args args = {
2552
+ .state = &parser->state,
2553
+ .config = &parser->config,
2554
+ .parser = self,
2555
+ };
2556
+ int status;
2557
+ const char *initial_cursor = parser->state.cursor;
2558
+ parser->complete = rb_protect(json_parse_any_resumable_safe, (VALUE)&args, &status);
2559
+
2560
+ if (status) {
2561
+ parser->complete = true; // a parse error is considered complete
2562
+ }
2563
+
2564
+ parser->parsed_bytes += parser->state.cursor - initial_cursor;
2565
+ parser->incomplete_bytes = parser->complete ? 0 : parser->state.end - parser->state.cursor;
2566
+
2567
+ json_eat_whitespace(&parser->state, &parser->config, false);
2568
+ if (eos(&parser->state)) {
2569
+ json_str_clear(parser->buffer);
2570
+ parser->buffer = Qfalse;
2571
+ parser->state.start = parser->state.cursor = parser->state.end = 0;
2572
+ }
2573
+ parser->in_use = false;
2574
+
2575
+ if (status) {
2576
+ rb_jump_tag(status); // reraise
2577
+ }
2578
+ RB_GC_GUARD(Vsource);
2579
+ return parser->complete ? Qtrue : Qfalse;
2580
+ }
2581
+
2582
+ /*
2583
+ * call-seq: value? -> true or false
2584
+ *
2585
+ * Returns whether a parsed value is available.
2586
+ */
2587
+ static VALUE cResumableParser_value_p(VALUE self)
2588
+ {
2589
+ JSON_ResumableParser *parser = ResumableParser_acquire(self, false);
2590
+
2591
+ if (parser->value_stack.head > 0) {
2592
+ json_frame *frame = json_frame_stack_peek(&parser->frames);
2593
+ if (frame->phase == JSON_PHASE_DONE) {
2594
+ return Qtrue;
2595
+ }
2596
+ }
2597
+ return Qfalse;
2598
+ }
2599
+
2600
+ /*
2601
+ * call-seq: value -> object
2602
+ *
2603
+ * Returns and consume the last parsed value.
2604
+ * Raises ArgumentError if there is no parsed value or if it was already retrieved:
2605
+ * parser << '[1][2]'
2606
+ * parser.value # ArgumentError no ready value
2607
+ * parser.parse # => true
2608
+ * parser.value # => [1]
2609
+ * parser.value # ArgumentError no ready value
2610
+ */
2611
+ static VALUE cResumableParser_value(VALUE self)
2612
+ {
2613
+ JSON_ResumableParser *parser = ResumableParser_acquire(self, false);
2614
+
2615
+ if (parser->frames.head > 0) {
2616
+ json_frame *frame = json_frame_stack_peek(&parser->frames);
2617
+
2618
+ if (frame->phase == JSON_PHASE_DONE) {
2619
+ VALUE result = *rvalue_stack_peek(parser->state.value_stack, 1);
2620
+ rvalue_stack_pop(parser->state.value_stack, 1);
2621
+ json_frame_stack_pop(parser->state.frames);
2622
+ return result;
2623
+ }
2624
+ }
2625
+ rb_raise(rb_eArgError, "no ready value");
2626
+ }
2627
+
2628
+ /*
2629
+ * call-seq: clear -> self
2630
+ *
2631
+ * Entirely reset the parser state and buffer.
2632
+ */
2633
+ static VALUE cResumableParser_clear(VALUE self)
2634
+ {
2635
+ JSON_ResumableParser *parser = ResumableParser_acquire(self, false);
2636
+ parser->buffer = 0;
2637
+ parser->complete = true;
2638
+ parser->parsed_bytes = 0;
2639
+ parser->incomplete_bytes = 0;
2640
+ parser->frames.head = 0;
2641
+ parser->value_stack.head = 0;
2642
+ parser->state.name_cache.length = 0;
2643
+ parser->state.current_nesting = 0;
2644
+ parser->state.in_array = 1;
2645
+ parser->state.start = parser->state.cursor = parser->state.end = NULL;
2646
+ return self;
2647
+ }
2648
+
2649
+ static VALUE cResumableParser_partial_value_body(VALUE self)
2650
+ {
2651
+ JSON_ResumableParser *original_parser = cResumableParser_get(self);
2652
+ JSON_ResumableParser parser = *original_parser;
2653
+
2654
+ parser.state.frames = &parser.frames;
2655
+ parser.state.value_stack = &parser.value_stack;
2656
+
2657
+ if (parser.value_stack.head == 0) {
2658
+ return Qnil;
2659
+ }
2660
+
2661
+ json_frame *frame = json_frame_stack_peek(parser.state.frames);
2662
+ long missing_object_value = 0;
2663
+ if (frame->type == JSON_FRAME_OBJECT && (frame->phase == JSON_PHASE_VALUE || frame->phase == JSON_PHASE_OBJECT_COLON)) {
2664
+ missing_object_value = 1;
2665
+ }
2666
+
2667
+ // Copy the value stack as we need to mutate it. The collapse loop folds each
2668
+ // open container by popping its entries and pushing the single result, so a
2669
+ // parent always reclaims its child's slot; head exceeds its live size by at
2670
+ // most one, either for the missing-value placeholder pushed below or for the
2671
+ // result of folding an empty innermost container. That one spare slot keeps
2672
+ // rvalue_stack_push from growing (reallocating) this ALLOCV buffer.
2673
+ long capa = parser.value_stack.head;
2674
+ parser.value_stack.capa = capa + 1;
2675
+ VALUE tmpbuf, *value_stack_buffer = ALLOCV_N(VALUE, tmpbuf, parser.value_stack.capa);
2676
+ MEMCPY(value_stack_buffer, parser.value_stack.ptr, VALUE, capa);
2677
+ parser.value_stack.ptr = value_stack_buffer;
2678
+
2679
+ JSON_ParserState *state = &parser.state;
2680
+ JSON_ParserConfig *config = &parser.config;
2681
+
2682
+ if (missing_object_value) {
2683
+ rvalue_stack_push(state->value_stack, Qnil, NULL, &state->value_stack);
2684
+ }
2685
+
2686
+ VALUE partial_result = Qundef;
2687
+
2688
+ while (UNDEF_P(partial_result)) {
2689
+ frame = json_frame_stack_peek(state->frames);
2690
+
2691
+ switch (frame->type) {
2692
+ case JSON_FRAME_ROOT: {
2693
+ partial_result = *rvalue_stack_peek(state->value_stack, 1);
2694
+ break;
2695
+ }
2696
+
2697
+ case JSON_FRAME_ARRAY: {
2698
+ long count = json_frame_entry_count(frame, state->value_stack);
2699
+ json_push_value(state, config, json_decode_array(state, config, count));
2700
+ json_frame_stack_pop(state->frames);
2701
+
2702
+ break;
2703
+ }
2704
+
2705
+ case JSON_FRAME_OBJECT: {
2706
+ long count = json_frame_entry_count(frame, state->value_stack);
2707
+ json_push_value(state, config, json_decode_object(state, config, count));
2708
+ json_frame_stack_pop(state->frames);
2709
+ break;
2710
+ }
2711
+
2712
+ default: {
2713
+ JSON_UNREACHABLE_RETURN(Qundef);
2714
+ break;
2715
+ }
2716
+ }
2717
+ }
2718
+
2719
+ ALLOCV_END(tmpbuf);
2720
+ return partial_result;
2721
+ }
2722
+
2723
+ /*
2724
+ * call-seq: partial_value -> object
2725
+ *
2726
+ * Returns the Ruby objects parsed up to this point:
2727
+ * parser << '[1, [2, 3,'
2728
+ * parser.parse # => false
2729
+ * parser.value # ArgumentError no ready value
2730
+ * parser.partial_value # => [1, [2, 3]]
2731
+ */
2732
+ static VALUE cResumableParser_partial_value(VALUE self)
2733
+ {
2734
+ JSON_ResumableParser *parser = ResumableParser_acquire(self, true);
2735
+
2736
+ int status;
2737
+ VALUE result = rb_protect(cResumableParser_partial_value_body, self, &status);
2738
+ parser->in_use = false;
2739
+ if (status) {
2740
+ rb_jump_tag(status);
2741
+ }
2742
+ return result;
2743
+ }
2744
+
2745
+ /*
2746
+ * call-seq: rest -> string
2747
+ *
2748
+ * Returns a string containing what remains to be parsed in the buffer
2749
+ * parser << '{ "message": "unterminated message'
2750
+ * parser.parse # => false
2751
+ * parser.rest # => '"unterminated message"'
2752
+ */
2753
+ static VALUE cResumableParser_rest(VALUE self)
2754
+ {
2755
+ JSON_ResumableParser *parser = cResumableParser_get(self);
2756
+
2757
+ if (!parser->buffer) {
2758
+ return rb_utf8_str_new("", 0);
2759
+ }
2760
+
2761
+ size_t offset = parser->state.cursor - parser->state.start;
2762
+ const char *ptr;
2763
+ long len;
2764
+ RSTRING_GETMEM(parser->buffer, ptr, len);
2765
+ return rb_utf8_str_new(ptr + offset, len - offset);
2766
+ }
2767
+
2768
+ /*
2769
+ * call-seq: eos? -> true or false
2770
+ *
2771
+ * Returns whether the internal buffer has been entirely consumed.
2772
+ */
2773
+ static VALUE cResumableParser_eos_p(VALUE self)
2774
+ {
2775
+ JSON_ResumableParser *parser = cResumableParser_get(self);
2776
+ return eos(&parser->state) ? Qtrue : Qfalse;
2777
+ }
2778
+
2779
+ /*
2780
+ * call-seq: partial_value? -> true or false
2781
+ *
2782
+ * Returns whether a document is currently under construction: an unclosed
2783
+ * container, a key awaiting its value, etc.
2784
+ *
2785
+ * It answers the same question as <tt>!partial_value.nil?</tt>, but as a
2786
+ * cheap predicate on the parser's internal state, without materializing the
2787
+ * partially parsed Ruby objects:
2788
+ * parser << '{"a":1,'
2789
+ * parser.parse # => false
2790
+ * parser.partial_value? # => true
2791
+ *
2792
+ * A fully parsed document whose value hasn't been retrieved yet is not under
2793
+ * construction: #value? returns true and #partial_value? returns false.
2794
+ */
2795
+ static VALUE cResumableParser_partial_value_p(VALUE self)
2796
+ {
2797
+ JSON_ResumableParser *parser = cResumableParser_get(self);
2798
+
2799
+ // Mirror of #value?: values on the stack while the document isn't DONE
2800
+ // belong to a partially built document. A container whose first key or
2801
+ // element hasn't been parsed yet has no frame nor value registered (the
2802
+ // tokenizer rewinds to the container start on EOS), so that state is
2803
+ // observable through the buffer (#eos?/#rest) instead, keeping this
2804
+ // predicate consistent with #partial_value returning nil.
2805
+ if (parser->value_stack.head > 0) {
2806
+ json_frame *frame = json_frame_stack_peek(&parser->frames);
2807
+ if (frame->phase != JSON_PHASE_DONE) {
2808
+ return Qtrue;
2809
+ }
2810
+ }
2811
+ return Qfalse;
2812
+ }
2813
+
2814
+ /*
2815
+ * call-seq: parsed_bytes -> integer
2816
+ *
2817
+ * Returns the number of bytes parsed since the start of the current partial value.
2818
+ * This is intended to be used for securing against untrusted input:
2819
+ *
2820
+ * loop do
2821
+ * if parser.parsed_bytes > DOCUMENT_MAX_SIZE
2822
+ * raise "document too large"
2823
+ * end
2824
+ *
2825
+ * parser << read_chunk
2826
+ * while parser.parse
2827
+ * process(parser.value)
2828
+ * end
2829
+ * end
2830
+ */
2831
+ static VALUE cResumableParser_parsed_bytes(VALUE self)
2832
+ {
2833
+ JSON_ResumableParser *parser = cResumableParser_get(self);
2834
+ return ULL2NUM(parser->parsed_bytes + parser->incomplete_bytes);
2835
+ }
2836
+
1364
2837
  void Init_parser(void)
1365
2838
  {
1366
2839
  #ifdef HAVE_RB_EXT_RACTOR_SAFE
@@ -1372,45 +2845,76 @@ void Init_parser(void)
1372
2845
  mJSON = rb_define_module("JSON");
1373
2846
  VALUE mExt = rb_define_module_under(mJSON, "Ext");
1374
2847
  VALUE cParserConfig = rb_define_class_under(mExt, "ParserConfig", rb_cObject);
2848
+
2849
+ rb_global_variable(&eParserError);
2850
+ eParserError = rb_path2class("JSON::ParserError");
2851
+
2852
+ rb_global_variable(&eNestingError);
1375
2853
  eNestingError = rb_path2class("JSON::NestingError");
1376
- rb_gc_register_mark_object(eNestingError);
2854
+
1377
2855
  rb_define_alloc_func(cParserConfig, cJSON_parser_s_allocate);
1378
- rb_define_method(cParserConfig, "initialize", cParserConfig_initialize, 1);
2856
+ rb_define_private_method(cParserConfig, "initialize", cParserConfig_initialize, 1);
1379
2857
  rb_define_method(cParserConfig, "parse", cParserConfig_parse, 1);
1380
2858
 
1381
2859
  VALUE cParser = rb_define_class_under(mExt, "Parser", rb_cObject);
1382
2860
  rb_define_singleton_method(cParser, "parse", cParser_m_parse, 2);
1383
2861
 
2862
+ VALUE cResumableParser = rb_define_class_under(mJSON, "ResumableParser", rb_cObject);
2863
+ rb_define_alloc_func(cResumableParser, cResumableParser_allocate);
2864
+ rb_define_private_method(cResumableParser, "initialize", cResumableParser_initialize, -1);
2865
+ rb_define_method(cResumableParser, "<<", cResumableParser_feed, 1);
2866
+ rb_define_method(cResumableParser, "parse", cResumableParser_parse, 0);
2867
+ rb_define_method(cResumableParser, "value", cResumableParser_value, 0);
2868
+ rb_define_method(cResumableParser, "value?", cResumableParser_value_p, 0);
2869
+ rb_define_method(cResumableParser, "partial_value", cResumableParser_partial_value, 0);
2870
+ rb_define_method(cResumableParser, "partial_value?", cResumableParser_partial_value_p, 0);
2871
+ rb_define_method(cResumableParser, "clear", cResumableParser_clear, 0);
2872
+ rb_define_method(cResumableParser, "rest", cResumableParser_rest, 0);
2873
+ rb_define_method(cResumableParser, "eos?", cResumableParser_eos_p, 0);
2874
+ rb_define_method(cResumableParser, "parsed_bytes", cResumableParser_parsed_bytes, 0);
2875
+
2876
+ rb_global_variable(&CNaN);
1384
2877
  CNaN = rb_const_get(mJSON, rb_intern("NaN"));
1385
- rb_gc_register_mark_object(CNaN);
1386
2878
 
2879
+ rb_global_variable(&CInfinity);
1387
2880
  CInfinity = rb_const_get(mJSON, rb_intern("Infinity"));
1388
- rb_gc_register_mark_object(CInfinity);
1389
2881
 
2882
+ rb_global_variable(&CMinusInfinity);
1390
2883
  CMinusInfinity = rb_const_get(mJSON, rb_intern("MinusInfinity"));
1391
- rb_gc_register_mark_object(CMinusInfinity);
1392
2884
 
1393
2885
  rb_global_variable(&Encoding_UTF_8);
1394
2886
  Encoding_UTF_8 = rb_const_get(rb_path2class("Encoding"), rb_intern("UTF_8"));
1395
2887
 
2888
+ rb_global_variable(&JSON_empty_string);
2889
+ JSON_empty_string = rb_obj_hide(rb_utf8_str_new("", 0));
2890
+
1396
2891
  sym_max_nesting = ID2SYM(rb_intern("max_nesting"));
1397
2892
  sym_allow_nan = ID2SYM(rb_intern("allow_nan"));
1398
2893
  sym_allow_trailing_comma = ID2SYM(rb_intern("allow_trailing_comma"));
2894
+ sym_allow_comments = ID2SYM(rb_intern("allow_comments"));
2895
+ sym_allow_control_characters = ID2SYM(rb_intern("allow_control_characters"));
2896
+ sym_allow_invalid_escape = ID2SYM(rb_intern("allow_invalid_escape"));
1399
2897
  sym_symbolize_names = ID2SYM(rb_intern("symbolize_names"));
1400
2898
  sym_freeze = ID2SYM(rb_intern("freeze"));
1401
2899
  sym_on_load = ID2SYM(rb_intern("on_load"));
1402
2900
  sym_decimal_class = ID2SYM(rb_intern("decimal_class"));
2901
+ sym_allow_duplicate_key = ID2SYM(rb_intern("allow_duplicate_key"));
1403
2902
 
1404
- i_chr = rb_intern("chr");
1405
- i_aset = rb_intern("[]=");
1406
- i_aref = rb_intern("[]");
1407
- i_leftshift = rb_intern("<<");
1408
2903
  i_new = rb_intern("new");
1409
2904
  i_try_convert = rb_intern("try_convert");
2905
+ #ifndef HAVE_RB_STR_TO_INTERNED_STR
1410
2906
  i_uminus = rb_intern("-@");
2907
+ #endif
1411
2908
  i_encode = rb_intern("encode");
2909
+ i_at_line = rb_intern("@line");
2910
+ i_at_column = rb_intern("@column");
2911
+ i_at_json_path = rb_intern("@json_path");
1412
2912
 
1413
2913
  binary_encindex = rb_ascii8bit_encindex();
1414
2914
  utf8_encindex = rb_utf8_encindex();
1415
2915
  enc_utf8 = rb_utf8_encoding();
2916
+
2917
+ #ifdef HAVE_SIMD
2918
+ simd_impl = find_simd_implementation();
2919
+ #endif
1416
2920
  }