corvus_json_schema 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +7 -0
- data/Cargo.lock +390 -0
- data/Cargo.toml +8 -0
- data/LICENSE +201 -0
- data/README.md +111 -0
- data/VERSIONHISTORY.md +7 -0
- data/ext/corvus_json_schema/Cargo.toml +18 -0
- data/ext/corvus_json_schema/extconf.rb +6 -0
- data/ext/corvus_json_schema/rustfmt.toml +2 -0
- data/ext/corvus_json_schema/src/lib.rs +643 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/Cargo.toml +35 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/LICENSE +201 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/README.md +139 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/compiler.rs +1118 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/dialect.rs +108 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/document.rs +905 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/eval/plan/fused.rs +1187 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/eval/plan.rs +1728 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/eval.rs +1451 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/formats.rs +1021 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/instance.rs +228 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/lib.rs +189 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/loader.rs +403 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/metaschemas.rs +29 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/node.rs +417 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/numbers.rs +244 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/options.rs +108 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/pattern.rs +1591 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/results.rs +316 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/uri.rs +274 -0
- data/lib/corvus_json_schema/version.rb +5 -0
- data/lib/corvus_json_schema.rb +65 -0
- metadata +94 -0
|
@@ -0,0 +1,905 @@
|
|
|
1
|
+
//! A parsed JSON document that the evaluator reads in place: JSON text parsed once into one flat array of nodes, with
|
|
2
|
+
//! strings borrowed from the text where they have no escapes. Parsing allocates a few buffers per document, not one
|
|
3
|
+
//! per value as `serde_json::Value` does, so it is several times faster to build.
|
|
4
|
+
//!
|
|
5
|
+
//! ```
|
|
6
|
+
//! let validator = corvus_json_schema::compile(&serde_json::json!({ "type": "array", "items": { "type": "integer" } }))
|
|
7
|
+
//! .unwrap();
|
|
8
|
+
//! let document = corvus_json_schema::JsonDocument::parse("[1, 2, 3]").unwrap();
|
|
9
|
+
//! assert!(validator.validate_instance(document.root()).unwrap());
|
|
10
|
+
//! ```
|
|
11
|
+
|
|
12
|
+
use std::cell::Cell;
|
|
13
|
+
use std::fmt;
|
|
14
|
+
|
|
15
|
+
use serde_json::{Map, Number, Value};
|
|
16
|
+
|
|
17
|
+
use crate::instance::{ArrayView, Instance, Kind, ObjectView, View, str_eq};
|
|
18
|
+
|
|
19
|
+
/// The deepest nesting of arrays and objects accepted: serde_json's default recursion limit admits 127 levels.
|
|
20
|
+
const MAX_DEPTH: usize = 127;
|
|
21
|
+
|
|
22
|
+
/// A value: its kind, flags (the number representation, or where a string's bytes are), a count (array items, object
|
|
23
|
+
/// properties) or byte length (strings), and its data (a number's bits, a string's offset, the index of a
|
|
24
|
+
/// container's first child). The children of an array or object are consecutive; an object's are key and value
|
|
25
|
+
/// pairs.
|
|
26
|
+
#[derive(Clone, Copy)]
|
|
27
|
+
struct Node {
|
|
28
|
+
kind: Kind,
|
|
29
|
+
flags: u8,
|
|
30
|
+
len: u32,
|
|
31
|
+
data: u64,
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
const NUM_U64: u8 = 0;
|
|
35
|
+
const NUM_I64: u8 = 1;
|
|
36
|
+
const NUM_F64: u8 = 2;
|
|
37
|
+
|
|
38
|
+
/// The string's bytes are in the source text.
|
|
39
|
+
const STR_SOURCE: u8 = 0;
|
|
40
|
+
/// The string's bytes are in the document's buffer of unescaped strings.
|
|
41
|
+
const STR_TEXT: u8 = 1;
|
|
42
|
+
|
|
43
|
+
impl Node {
|
|
44
|
+
#[inline(always)]
|
|
45
|
+
const fn new(kind: Kind, flags: u8, len: u32, data: u64) -> Node {
|
|
46
|
+
Node { kind, flags, len, data }
|
|
47
|
+
}
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
/// JSON text parsed for evaluation. It borrows the text, for the strings that need no unescaping.
|
|
51
|
+
pub struct JsonDocument<'s> {
|
|
52
|
+
source: &'s str,
|
|
53
|
+
/// Exactly sized for a parsed document; the parser's own buffer for one lent by `with_document`.
|
|
54
|
+
nodes: Vec<Node>,
|
|
55
|
+
root: Node,
|
|
56
|
+
/// The unescaped strings.
|
|
57
|
+
text: String,
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
/// The text is not valid JSON, or nests arrays and objects too deeply.
|
|
61
|
+
#[derive(Clone, Debug, PartialEq, Eq)]
|
|
62
|
+
pub struct JsonParseError {
|
|
63
|
+
offset: usize,
|
|
64
|
+
message: &'static str,
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
impl JsonParseError {
|
|
68
|
+
/// The byte offset in the text at which the error was found.
|
|
69
|
+
pub fn offset(&self) -> usize {
|
|
70
|
+
self.offset
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
/// What is wrong.
|
|
74
|
+
pub fn message(&self) -> &str {
|
|
75
|
+
self.message
|
|
76
|
+
}
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
impl fmt::Display for JsonParseError {
|
|
80
|
+
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
|
|
81
|
+
write!(f, "{} at byte {}", self.message, self.offset)
|
|
82
|
+
}
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
impl std::error::Error for JsonParseError {}
|
|
86
|
+
|
|
87
|
+
impl<'s> JsonDocument<'s> {
|
|
88
|
+
/// Parses JSON text. As serde_json does, it rejects lone surrogates in `\u` escapes, numbers out of the range of a
|
|
89
|
+
/// double, nesting deeper than 127 levels and anything but whitespace after the value; of duplicate property
|
|
90
|
+
/// names, the last value is kept, at the position of the first.
|
|
91
|
+
pub fn parse(source: &'s str) -> Result<JsonDocument<'s>, JsonParseError> {
|
|
92
|
+
let mut parser = Parser::new(source, BUFFERS.take().unwrap_or_default());
|
|
93
|
+
let root = parser.parse();
|
|
94
|
+
// One allocation for the document's nodes, exactly sized.
|
|
95
|
+
let document = root.map(|root| JsonDocument {
|
|
96
|
+
source,
|
|
97
|
+
nodes: parser.nodes.as_slice().to_vec(),
|
|
98
|
+
root,
|
|
99
|
+
text: parser.text.as_str().to_owned(),
|
|
100
|
+
});
|
|
101
|
+
let Parser { nodes, scratch, frames, text, .. } = parser;
|
|
102
|
+
release(Buffers { nodes, scratch, frames, text });
|
|
103
|
+
document
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
/// Parses JSON text into the thread's reused buffers and calls `f` with the document, without the allocation
|
|
107
|
+
/// `parse` makes for a document that outlives the call: a validation of JSON text allocates nothing in the
|
|
108
|
+
/// steady state. A call made from within `f` (a format callback validating JSON itself) gets buffers of its own.
|
|
109
|
+
pub(crate) fn with_document<R>(source: &str, f: impl FnOnce(&JsonDocument<'_>) -> R) -> Result<R, JsonParseError> {
|
|
110
|
+
let mut parser = Parser::new(source, BUFFERS.take().unwrap_or_default());
|
|
111
|
+
let root = parser.parse();
|
|
112
|
+
let Parser { nodes, scratch, frames, text, .. } = parser;
|
|
113
|
+
let (nodes, text, out) = match root {
|
|
114
|
+
Ok(root) => {
|
|
115
|
+
let document = JsonDocument { source, nodes, root, text };
|
|
116
|
+
let out = f(&document);
|
|
117
|
+
(document.nodes, document.text, Ok(out))
|
|
118
|
+
}
|
|
119
|
+
Err(e) => (nodes, text, Err(e)),
|
|
120
|
+
};
|
|
121
|
+
release(Buffers { nodes, scratch, frames, text });
|
|
122
|
+
out
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
/// The root value.
|
|
126
|
+
#[inline]
|
|
127
|
+
pub fn root(&self) -> JsonDocumentValue<'_> {
|
|
128
|
+
JsonDocumentValue { doc: self.erase(), node: &self.root }
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
/// The document as a `serde_json::Value`.
|
|
132
|
+
pub fn to_value(&self) -> Value {
|
|
133
|
+
self.root().to_value()
|
|
134
|
+
}
|
|
135
|
+
|
|
136
|
+
/// The document with its source lifetime shortened to the borrow (`JsonDocument` is covariant in it).
|
|
137
|
+
#[inline(always)]
|
|
138
|
+
fn erase<'d>(&'d self) -> &'d JsonDocument<'d> {
|
|
139
|
+
self
|
|
140
|
+
}
|
|
141
|
+
|
|
142
|
+
#[inline(always)]
|
|
143
|
+
fn str_of(&self, n: &Node) -> &str {
|
|
144
|
+
let (offset, len) = (n.data as usize, n.len as usize);
|
|
145
|
+
let buffer = if n.flags == STR_TEXT { self.text.as_str() } else { self.source };
|
|
146
|
+
debug_assert!(buffer.is_char_boundary(offset) && buffer.is_char_boundary(offset + len));
|
|
147
|
+
// SAFETY: the parser records string nodes at character boundaries within their buffer.
|
|
148
|
+
unsafe { buffer.get_unchecked(offset..offset + len) }
|
|
149
|
+
}
|
|
150
|
+
|
|
151
|
+
#[inline(always)]
|
|
152
|
+
fn children(&self, n: &Node, count: usize) -> &[Node] {
|
|
153
|
+
let start = n.data as usize;
|
|
154
|
+
debug_assert!(start + count <= self.nodes.len());
|
|
155
|
+
// SAFETY: the parser records a container's children as a range within `nodes`.
|
|
156
|
+
unsafe { self.nodes.get_unchecked(start..start + count) }
|
|
157
|
+
}
|
|
158
|
+
|
|
159
|
+
/// An object's key and value nodes, as pairs (iterating pairs, not two-node chunks, needs no per-step checks).
|
|
160
|
+
#[inline(always)]
|
|
161
|
+
fn pairs(&self, n: &Node) -> &[[Node; 2]] {
|
|
162
|
+
let items = self.children(n, 2 * n.len as usize);
|
|
163
|
+
// SAFETY: `[Node; 2]` has the layout of two consecutive `Node`s, and `items` holds exactly `n.len` pairs.
|
|
164
|
+
unsafe { std::slice::from_raw_parts(items.as_ptr().cast::<[Node; 2]>(), n.len as usize) }
|
|
165
|
+
}
|
|
166
|
+
}
|
|
167
|
+
|
|
168
|
+
impl fmt::Debug for JsonDocument<'_> {
|
|
169
|
+
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
|
|
170
|
+
f.debug_struct("JsonDocument").field("nodes", &self.nodes.len()).finish()
|
|
171
|
+
}
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
/// A value in a [`JsonDocument`]: an [`Instance`] for [`crate::Validator::validate_instance`].
|
|
175
|
+
#[derive(Clone, Copy)]
|
|
176
|
+
pub struct JsonDocumentValue<'d> {
|
|
177
|
+
doc: &'d JsonDocument<'d>,
|
|
178
|
+
node: &'d Node,
|
|
179
|
+
}
|
|
180
|
+
|
|
181
|
+
/// The items of an array in a [`JsonDocument`].
|
|
182
|
+
#[derive(Clone, Copy)]
|
|
183
|
+
pub struct JsonDocumentArray<'d> {
|
|
184
|
+
doc: &'d JsonDocument<'d>,
|
|
185
|
+
items: &'d [Node],
|
|
186
|
+
}
|
|
187
|
+
|
|
188
|
+
/// The properties of an object in a [`JsonDocument`].
|
|
189
|
+
#[derive(Clone, Copy)]
|
|
190
|
+
pub struct JsonDocumentObject<'d> {
|
|
191
|
+
doc: &'d JsonDocument<'d>,
|
|
192
|
+
/// Key and value nodes.
|
|
193
|
+
pairs: &'d [[Node; 2]],
|
|
194
|
+
}
|
|
195
|
+
|
|
196
|
+
impl<'d> JsonDocumentValue<'d> {
|
|
197
|
+
#[inline(always)]
|
|
198
|
+
fn number(n: &Node) -> Number {
|
|
199
|
+
match n.flags {
|
|
200
|
+
NUM_U64 => Number::from(n.data),
|
|
201
|
+
NUM_I64 => Number::from(n.data as i64),
|
|
202
|
+
// The parser only records finite doubles.
|
|
203
|
+
_ => Number::from_f64(f64::from_bits(n.data)).unwrap_or_else(|| Number::from(0)),
|
|
204
|
+
}
|
|
205
|
+
}
|
|
206
|
+
|
|
207
|
+
/// The value as a `serde_json::Value`.
|
|
208
|
+
pub fn to_value(self) -> Value {
|
|
209
|
+
match self.view() {
|
|
210
|
+
View::Null => Value::Null,
|
|
211
|
+
View::Bool(b) => Value::Bool(b),
|
|
212
|
+
View::Number(n) => Value::Number(n),
|
|
213
|
+
View::String(s) => Value::String(s.to_owned()),
|
|
214
|
+
View::Array(a) => Value::Array(a.iter().map(JsonDocumentValue::to_value).collect()),
|
|
215
|
+
View::Object(o) => {
|
|
216
|
+
let mut map = Map::with_capacity(o.len());
|
|
217
|
+
for (k, v) in o.iter() {
|
|
218
|
+
map.insert(k.to_owned(), v.to_value());
|
|
219
|
+
}
|
|
220
|
+
Value::Object(map)
|
|
221
|
+
}
|
|
222
|
+
}
|
|
223
|
+
}
|
|
224
|
+
}
|
|
225
|
+
|
|
226
|
+
impl<'d> Instance<'d> for JsonDocumentValue<'d> {
|
|
227
|
+
type Array = JsonDocumentArray<'d>;
|
|
228
|
+
type Object = JsonDocumentObject<'d>;
|
|
229
|
+
|
|
230
|
+
#[inline(always)]
|
|
231
|
+
fn view(self) -> View<'d, Self> {
|
|
232
|
+
let (doc, n) = (self.doc, self.node);
|
|
233
|
+
match n.kind {
|
|
234
|
+
Kind::Null => View::Null,
|
|
235
|
+
Kind::Bool => View::Bool(n.data != 0),
|
|
236
|
+
Kind::Number => View::Number(Self::number(n)),
|
|
237
|
+
Kind::String => View::String(doc.str_of(n)),
|
|
238
|
+
Kind::Array => View::Array(JsonDocumentArray { doc, items: doc.children(n, n.len as usize) }),
|
|
239
|
+
Kind::Object => View::Object(JsonDocumentObject { doc, pairs: doc.pairs(n) }),
|
|
240
|
+
}
|
|
241
|
+
}
|
|
242
|
+
|
|
243
|
+
#[inline(always)]
|
|
244
|
+
fn kind(self) -> Kind {
|
|
245
|
+
self.node.kind
|
|
246
|
+
}
|
|
247
|
+
}
|
|
248
|
+
|
|
249
|
+
impl<'d> ArrayView<'d> for JsonDocumentArray<'d> {
|
|
250
|
+
type Item = JsonDocumentValue<'d>;
|
|
251
|
+
|
|
252
|
+
#[inline(always)]
|
|
253
|
+
fn len(self) -> usize {
|
|
254
|
+
self.items.len()
|
|
255
|
+
}
|
|
256
|
+
|
|
257
|
+
#[inline(always)]
|
|
258
|
+
fn get(self, index: usize) -> JsonDocumentValue<'d> {
|
|
259
|
+
JsonDocumentValue { doc: self.doc, node: &self.items[index] }
|
|
260
|
+
}
|
|
261
|
+
|
|
262
|
+
#[inline(always)]
|
|
263
|
+
fn iter(self) -> impl Iterator<Item = JsonDocumentValue<'d>> {
|
|
264
|
+
let doc = self.doc;
|
|
265
|
+
self.items.iter().map(move |node| JsonDocumentValue { doc, node })
|
|
266
|
+
}
|
|
267
|
+
|
|
268
|
+
#[inline(always)]
|
|
269
|
+
fn tail(self, start: usize) -> impl Iterator<Item = JsonDocumentValue<'d>> {
|
|
270
|
+
let doc = self.doc;
|
|
271
|
+
self.items.get(start..).unwrap_or_default().iter().map(move |node| JsonDocumentValue { doc, node })
|
|
272
|
+
}
|
|
273
|
+
}
|
|
274
|
+
|
|
275
|
+
impl<'d> ObjectView<'d> for JsonDocumentObject<'d> {
|
|
276
|
+
type Item = JsonDocumentValue<'d>;
|
|
277
|
+
|
|
278
|
+
#[inline(always)]
|
|
279
|
+
fn len(self) -> usize {
|
|
280
|
+
self.pairs.len()
|
|
281
|
+
}
|
|
282
|
+
|
|
283
|
+
#[inline]
|
|
284
|
+
fn get(self, name: &str) -> Option<JsonDocumentValue<'d>> {
|
|
285
|
+
let doc = self.doc;
|
|
286
|
+
self.pairs.iter().find(|[k, _]| str_eq(doc.str_of(k), name)).map(|[_, v]| JsonDocumentValue { doc, node: v })
|
|
287
|
+
}
|
|
288
|
+
|
|
289
|
+
#[inline]
|
|
290
|
+
fn contains_key(self, name: &str) -> bool {
|
|
291
|
+
let doc = self.doc;
|
|
292
|
+
self.pairs.iter().any(|[k, _]| str_eq(doc.str_of(k), name))
|
|
293
|
+
}
|
|
294
|
+
|
|
295
|
+
#[inline(always)]
|
|
296
|
+
fn iter(self) -> impl Iterator<Item = (&'d str, JsonDocumentValue<'d>)> {
|
|
297
|
+
let doc = self.doc;
|
|
298
|
+
self.pairs.iter().map(move |[k, v]| (doc.str_of(k), JsonDocumentValue { doc, node: v }))
|
|
299
|
+
}
|
|
300
|
+
|
|
301
|
+
#[inline(always)]
|
|
302
|
+
fn values(self) -> impl Iterator<Item = JsonDocumentValue<'d>> {
|
|
303
|
+
let doc = self.doc;
|
|
304
|
+
self.pairs.iter().map(move |[_, v]| JsonDocumentValue { doc, node: v })
|
|
305
|
+
}
|
|
306
|
+
}
|
|
307
|
+
|
|
308
|
+
/// The parser's working buffers, kept between parses on a thread so that a document costs one allocation (two when
|
|
309
|
+
/// strings need unescaping): an allocator's cost per call, which musl's makes high, otherwise dominates small
|
|
310
|
+
/// documents.
|
|
311
|
+
#[derive(Default)]
|
|
312
|
+
struct Buffers {
|
|
313
|
+
nodes: Vec<Node>,
|
|
314
|
+
scratch: Vec<Node>,
|
|
315
|
+
frames: Vec<Frame>,
|
|
316
|
+
text: String,
|
|
317
|
+
}
|
|
318
|
+
|
|
319
|
+
thread_local! {
|
|
320
|
+
/// Taken for the length of a parse, so a parse that a parse somehow re-entered would start with new buffers.
|
|
321
|
+
static BUFFERS: Cell<Option<Buffers>> = const { Cell::new(None) };
|
|
322
|
+
}
|
|
323
|
+
|
|
324
|
+
/// Returns the parser's buffers to the thread for the next parse, unless they grew too large to keep.
|
|
325
|
+
fn release(mut b: Buffers) {
|
|
326
|
+
if b.nodes.capacity() <= MAX_KEPT_NODES && b.text.capacity() <= MAX_KEPT_TEXT {
|
|
327
|
+
b.nodes.clear();
|
|
328
|
+
b.scratch.clear();
|
|
329
|
+
b.frames.clear();
|
|
330
|
+
b.text.clear();
|
|
331
|
+
BUFFERS.set(Some(b));
|
|
332
|
+
}
|
|
333
|
+
}
|
|
334
|
+
|
|
335
|
+
/// Buffers that grew beyond this many nodes (16 MiB) for a large document are released, not kept.
|
|
336
|
+
const MAX_KEPT_NODES: usize = 1 << 20;
|
|
337
|
+
/// Nor is an unescaped-string buffer beyond 16 MiB.
|
|
338
|
+
const MAX_KEPT_TEXT: usize = 1 << 24;
|
|
339
|
+
|
|
340
|
+
/// An open array or object: where its children start in the scratch stack.
|
|
341
|
+
#[derive(Clone, Copy)]
|
|
342
|
+
struct Frame {
|
|
343
|
+
start: u32,
|
|
344
|
+
object: bool,
|
|
345
|
+
}
|
|
346
|
+
|
|
347
|
+
struct Parser<'s> {
|
|
348
|
+
source: &'s str,
|
|
349
|
+
b: &'s [u8],
|
|
350
|
+
i: usize,
|
|
351
|
+
/// Finished children of closed containers, each container's consecutive.
|
|
352
|
+
nodes: Vec<Node>,
|
|
353
|
+
/// The values of the open containers, innermost last; a container's run moves to `nodes` when it closes.
|
|
354
|
+
scratch: Vec<Node>,
|
|
355
|
+
frames: Vec<Frame>,
|
|
356
|
+
text: String,
|
|
357
|
+
}
|
|
358
|
+
|
|
359
|
+
/// Bytes equal to `n` in a word (exact at the lowest set bit, which is all the scanner uses).
|
|
360
|
+
#[inline(always)]
|
|
361
|
+
const fn eq_bytes(w: u64, n: u8) -> u64 {
|
|
362
|
+
let x = w ^ (0x0101_0101_0101_0101 * n as u64);
|
|
363
|
+
x.wrapping_sub(0x0101_0101_0101_0101) & !x & 0x8080_8080_8080_8080
|
|
364
|
+
}
|
|
365
|
+
|
|
366
|
+
/// Bytes below 0x20 in a word (exact at the lowest set bit).
|
|
367
|
+
#[inline(always)]
|
|
368
|
+
const fn control_bytes(w: u64) -> u64 {
|
|
369
|
+
w.wrapping_sub(0x2020_2020_2020_2020) & !w & 0x8080_8080_8080_8080
|
|
370
|
+
}
|
|
371
|
+
|
|
372
|
+
impl<'s> Parser<'s> {
|
|
373
|
+
fn new(source: &'s str, buffers: Buffers) -> Parser<'s> {
|
|
374
|
+
Parser {
|
|
375
|
+
source,
|
|
376
|
+
b: source.as_bytes(),
|
|
377
|
+
i: 0,
|
|
378
|
+
nodes: buffers.nodes,
|
|
379
|
+
scratch: buffers.scratch,
|
|
380
|
+
frames: buffers.frames,
|
|
381
|
+
text: buffers.text,
|
|
382
|
+
}
|
|
383
|
+
}
|
|
384
|
+
|
|
385
|
+
#[cold]
|
|
386
|
+
fn error<T>(&self, message: &'static str) -> Result<T, JsonParseError> {
|
|
387
|
+
Err(JsonParseError { offset: self.i, message })
|
|
388
|
+
}
|
|
389
|
+
|
|
390
|
+
#[inline(always)]
|
|
391
|
+
fn skip_ws(&mut self) {
|
|
392
|
+
while let Some(&c) = self.b.get(self.i) {
|
|
393
|
+
if !matches!(c, b' ' | b'\n' | b'\r' | b'\t') {
|
|
394
|
+
break;
|
|
395
|
+
}
|
|
396
|
+
self.i += 1;
|
|
397
|
+
}
|
|
398
|
+
}
|
|
399
|
+
|
|
400
|
+
#[inline(always)]
|
|
401
|
+
fn peek(&self) -> Option<u8> {
|
|
402
|
+
self.b.get(self.i).copied()
|
|
403
|
+
}
|
|
404
|
+
|
|
405
|
+
/// Parses the text into `nodes`, returning the root.
|
|
406
|
+
fn parse(&mut self) -> Result<Node, JsonParseError> {
|
|
407
|
+
if self.source.len() > u32::MAX as usize {
|
|
408
|
+
return self.error("document too large");
|
|
409
|
+
}
|
|
410
|
+
self.skip_ws();
|
|
411
|
+
'value: loop {
|
|
412
|
+
// A value.
|
|
413
|
+
match self.peek() {
|
|
414
|
+
Some(b'{') => {
|
|
415
|
+
self.i += 1;
|
|
416
|
+
self.skip_ws();
|
|
417
|
+
if self.peek() == Some(b'}') {
|
|
418
|
+
self.i += 1;
|
|
419
|
+
self.check_depth()?;
|
|
420
|
+
self.scratch.push(Node::new(Kind::Object, 0, 0, 0));
|
|
421
|
+
} else {
|
|
422
|
+
self.open(true)?;
|
|
423
|
+
self.key()?;
|
|
424
|
+
continue 'value;
|
|
425
|
+
}
|
|
426
|
+
}
|
|
427
|
+
Some(b'[') => {
|
|
428
|
+
self.i += 1;
|
|
429
|
+
self.skip_ws();
|
|
430
|
+
if self.peek() == Some(b']') {
|
|
431
|
+
self.i += 1;
|
|
432
|
+
self.check_depth()?;
|
|
433
|
+
self.scratch.push(Node::new(Kind::Array, 0, 0, 0));
|
|
434
|
+
} else {
|
|
435
|
+
self.open(false)?;
|
|
436
|
+
continue 'value;
|
|
437
|
+
}
|
|
438
|
+
}
|
|
439
|
+
Some(b'"') => {
|
|
440
|
+
let n = self.string()?;
|
|
441
|
+
self.scratch.push(n);
|
|
442
|
+
}
|
|
443
|
+
Some(b't') => self.literal(b"true", Node::new(Kind::Bool, 0, 0, 1))?,
|
|
444
|
+
Some(b'f') => self.literal(b"false", Node::new(Kind::Bool, 0, 0, 0))?,
|
|
445
|
+
Some(b'n') => self.literal(b"null", Node::new(Kind::Null, 0, 0, 0))?,
|
|
446
|
+
Some(b'-' | b'0'..=b'9') => {
|
|
447
|
+
let n = self.number()?;
|
|
448
|
+
self.scratch.push(n);
|
|
449
|
+
}
|
|
450
|
+
Some(_) => return self.error("expected a value"),
|
|
451
|
+
None => return self.error("unexpected end of input"),
|
|
452
|
+
}
|
|
453
|
+
// After a value: separators and closing brackets, until the next value or the end.
|
|
454
|
+
loop {
|
|
455
|
+
let Some(&frame) = self.frames.last() else {
|
|
456
|
+
self.skip_ws();
|
|
457
|
+
if self.i != self.b.len() {
|
|
458
|
+
return self.error("trailing characters");
|
|
459
|
+
}
|
|
460
|
+
return Ok(self.scratch[0]);
|
|
461
|
+
};
|
|
462
|
+
self.skip_ws();
|
|
463
|
+
match self.peek() {
|
|
464
|
+
Some(b',') => {
|
|
465
|
+
self.i += 1;
|
|
466
|
+
self.skip_ws();
|
|
467
|
+
if frame.object {
|
|
468
|
+
self.key()?;
|
|
469
|
+
}
|
|
470
|
+
continue 'value;
|
|
471
|
+
}
|
|
472
|
+
Some(b'}') if frame.object => {
|
|
473
|
+
self.i += 1;
|
|
474
|
+
self.close(frame);
|
|
475
|
+
}
|
|
476
|
+
Some(b']') if !frame.object => {
|
|
477
|
+
self.i += 1;
|
|
478
|
+
self.close(frame);
|
|
479
|
+
}
|
|
480
|
+
Some(_) => {
|
|
481
|
+
return self.error(if frame.object { "expected ',' or '}'" } else { "expected ',' or ']'" });
|
|
482
|
+
}
|
|
483
|
+
None => return self.error("unexpected end of input"),
|
|
484
|
+
}
|
|
485
|
+
}
|
|
486
|
+
}
|
|
487
|
+
}
|
|
488
|
+
|
|
489
|
+
#[inline(always)]
|
|
490
|
+
fn open(&mut self, object: bool) -> Result<(), JsonParseError> {
|
|
491
|
+
self.check_depth()?;
|
|
492
|
+
self.frames.push(Frame { start: self.scratch.len() as u32, object });
|
|
493
|
+
Ok(())
|
|
494
|
+
}
|
|
495
|
+
|
|
496
|
+
/// An array or object may open at the current depth (an empty one counts, as in serde_json).
|
|
497
|
+
#[inline(always)]
|
|
498
|
+
fn check_depth(&self) -> Result<(), JsonParseError> {
|
|
499
|
+
if self.frames.len() >= MAX_DEPTH { self.error("recursion limit exceeded") } else { Ok(()) }
|
|
500
|
+
}
|
|
501
|
+
|
|
502
|
+
/// A property name and its colon, leaving the parser at the value.
|
|
503
|
+
#[inline(always)]
|
|
504
|
+
fn key(&mut self) -> Result<(), JsonParseError> {
|
|
505
|
+
if self.peek() != Some(b'"') {
|
|
506
|
+
return self.error("expected a property name");
|
|
507
|
+
}
|
|
508
|
+
let k = self.string()?;
|
|
509
|
+
self.scratch.push(k);
|
|
510
|
+
self.skip_ws();
|
|
511
|
+
if self.peek() != Some(b':') {
|
|
512
|
+
return self.error("expected ':'");
|
|
513
|
+
}
|
|
514
|
+
self.i += 1;
|
|
515
|
+
self.skip_ws();
|
|
516
|
+
Ok(())
|
|
517
|
+
}
|
|
518
|
+
|
|
519
|
+
/// Moves the closed container's children to `nodes` and pushes the container in their place.
|
|
520
|
+
#[inline(always)]
|
|
521
|
+
fn close(&mut self, frame: Frame) {
|
|
522
|
+
self.frames.pop();
|
|
523
|
+
let start = frame.start as usize;
|
|
524
|
+
if frame.object && self.scratch.len() - start > 2 {
|
|
525
|
+
self.dedupe(start);
|
|
526
|
+
}
|
|
527
|
+
let first = self.nodes.len() as u64;
|
|
528
|
+
let children = self.scratch.len() - start;
|
|
529
|
+
self.nodes.extend_from_slice(&self.scratch[start..]);
|
|
530
|
+
self.scratch.truncate(start);
|
|
531
|
+
let (kind, count) = if frame.object { (Kind::Object, children / 2) } else { (Kind::Array, children) };
|
|
532
|
+
self.scratch.push(Node::new(kind, 0, count as u32, first));
|
|
533
|
+
}
|
|
534
|
+
|
|
535
|
+
fn key_str(&self, n: &Node) -> &str {
|
|
536
|
+
let (offset, len) = (n.data as usize, n.len as usize);
|
|
537
|
+
if n.flags == STR_TEXT { &self.text[offset..offset + len] } else { &self.source[offset..offset + len] }
|
|
538
|
+
}
|
|
539
|
+
|
|
540
|
+
/// Of duplicate property names in the object whose pairs start at `start`, keeps the last value at the first
|
|
541
|
+
/// position (as `serde_json::Map` with `preserve_order` does).
|
|
542
|
+
fn dedupe(&mut self, start: usize) {
|
|
543
|
+
let pairs = &self.scratch[start..];
|
|
544
|
+
let count = pairs.len() / 2;
|
|
545
|
+
let duplicate = if count <= 16 {
|
|
546
|
+
(1..count).any(|j| (0..j).any(|i| str_eq(self.key_str(&pairs[2 * i]), self.key_str(&pairs[2 * j]))))
|
|
547
|
+
} else {
|
|
548
|
+
let mut seen = std::collections::HashSet::with_capacity(count);
|
|
549
|
+
(0..count).any(|j| !seen.insert(self.key_str(&pairs[2 * j])))
|
|
550
|
+
};
|
|
551
|
+
if !duplicate {
|
|
552
|
+
return;
|
|
553
|
+
}
|
|
554
|
+
let mut kept: Vec<Node> = Vec::with_capacity(pairs.len());
|
|
555
|
+
for p in pairs.chunks_exact(2) {
|
|
556
|
+
let name = self.key_str(&p[0]);
|
|
557
|
+
match kept.chunks_exact(2).position(|q| self.key_str(&q[0]) == name) {
|
|
558
|
+
Some(at) => kept[2 * at + 1] = p[1],
|
|
559
|
+
None => kept.extend_from_slice(p),
|
|
560
|
+
}
|
|
561
|
+
}
|
|
562
|
+
self.scratch.truncate(start);
|
|
563
|
+
self.scratch.extend_from_slice(&kept);
|
|
564
|
+
}
|
|
565
|
+
|
|
566
|
+
#[inline(always)]
|
|
567
|
+
fn literal(&mut self, word: &[u8], node: Node) -> Result<(), JsonParseError> {
|
|
568
|
+
if self.b.get(self.i..self.i + word.len()) != Some(word) {
|
|
569
|
+
return self.error("expected a value");
|
|
570
|
+
}
|
|
571
|
+
self.i += word.len();
|
|
572
|
+
self.scratch.push(node);
|
|
573
|
+
Ok(())
|
|
574
|
+
}
|
|
575
|
+
|
|
576
|
+
/// A string, from its opening quote.
|
|
577
|
+
#[inline(always)]
|
|
578
|
+
fn string(&mut self) -> Result<Node, JsonParseError> {
|
|
579
|
+
let start = self.i + 1;
|
|
580
|
+
let mut j = start;
|
|
581
|
+
let b = self.b;
|
|
582
|
+
// Eight bytes at a time to the first quote, backslash or control character.
|
|
583
|
+
while j + 8 <= b.len() {
|
|
584
|
+
let w = u64::from_le_bytes(b[j..j + 8].try_into().unwrap());
|
|
585
|
+
let special = eq_bytes(w, b'"') | eq_bytes(w, b'\\') | control_bytes(w);
|
|
586
|
+
if special != 0 {
|
|
587
|
+
j += (special.trailing_zeros() / 8) as usize;
|
|
588
|
+
return self.string_at(start, j);
|
|
589
|
+
}
|
|
590
|
+
j += 8;
|
|
591
|
+
}
|
|
592
|
+
while j < b.len() && !matches!(b[j], b'"' | b'\\' | 0..0x20) {
|
|
593
|
+
j += 1;
|
|
594
|
+
}
|
|
595
|
+
self.string_at(start, j)
|
|
596
|
+
}
|
|
597
|
+
|
|
598
|
+
#[inline(always)]
|
|
599
|
+
fn string_at(&mut self, start: usize, j: usize) -> Result<Node, JsonParseError> {
|
|
600
|
+
match self.b.get(j) {
|
|
601
|
+
Some(b'"') => {
|
|
602
|
+
self.i = j + 1;
|
|
603
|
+
Ok(Node::new(Kind::String, STR_SOURCE, (j - start) as u32, start as u64))
|
|
604
|
+
}
|
|
605
|
+
Some(b'\\') => self.escaped(start, j),
|
|
606
|
+
Some(_) => {
|
|
607
|
+
self.i = j;
|
|
608
|
+
self.error("control character in a string")
|
|
609
|
+
}
|
|
610
|
+
None => {
|
|
611
|
+
self.i = j;
|
|
612
|
+
self.error("unterminated string")
|
|
613
|
+
}
|
|
614
|
+
}
|
|
615
|
+
}
|
|
616
|
+
|
|
617
|
+
/// The rest of a string with escapes, unescaped into the text buffer: `j` is at the first backslash.
|
|
618
|
+
#[cold]
|
|
619
|
+
fn escaped(&mut self, start: usize, mut j: usize) -> Result<Node, JsonParseError> {
|
|
620
|
+
let offset = self.text.len();
|
|
621
|
+
let b = self.b;
|
|
622
|
+
let mut run = start;
|
|
623
|
+
loop {
|
|
624
|
+
match b.get(j) {
|
|
625
|
+
Some(b'"') => {
|
|
626
|
+
self.text.push_str(&self.source[run..j]);
|
|
627
|
+
self.i = j + 1;
|
|
628
|
+
let len = self.text.len() - offset;
|
|
629
|
+
return Ok(Node::new(Kind::String, STR_TEXT, len as u32, offset as u64));
|
|
630
|
+
}
|
|
631
|
+
Some(b'\\') => {
|
|
632
|
+
self.text.push_str(&self.source[run..j]);
|
|
633
|
+
let c = match b.get(j + 1) {
|
|
634
|
+
Some(b'"') => '"',
|
|
635
|
+
Some(b'\\') => '\\',
|
|
636
|
+
Some(b'/') => '/',
|
|
637
|
+
Some(b'b') => '\u{8}',
|
|
638
|
+
Some(b'f') => '\u{c}',
|
|
639
|
+
Some(b'n') => '\n',
|
|
640
|
+
Some(b'r') => '\r',
|
|
641
|
+
Some(b't') => '\t',
|
|
642
|
+
Some(b'u') => {
|
|
643
|
+
let (c, next) = self.unicode_escape(j)?;
|
|
644
|
+
self.text.push(c);
|
|
645
|
+
j = next;
|
|
646
|
+
run = j;
|
|
647
|
+
continue;
|
|
648
|
+
}
|
|
649
|
+
_ => {
|
|
650
|
+
self.i = j;
|
|
651
|
+
return self.error("invalid escape");
|
|
652
|
+
}
|
|
653
|
+
};
|
|
654
|
+
self.text.push(c);
|
|
655
|
+
j += 2;
|
|
656
|
+
run = j;
|
|
657
|
+
}
|
|
658
|
+
Some(0..0x20) => {
|
|
659
|
+
self.i = j;
|
|
660
|
+
return self.error("control character in a string");
|
|
661
|
+
}
|
|
662
|
+
Some(_) => j += 1,
|
|
663
|
+
None => {
|
|
664
|
+
self.i = j;
|
|
665
|
+
return self.error("unterminated string");
|
|
666
|
+
}
|
|
667
|
+
}
|
|
668
|
+
}
|
|
669
|
+
}
|
|
670
|
+
|
|
671
|
+
/// A `\u` escape at `j` (and its low surrogate, for a high one): the character and the index after it.
|
|
672
|
+
fn unicode_escape(&mut self, j: usize) -> Result<(char, usize), JsonParseError> {
|
|
673
|
+
let hex = |p: &Self, at: usize| -> Option<u32> {
|
|
674
|
+
let digits = p.b.get(at..at + 4)?;
|
|
675
|
+
let mut v = 0u32;
|
|
676
|
+
for &d in digits {
|
|
677
|
+
v = v * 16 + (d as char).to_digit(16)?;
|
|
678
|
+
}
|
|
679
|
+
Some(v)
|
|
680
|
+
};
|
|
681
|
+
let Some(u) = hex(self, j + 2) else {
|
|
682
|
+
self.i = j;
|
|
683
|
+
return self.error("invalid \\u escape");
|
|
684
|
+
};
|
|
685
|
+
match u {
|
|
686
|
+
0xD800..=0xDBFF => {
|
|
687
|
+
let low = (self.b.get(j + 6) == Some(&b'\\') && self.b.get(j + 7) == Some(&b'u'))
|
|
688
|
+
.then(|| hex(self, j + 8))
|
|
689
|
+
.flatten();
|
|
690
|
+
match low {
|
|
691
|
+
Some(l @ 0xDC00..=0xDFFF) => {
|
|
692
|
+
let c = 0x10000 + ((u - 0xD800) << 10) + (l - 0xDC00);
|
|
693
|
+
Ok((char::from_u32(c).unwrap(), j + 12))
|
|
694
|
+
}
|
|
695
|
+
_ => {
|
|
696
|
+
self.i = j;
|
|
697
|
+
self.error("lone leading surrogate in hex escape")
|
|
698
|
+
}
|
|
699
|
+
}
|
|
700
|
+
}
|
|
701
|
+
0xDC00..=0xDFFF => {
|
|
702
|
+
self.i = j;
|
|
703
|
+
self.error("lone trailing surrogate in hex escape")
|
|
704
|
+
}
|
|
705
|
+
_ => Ok((char::from_u32(u).unwrap(), j + 6)),
|
|
706
|
+
}
|
|
707
|
+
}
|
|
708
|
+
|
|
709
|
+
/// A number: unsigned or negative integers that fit 64 bits as integers, anything else as a double (as serde_json
|
|
710
|
+
/// classifies them).
|
|
711
|
+
#[inline(always)]
|
|
712
|
+
fn number(&mut self) -> Result<Node, JsonParseError> {
|
|
713
|
+
let b = self.b;
|
|
714
|
+
let start = self.i;
|
|
715
|
+
let mut j = start;
|
|
716
|
+
let negative = b[j] == b'-';
|
|
717
|
+
if negative {
|
|
718
|
+
j += 1;
|
|
719
|
+
}
|
|
720
|
+
let mut value: u64 = 0;
|
|
721
|
+
let mut overflow = false;
|
|
722
|
+
match b.get(j) {
|
|
723
|
+
Some(b'0') => j += 1,
|
|
724
|
+
Some(b'1'..=b'9') => {
|
|
725
|
+
while let Some(&d @ b'0'..=b'9') = b.get(j) {
|
|
726
|
+
match value.checked_mul(10).and_then(|v| v.checked_add((d - b'0') as u64)) {
|
|
727
|
+
Some(v) => value = v,
|
|
728
|
+
None => overflow = true,
|
|
729
|
+
}
|
|
730
|
+
j += 1;
|
|
731
|
+
}
|
|
732
|
+
}
|
|
733
|
+
_ => {
|
|
734
|
+
self.i = j;
|
|
735
|
+
return self.error("invalid number");
|
|
736
|
+
}
|
|
737
|
+
}
|
|
738
|
+
let mut float = false;
|
|
739
|
+
if b.get(j) == Some(&b'.') {
|
|
740
|
+
j += 1;
|
|
741
|
+
if !matches!(b.get(j), Some(b'0'..=b'9')) {
|
|
742
|
+
self.i = j;
|
|
743
|
+
return self.error("invalid number");
|
|
744
|
+
}
|
|
745
|
+
while matches!(b.get(j), Some(b'0'..=b'9')) {
|
|
746
|
+
j += 1;
|
|
747
|
+
}
|
|
748
|
+
float = true;
|
|
749
|
+
}
|
|
750
|
+
if matches!(b.get(j), Some(b'e' | b'E')) {
|
|
751
|
+
j += 1;
|
|
752
|
+
if matches!(b.get(j), Some(b'+' | b'-')) {
|
|
753
|
+
j += 1;
|
|
754
|
+
}
|
|
755
|
+
if !matches!(b.get(j), Some(b'0'..=b'9')) {
|
|
756
|
+
self.i = j;
|
|
757
|
+
return self.error("invalid number");
|
|
758
|
+
}
|
|
759
|
+
while matches!(b.get(j), Some(b'0'..=b'9')) {
|
|
760
|
+
j += 1;
|
|
761
|
+
}
|
|
762
|
+
float = true;
|
|
763
|
+
}
|
|
764
|
+
self.i = j;
|
|
765
|
+
if !float && !overflow {
|
|
766
|
+
if !negative {
|
|
767
|
+
return Ok(Node::new(Kind::Number, NUM_U64, 0, value));
|
|
768
|
+
}
|
|
769
|
+
// serde_json reads -0 as the double.
|
|
770
|
+
if value != 0 && value <= i64::MAX as u64 + 1 {
|
|
771
|
+
return Ok(Node::new(Kind::Number, NUM_I64, 0, (value as i64).wrapping_neg() as u64));
|
|
772
|
+
}
|
|
773
|
+
}
|
|
774
|
+
let f: f64 = self.source[start..j].parse().unwrap_or(f64::INFINITY);
|
|
775
|
+
if !f.is_finite() {
|
|
776
|
+
self.i = start;
|
|
777
|
+
return self.error("number out of range");
|
|
778
|
+
}
|
|
779
|
+
Ok(Node::new(Kind::Number, NUM_F64, 0, f.to_bits()))
|
|
780
|
+
}
|
|
781
|
+
}
|
|
782
|
+
|
|
783
|
+
#[cfg(test)]
|
|
784
|
+
mod tests {
|
|
785
|
+
use super::*;
|
|
786
|
+
|
|
787
|
+
fn same(text: &str) {
|
|
788
|
+
let expected: Value = serde_json::from_str(text).unwrap();
|
|
789
|
+
let document = JsonDocument::parse(text).unwrap_or_else(|e| panic!("{text}: {e}"));
|
|
790
|
+
assert_eq!(document.to_value(), expected, "{text}");
|
|
791
|
+
// The number representations match too (serde_json's Value equality compares them).
|
|
792
|
+
assert_eq!(serde_json::to_string(&document.to_value()).unwrap(), serde_json::to_string(&expected).unwrap());
|
|
793
|
+
}
|
|
794
|
+
|
|
795
|
+
fn both_reject(text: &str) {
|
|
796
|
+
assert!(serde_json::from_str::<Value>(text).is_err(), "serde_json accepts {text}");
|
|
797
|
+
assert!(JsonDocument::parse(text).is_err(), "accepted {text}");
|
|
798
|
+
}
|
|
799
|
+
|
|
800
|
+
#[test]
|
|
801
|
+
fn values_match_serde_json() {
|
|
802
|
+
for text in [
|
|
803
|
+
"null",
|
|
804
|
+
" true ",
|
|
805
|
+
"false",
|
|
806
|
+
"0",
|
|
807
|
+
"-0",
|
|
808
|
+
"-0.0",
|
|
809
|
+
"1",
|
|
810
|
+
"-1",
|
|
811
|
+
"18446744073709551615",
|
|
812
|
+
"18446744073709551616",
|
|
813
|
+
"-9223372036854775808",
|
|
814
|
+
"-9223372036854775809",
|
|
815
|
+
"1.5",
|
|
816
|
+
"1e3",
|
|
817
|
+
"1E+3",
|
|
818
|
+
"1.0e-3",
|
|
819
|
+
"123456789012345678901234567890",
|
|
820
|
+
"2.2250738585072014e-308",
|
|
821
|
+
"\"\"",
|
|
822
|
+
"\"plain text, longer than eight bytes\"",
|
|
823
|
+
"\"caf\u{e9} \u{1f600}\"",
|
|
824
|
+
r#""esc \" \\ \/ \b \f \n \r \t é 😀 end""#,
|
|
825
|
+
r#""\u0000""#,
|
|
826
|
+
"[]",
|
|
827
|
+
"{}",
|
|
828
|
+
"[1, [2, [3, []]], {}]",
|
|
829
|
+
r#"{"a": 1, "b": [true, null], "c": {"d": "e"}}"#,
|
|
830
|
+
r#"{"a": 1, "b": 2, "a": 3}"#,
|
|
831
|
+
r#"{"x": 1, "y": 2, "x": {"z": 1}, "y": 4, "w": 5}"#,
|
|
832
|
+
] {
|
|
833
|
+
same(text);
|
|
834
|
+
}
|
|
835
|
+
let many: String =
|
|
836
|
+
format!("{{{}}}", (0..40).map(|i| format!("\"k{}\": {i}", i % 25)).collect::<Vec<_>>().join(","));
|
|
837
|
+
same(&many);
|
|
838
|
+
}
|
|
839
|
+
|
|
840
|
+
#[test]
|
|
841
|
+
fn doubles_are_correctly_rounded() {
|
|
842
|
+
// serde_json without its float_roundtrip feature reads this one ulp off.
|
|
843
|
+
let text = "-90.50242899999999";
|
|
844
|
+
let d = JsonDocument::parse(text).unwrap();
|
|
845
|
+
assert_eq!(d.to_value().as_f64().unwrap().to_bits(), text.parse::<f64>().unwrap().to_bits());
|
|
846
|
+
}
|
|
847
|
+
|
|
848
|
+
#[test]
|
|
849
|
+
fn rejects_what_serde_json_rejects() {
|
|
850
|
+
for text in [
|
|
851
|
+
"",
|
|
852
|
+
" ",
|
|
853
|
+
"nul",
|
|
854
|
+
"tru",
|
|
855
|
+
"01",
|
|
856
|
+
"-",
|
|
857
|
+
"1.",
|
|
858
|
+
".5",
|
|
859
|
+
"1e",
|
|
860
|
+
"+1",
|
|
861
|
+
"1e400",
|
|
862
|
+
"[1,]",
|
|
863
|
+
"[1 2]",
|
|
864
|
+
"{\"a\" 1}",
|
|
865
|
+
"{\"a\": 1,}",
|
|
866
|
+
"{a: 1}",
|
|
867
|
+
"\"unterminated",
|
|
868
|
+
"\"tab\there\"",
|
|
869
|
+
r#""\x""#,
|
|
870
|
+
r#""\ud800""#,
|
|
871
|
+
r#""\udc00""#,
|
|
872
|
+
r#""\ud800A""#,
|
|
873
|
+
"[1] 2",
|
|
874
|
+
"[",
|
|
875
|
+
"{",
|
|
876
|
+
] {
|
|
877
|
+
both_reject(text);
|
|
878
|
+
}
|
|
879
|
+
}
|
|
880
|
+
|
|
881
|
+
#[test]
|
|
882
|
+
fn depth_limit() {
|
|
883
|
+
for inner in ["", "1"] {
|
|
884
|
+
let nest = |n: usize| format!("{}{inner}{}", "[".repeat(n), "]".repeat(n));
|
|
885
|
+
assert!(JsonDocument::parse(&nest(MAX_DEPTH)).is_ok());
|
|
886
|
+
assert!(JsonDocument::parse(&nest(MAX_DEPTH + 1)).is_err());
|
|
887
|
+
assert!(serde_json::from_str::<Value>(&nest(MAX_DEPTH)).is_ok());
|
|
888
|
+
assert!(serde_json::from_str::<Value>(&nest(MAX_DEPTH + 1)).is_err());
|
|
889
|
+
}
|
|
890
|
+
}
|
|
891
|
+
|
|
892
|
+
#[test]
|
|
893
|
+
fn reads_in_place() {
|
|
894
|
+
let d = JsonDocument::parse(r#"{"a": [1, "x\n"], "b": {"c": null}}"#).unwrap();
|
|
895
|
+
let View::Object(o) = d.root().view() else { panic!() };
|
|
896
|
+
assert_eq!(o.len(), 2);
|
|
897
|
+
assert!(o.contains_key("b") && !o.contains_key("c"));
|
|
898
|
+
let View::Array(a) = o.get("a").unwrap().view() else { panic!() };
|
|
899
|
+
assert_eq!(a.len(), 2);
|
|
900
|
+
assert!(matches!(a.get(1).view(), View::String("x\n")));
|
|
901
|
+
assert_eq!(a.tail(1).count(), 1);
|
|
902
|
+
assert_eq!(a.tail(5).count(), 0);
|
|
903
|
+
assert_eq!(o.iter().map(|(k, _)| k).collect::<Vec<_>>(), ["a", "b"]);
|
|
904
|
+
}
|
|
905
|
+
}
|