corvus_json_schema 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +7 -0
- data/Cargo.lock +390 -0
- data/Cargo.toml +8 -0
- data/LICENSE +201 -0
- data/README.md +111 -0
- data/VERSIONHISTORY.md +7 -0
- data/ext/corvus_json_schema/Cargo.toml +18 -0
- data/ext/corvus_json_schema/extconf.rb +6 -0
- data/ext/corvus_json_schema/rustfmt.toml +2 -0
- data/ext/corvus_json_schema/src/lib.rs +643 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/Cargo.toml +35 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/LICENSE +201 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/README.md +139 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/compiler.rs +1118 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/dialect.rs +108 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/document.rs +905 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/eval/plan/fused.rs +1187 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/eval/plan.rs +1728 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/eval.rs +1451 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/formats.rs +1021 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/instance.rs +228 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/lib.rs +189 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/loader.rs +403 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/metaschemas.rs +29 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/node.rs +417 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/numbers.rs +244 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/options.rs +108 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/pattern.rs +1591 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/results.rs +316 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/uri.rs +274 -0
- data/lib/corvus_json_schema/version.rb +5 -0
- data/lib/corvus_json_schema.rb +65 -0
- metadata +94 -0
|
@@ -0,0 +1,316 @@
|
|
|
1
|
+
//! Results collection: a port of Corvus.Text.Json's `JsonSchemaResultsCollector` and `JsonSchemaAnnotationProducer`.
|
|
2
|
+
//!
|
|
3
|
+
//! The evaluator opens a context per subschema application and writes keyword rows into the open context. Closing a
|
|
4
|
+
//! context either commits it (a summary row, then its own rows newest first, after its committed descendants) or
|
|
5
|
+
//! pops it (everything it and its descendants wrote is discarded). Levels decide which rows exist and which carry
|
|
6
|
+
//! message text: Basic has failures without text, Detailed adds text to failures, Verbose keeps every row with text,
|
|
7
|
+
//! annotations included.
|
|
8
|
+
|
|
9
|
+
use std::collections::BTreeMap;
|
|
10
|
+
|
|
11
|
+
use serde_json::Value;
|
|
12
|
+
|
|
13
|
+
/// How much a results collector records.
|
|
14
|
+
#[derive(Clone, Copy, Debug, PartialEq, Eq, PartialOrd, Ord)]
|
|
15
|
+
pub enum ResultsLevel {
|
|
16
|
+
/// Failures only, without message text (the lowest overhead).
|
|
17
|
+
Basic,
|
|
18
|
+
/// Failures only, with message text.
|
|
19
|
+
Detailed,
|
|
20
|
+
/// Every evaluation, passing and failing, with message text, including annotations.
|
|
21
|
+
Verbose,
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
/// One result row.
|
|
25
|
+
#[derive(Clone, Debug, PartialEq, Eq)]
|
|
26
|
+
pub struct SchemaResult {
|
|
27
|
+
pub is_match: bool,
|
|
28
|
+
/// The message, or `""` when the level records none or the keyword has none. Annotation rows carry raw JSON.
|
|
29
|
+
pub message: String,
|
|
30
|
+
/// The path of keywords from the root schema (e.g. `/properties/name/type`).
|
|
31
|
+
pub evaluation_location: String,
|
|
32
|
+
/// The JSON pointer of the evaluated schema (or keyword) within its document (e.g. `/properties/name/type`).
|
|
33
|
+
pub schema_evaluation_location: String,
|
|
34
|
+
/// The JSON pointer of the instance location (e.g. `/name`).
|
|
35
|
+
pub document_evaluation_location: String,
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
/// A message for a result row, produced only when the level records message text.
|
|
39
|
+
pub(crate) enum Message<'a> {
|
|
40
|
+
None,
|
|
41
|
+
Static(&'static str),
|
|
42
|
+
Lazy(&'a dyn Fn() -> String),
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
struct Frame {
|
|
46
|
+
eval_len: usize,
|
|
47
|
+
schema_path: String,
|
|
48
|
+
doc_len: usize,
|
|
49
|
+
commit_index: usize,
|
|
50
|
+
rows_start: usize,
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
/// Collects the results of an evaluation (`JsonSchemaResultsCollector`).
|
|
54
|
+
pub struct JsonSchemaResultsCollector {
|
|
55
|
+
level: ResultsLevel,
|
|
56
|
+
committed: Vec<SchemaResult>,
|
|
57
|
+
frames: Vec<Frame>,
|
|
58
|
+
/// The rows written into open frames (each frame owns the tail from its `rows_start`).
|
|
59
|
+
pending: Vec<SchemaResult>,
|
|
60
|
+
eval_path: String,
|
|
61
|
+
schema_path: String,
|
|
62
|
+
doc_path: String,
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
/// Encodes a JSON pointer segment (`~` as `~0`, `/` as `~1`).
|
|
66
|
+
pub fn encode_pointer_segment(segment: &str) -> std::borrow::Cow<'_, str> {
|
|
67
|
+
crate::uri::escape_pointer_token(segment)
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
impl JsonSchemaResultsCollector {
|
|
71
|
+
/// Creates a collector at the given level.
|
|
72
|
+
pub fn new(level: ResultsLevel) -> Self {
|
|
73
|
+
JsonSchemaResultsCollector {
|
|
74
|
+
level,
|
|
75
|
+
committed: Vec::new(),
|
|
76
|
+
frames: Vec::new(),
|
|
77
|
+
pending: Vec::new(),
|
|
78
|
+
eval_path: String::new(),
|
|
79
|
+
schema_path: String::new(),
|
|
80
|
+
doc_path: String::new(),
|
|
81
|
+
}
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
pub fn level(&self) -> ResultsLevel {
|
|
85
|
+
self.level
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
/// The results, in commit order.
|
|
89
|
+
pub fn results(&self) -> &[SchemaResult] {
|
|
90
|
+
&self.committed
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
/// Removes and returns the results.
|
|
94
|
+
pub fn take_results(&mut self) -> Vec<SchemaResult> {
|
|
95
|
+
std::mem::take(&mut self.committed)
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
// ------------------------------------------------------------------------------------------------------------
|
|
99
|
+
// The evaluator's side
|
|
100
|
+
|
|
101
|
+
/// Opens a child context. The evaluation path is extended by `eval_segment` (verbatim), the schema path is
|
|
102
|
+
/// replaced by `schema_location`, and the document path is extended by `doc_segment` (already pointer-encoded).
|
|
103
|
+
pub(crate) fn begin_child_context(
|
|
104
|
+
&mut self,
|
|
105
|
+
eval_segment: Option<&str>,
|
|
106
|
+
schema_location: Option<&str>,
|
|
107
|
+
doc_segment: Option<&str>,
|
|
108
|
+
) {
|
|
109
|
+
self.frames.push(Frame {
|
|
110
|
+
eval_len: self.eval_path.len(),
|
|
111
|
+
schema_path: self.schema_path.clone(),
|
|
112
|
+
doc_len: self.doc_path.len(),
|
|
113
|
+
commit_index: self.committed.len(),
|
|
114
|
+
rows_start: self.pending.len(),
|
|
115
|
+
});
|
|
116
|
+
if let Some(s) = eval_segment {
|
|
117
|
+
self.eval_path.push('/');
|
|
118
|
+
self.eval_path.push_str(s);
|
|
119
|
+
}
|
|
120
|
+
if let Some(s) = schema_location {
|
|
121
|
+
self.schema_path.clear();
|
|
122
|
+
self.schema_path.push_str(s);
|
|
123
|
+
}
|
|
124
|
+
if let Some(s) = doc_segment {
|
|
125
|
+
self.doc_path.push('/');
|
|
126
|
+
self.doc_path.push_str(s);
|
|
127
|
+
}
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
/// Closes a child context. When the parent does not need the child's results (`parent_is_match`) they are
|
|
131
|
+
/// discarded below Verbose; otherwise the context's summary row is written and its rows are committed.
|
|
132
|
+
pub(crate) fn commit_child_context(&mut self, parent_is_match: bool, child_is_match: bool, message: Message<'_>) {
|
|
133
|
+
if parent_is_match && self.level != ResultsLevel::Verbose {
|
|
134
|
+
self.pop_child_context();
|
|
135
|
+
return;
|
|
136
|
+
}
|
|
137
|
+
let row =
|
|
138
|
+
self.row(child_is_match, message, self.eval_path.clone(), self.schema_path.clone(), self.doc_path.clone());
|
|
139
|
+
self.pending.push(row);
|
|
140
|
+
let frame = self.frames.pop().unwrap();
|
|
141
|
+
let rows = self.pending.split_off(frame.rows_start);
|
|
142
|
+
self.committed.extend(rows.into_iter().rev());
|
|
143
|
+
self.restore(frame);
|
|
144
|
+
}
|
|
145
|
+
|
|
146
|
+
/// Closes a child context and discards everything it and its descendants wrote.
|
|
147
|
+
pub(crate) fn pop_child_context(&mut self) {
|
|
148
|
+
let frame = self.frames.pop().unwrap();
|
|
149
|
+
self.committed.truncate(frame.commit_index);
|
|
150
|
+
self.pending.truncate(frame.rows_start);
|
|
151
|
+
self.restore(frame);
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
pub(crate) fn evaluated_keyword(&mut self, is_match: bool, message: Message<'_>, keyword: &str) {
|
|
155
|
+
if !is_match || self.level == ResultsLevel::Verbose {
|
|
156
|
+
let k = encode_pointer_segment(keyword);
|
|
157
|
+
let row = self.row(
|
|
158
|
+
is_match,
|
|
159
|
+
message,
|
|
160
|
+
format!("{}/{k}", self.eval_path),
|
|
161
|
+
format!("{}/{k}", self.schema_path),
|
|
162
|
+
self.doc_path.clone(),
|
|
163
|
+
);
|
|
164
|
+
self.pending.push(row);
|
|
165
|
+
}
|
|
166
|
+
}
|
|
167
|
+
|
|
168
|
+
pub(crate) fn evaluated_keyword_for_property(
|
|
169
|
+
&mut self,
|
|
170
|
+
is_match: bool,
|
|
171
|
+
message: Message<'_>,
|
|
172
|
+
property_name: &str,
|
|
173
|
+
keyword: &str,
|
|
174
|
+
) {
|
|
175
|
+
if !is_match || self.level == ResultsLevel::Verbose {
|
|
176
|
+
let k = encode_pointer_segment(keyword);
|
|
177
|
+
let row = self.row(
|
|
178
|
+
is_match,
|
|
179
|
+
message,
|
|
180
|
+
format!("{}/{k}", self.eval_path),
|
|
181
|
+
format!("{}/{k}", self.schema_path),
|
|
182
|
+
format!("{}/{}", self.doc_path, encode_pointer_segment(property_name)),
|
|
183
|
+
);
|
|
184
|
+
self.pending.push(row);
|
|
185
|
+
}
|
|
186
|
+
}
|
|
187
|
+
|
|
188
|
+
/// An annotation: Verbose only; the keyword extends the evaluation path but not the schema path.
|
|
189
|
+
pub(crate) fn ignored_keyword(&mut self, message: Message<'_>, keyword: &str) {
|
|
190
|
+
if self.level == ResultsLevel::Verbose {
|
|
191
|
+
let row = self.row(
|
|
192
|
+
true,
|
|
193
|
+
message,
|
|
194
|
+
format!("{}/{}", self.eval_path, encode_pointer_segment(keyword)),
|
|
195
|
+
self.schema_path.clone(),
|
|
196
|
+
self.doc_path.clone(),
|
|
197
|
+
);
|
|
198
|
+
self.pending.push(row);
|
|
199
|
+
}
|
|
200
|
+
}
|
|
201
|
+
|
|
202
|
+
pub(crate) fn evaluated_boolean_schema(&mut self, is_match: bool) {
|
|
203
|
+
if !is_match || self.level == ResultsLevel::Verbose {
|
|
204
|
+
let row = self.row(
|
|
205
|
+
is_match,
|
|
206
|
+
Message::None,
|
|
207
|
+
self.eval_path.clone(),
|
|
208
|
+
self.schema_path.clone(),
|
|
209
|
+
self.doc_path.clone(),
|
|
210
|
+
);
|
|
211
|
+
self.pending.push(row);
|
|
212
|
+
}
|
|
213
|
+
}
|
|
214
|
+
|
|
215
|
+
fn row(
|
|
216
|
+
&self,
|
|
217
|
+
is_match: bool,
|
|
218
|
+
message: Message<'_>,
|
|
219
|
+
evaluation_location: String,
|
|
220
|
+
schema_evaluation_location: String,
|
|
221
|
+
document_evaluation_location: String,
|
|
222
|
+
) -> SchemaResult {
|
|
223
|
+
let with_text = self.level == ResultsLevel::Verbose || (!is_match && self.level >= ResultsLevel::Detailed);
|
|
224
|
+
let message = if with_text {
|
|
225
|
+
match message {
|
|
226
|
+
Message::None => String::new(),
|
|
227
|
+
Message::Static(s) => s.to_string(),
|
|
228
|
+
Message::Lazy(f) => f(),
|
|
229
|
+
}
|
|
230
|
+
} else {
|
|
231
|
+
String::new()
|
|
232
|
+
};
|
|
233
|
+
SchemaResult {
|
|
234
|
+
is_match,
|
|
235
|
+
message,
|
|
236
|
+
evaluation_location,
|
|
237
|
+
schema_evaluation_location,
|
|
238
|
+
document_evaluation_location,
|
|
239
|
+
}
|
|
240
|
+
}
|
|
241
|
+
|
|
242
|
+
fn restore(&mut self, frame: Frame) {
|
|
243
|
+
self.eval_path.truncate(frame.eval_len);
|
|
244
|
+
self.schema_path = frame.schema_path;
|
|
245
|
+
self.doc_path.truncate(frame.doc_len);
|
|
246
|
+
}
|
|
247
|
+
}
|
|
248
|
+
|
|
249
|
+
/// An annotation extracted from verbose results.
|
|
250
|
+
#[derive(Clone, Debug, PartialEq, Eq)]
|
|
251
|
+
pub struct Annotation {
|
|
252
|
+
/// The instance location (JSON pointer).
|
|
253
|
+
pub instance_location: String,
|
|
254
|
+
pub keyword: String,
|
|
255
|
+
/// The JSON pointer of the schema object that holds the keyword.
|
|
256
|
+
pub schema_location: String,
|
|
257
|
+
/// The annotation value as JSON text.
|
|
258
|
+
pub value: String,
|
|
259
|
+
}
|
|
260
|
+
|
|
261
|
+
/// The annotations in a verbose collector's results (`JsonSchemaAnnotationProducer.EnumerateAnnotations`).
|
|
262
|
+
pub fn enumerate_annotations(collector: &JsonSchemaResultsCollector) -> impl Iterator<Item = Annotation> + '_ {
|
|
263
|
+
collector.results().iter().filter_map(|r| {
|
|
264
|
+
if !r.is_match || r.message.is_empty() {
|
|
265
|
+
return None;
|
|
266
|
+
}
|
|
267
|
+
let slash = r.evaluation_location.rfind('/')?;
|
|
268
|
+
if r.evaluation_location == r.schema_evaluation_location {
|
|
269
|
+
return None;
|
|
270
|
+
}
|
|
271
|
+
let keyword = &r.evaluation_location[slash + 1..];
|
|
272
|
+
let first = r.message.as_bytes()[0];
|
|
273
|
+
if keyword.is_empty()
|
|
274
|
+
|| !(matches!(first, b'"' | b'{' | b'[' | b't' | b'f' | b'n' | b'-') || first.is_ascii_digit())
|
|
275
|
+
{
|
|
276
|
+
return None;
|
|
277
|
+
}
|
|
278
|
+
Some(Annotation {
|
|
279
|
+
instance_location: r.document_evaluation_location.clone(),
|
|
280
|
+
keyword: keyword.to_string(),
|
|
281
|
+
schema_location: r.schema_evaluation_location.clone(),
|
|
282
|
+
value: r.message.clone(),
|
|
283
|
+
})
|
|
284
|
+
})
|
|
285
|
+
}
|
|
286
|
+
|
|
287
|
+
/// `#` followed by the schema location, percent-encoded as a URI fragment (upper-case hex, UTF-8).
|
|
288
|
+
pub fn schema_location_fragment(schema_location: &str) -> String {
|
|
289
|
+
let mut out = String::from("#");
|
|
290
|
+
for &byte in schema_location.as_bytes() {
|
|
291
|
+
let c = byte as char;
|
|
292
|
+
if byte < 128 && (c.is_ascii_alphanumeric() || "-._~!$&'()*+,;=:@/?".contains(c)) {
|
|
293
|
+
out.push(c);
|
|
294
|
+
} else {
|
|
295
|
+
out.push_str(&format!("%{byte:02X}"));
|
|
296
|
+
}
|
|
297
|
+
}
|
|
298
|
+
out
|
|
299
|
+
}
|
|
300
|
+
|
|
301
|
+
/// Annotations grouped by instance location, then keyword, then schema location fragment, with parsed values
|
|
302
|
+
/// (`JsonSchemaAnnotationProducer.WriteAnnotationsTo`): `{ "/name": { "title": { "#/properties/name": "Name" } } }`.
|
|
303
|
+
pub fn collect_annotations(
|
|
304
|
+
collector: &JsonSchemaResultsCollector,
|
|
305
|
+
) -> BTreeMap<String, BTreeMap<String, BTreeMap<String, Value>>> {
|
|
306
|
+
let mut out: BTreeMap<String, BTreeMap<String, BTreeMap<String, Value>>> = BTreeMap::new();
|
|
307
|
+
for a in enumerate_annotations(collector) {
|
|
308
|
+
let value = serde_json::from_str(&a.value).unwrap_or(Value::String(a.value.clone()));
|
|
309
|
+
out.entry(a.instance_location)
|
|
310
|
+
.or_default()
|
|
311
|
+
.entry(a.keyword)
|
|
312
|
+
.or_default()
|
|
313
|
+
.insert(schema_location_fragment(&a.schema_location), value);
|
|
314
|
+
}
|
|
315
|
+
out
|
|
316
|
+
}
|
|
@@ -0,0 +1,274 @@
|
|
|
1
|
+
//! URI handling for schema identification and reference resolution (RFC 3986 section 5).
|
|
2
|
+
//!
|
|
3
|
+
//! Mirrors the TypeScript port's `uri.ts`: reference resolution is implemented directly so that opaque bases
|
|
4
|
+
//! (`urn:`, `tag:`) resolve the same way everywhere.
|
|
5
|
+
|
|
6
|
+
use serde_json::Value;
|
|
7
|
+
|
|
8
|
+
struct UriParts<'a> {
|
|
9
|
+
scheme: Option<&'a str>,
|
|
10
|
+
authority: Option<&'a str>,
|
|
11
|
+
path: &'a str,
|
|
12
|
+
query: Option<&'a str>,
|
|
13
|
+
}
|
|
14
|
+
|
|
15
|
+
/// Splits a URI into scheme, authority, path and query (the fragment must already be removed).
|
|
16
|
+
fn parse(uri: &str) -> UriParts<'_> {
|
|
17
|
+
let mut rest = uri;
|
|
18
|
+
let mut scheme = None;
|
|
19
|
+
if let Some(colon) = scheme_end(rest) {
|
|
20
|
+
scheme = Some(&rest[..colon]);
|
|
21
|
+
rest = &rest[colon + 1..];
|
|
22
|
+
}
|
|
23
|
+
let mut authority = None;
|
|
24
|
+
if let Some(after) = rest.strip_prefix("//") {
|
|
25
|
+
let end = after.find(['/', '?']).unwrap_or(after.len());
|
|
26
|
+
authority = Some(&after[..end]);
|
|
27
|
+
rest = &after[end..];
|
|
28
|
+
}
|
|
29
|
+
let (path, query) = match rest.find('?') {
|
|
30
|
+
Some(q) => (&rest[..q], Some(&rest[q + 1..])),
|
|
31
|
+
None => (rest, None),
|
|
32
|
+
};
|
|
33
|
+
UriParts { scheme, authority, path, query }
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
/// The index of the ':' ending a URI scheme, if the text starts with one.
|
|
37
|
+
fn scheme_end(s: &str) -> Option<usize> {
|
|
38
|
+
let bytes = s.as_bytes();
|
|
39
|
+
if bytes.is_empty() || !bytes[0].is_ascii_alphabetic() {
|
|
40
|
+
return None;
|
|
41
|
+
}
|
|
42
|
+
for (i, &b) in bytes.iter().enumerate().skip(1) {
|
|
43
|
+
if b == b':' {
|
|
44
|
+
return Some(i);
|
|
45
|
+
}
|
|
46
|
+
if !(b.is_ascii_alphanumeric() || b == b'+' || b == b'-' || b == b'.') {
|
|
47
|
+
return None;
|
|
48
|
+
}
|
|
49
|
+
}
|
|
50
|
+
None
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
/// True when the reference starts with a URI scheme.
|
|
54
|
+
pub(crate) fn has_scheme(reference: &str) -> bool {
|
|
55
|
+
scheme_end(reference).is_some()
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
/// Splits a reference at its first '#'.
|
|
59
|
+
pub(crate) fn split(reference: &str) -> (&str, &str) {
|
|
60
|
+
match reference.find('#') {
|
|
61
|
+
Some(i) => (&reference[..i], &reference[i + 1..]),
|
|
62
|
+
None => (reference, ""),
|
|
63
|
+
}
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
fn format(scheme: Option<&str>, authority: Option<&str>, path: &str, query: Option<&str>) -> String {
|
|
67
|
+
let mut s = String::with_capacity(path.len() + 32);
|
|
68
|
+
if let Some(sc) = scheme {
|
|
69
|
+
s.push_str(sc);
|
|
70
|
+
s.push(':');
|
|
71
|
+
}
|
|
72
|
+
if let Some(a) = authority {
|
|
73
|
+
s.push_str("//");
|
|
74
|
+
s.push_str(a);
|
|
75
|
+
}
|
|
76
|
+
s.push_str(path);
|
|
77
|
+
if let Some(q) = query {
|
|
78
|
+
s.push('?');
|
|
79
|
+
s.push_str(q);
|
|
80
|
+
}
|
|
81
|
+
s
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
fn remove_dot_segments(path: &str) -> String {
|
|
85
|
+
if !path.contains('.') {
|
|
86
|
+
return path.to_string();
|
|
87
|
+
}
|
|
88
|
+
let input: Vec<&str> = path.split('/').collect();
|
|
89
|
+
let mut out: Vec<&str> = Vec::with_capacity(input.len());
|
|
90
|
+
let last = input.len() - 1;
|
|
91
|
+
for (i, seg) in input.iter().enumerate() {
|
|
92
|
+
match *seg {
|
|
93
|
+
"." => {
|
|
94
|
+
if i == last {
|
|
95
|
+
out.push("");
|
|
96
|
+
}
|
|
97
|
+
}
|
|
98
|
+
".." => {
|
|
99
|
+
if out.len() > 1 || (out.len() == 1 && !out[0].is_empty()) {
|
|
100
|
+
out.pop();
|
|
101
|
+
}
|
|
102
|
+
if i == last {
|
|
103
|
+
out.push("");
|
|
104
|
+
}
|
|
105
|
+
}
|
|
106
|
+
s => out.push(s),
|
|
107
|
+
}
|
|
108
|
+
}
|
|
109
|
+
out.join("/")
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
fn merge(base: &UriParts<'_>, ref_path: &str) -> String {
|
|
113
|
+
if base.authority.is_some() && base.path.is_empty() {
|
|
114
|
+
return format!("/{ref_path}");
|
|
115
|
+
}
|
|
116
|
+
match base.path.rfind('/') {
|
|
117
|
+
Some(i) => format!("{}{}", &base.path[..=i], ref_path),
|
|
118
|
+
None => ref_path.to_string(),
|
|
119
|
+
}
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
fn normalize_parts(scheme: Option<&str>, authority: Option<&str>, path: &str, query: Option<&str>) -> String {
|
|
123
|
+
let scheme = scheme.map(|s| s.to_ascii_lowercase());
|
|
124
|
+
let mut path = remove_dot_segments(path);
|
|
125
|
+
let authority = authority.map(|a| {
|
|
126
|
+
let mut a = a.to_ascii_lowercase();
|
|
127
|
+
let default_port = match scheme.as_deref() {
|
|
128
|
+
Some("http") => Some(":80"),
|
|
129
|
+
Some("https") => Some(":443"),
|
|
130
|
+
_ => None,
|
|
131
|
+
};
|
|
132
|
+
if let Some(port) = default_port {
|
|
133
|
+
if a.ends_with(port) {
|
|
134
|
+
a.truncate(a.len() - port.len());
|
|
135
|
+
}
|
|
136
|
+
}
|
|
137
|
+
if path.is_empty() {
|
|
138
|
+
path = "/".to_string();
|
|
139
|
+
}
|
|
140
|
+
a
|
|
141
|
+
});
|
|
142
|
+
format(scheme.as_deref(), authority.as_deref(), &path, query)
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
/// Normalises an absolute URI (dropping its fragment) so that equivalent spellings compare equal.
|
|
146
|
+
pub(crate) fn normalize(uri: &str) -> String {
|
|
147
|
+
let (uri_part, _) = split(uri);
|
|
148
|
+
if !has_scheme(uri_part) {
|
|
149
|
+
return uri_part.to_string();
|
|
150
|
+
}
|
|
151
|
+
let p = parse(uri_part);
|
|
152
|
+
normalize_parts(p.scheme, p.authority, p.path, p.query)
|
|
153
|
+
}
|
|
154
|
+
|
|
155
|
+
/// Resolves a reference (without fragment) against a base URI, returning the normalised absolute URI.
|
|
156
|
+
pub(crate) fn resolve(base_uri: &str, reference: &str) -> String {
|
|
157
|
+
if reference.is_empty() {
|
|
158
|
+
return base_uri.to_string();
|
|
159
|
+
}
|
|
160
|
+
let r = parse(reference);
|
|
161
|
+
if r.scheme.is_some() {
|
|
162
|
+
return normalize_parts(r.scheme, r.authority, r.path, r.query);
|
|
163
|
+
}
|
|
164
|
+
if base_uri.is_empty() {
|
|
165
|
+
return reference.to_string();
|
|
166
|
+
}
|
|
167
|
+
let b = parse(base_uri);
|
|
168
|
+
let (authority, path, query);
|
|
169
|
+
if r.authority.is_some() {
|
|
170
|
+
authority = r.authority;
|
|
171
|
+
path = remove_dot_segments(r.path);
|
|
172
|
+
query = r.query;
|
|
173
|
+
} else {
|
|
174
|
+
if r.path.is_empty() {
|
|
175
|
+
path = b.path.to_string();
|
|
176
|
+
query = if r.query.is_some() { r.query } else { b.query };
|
|
177
|
+
} else {
|
|
178
|
+
path = if r.path.starts_with('/') {
|
|
179
|
+
remove_dot_segments(r.path)
|
|
180
|
+
} else {
|
|
181
|
+
remove_dot_segments(&merge(&b, r.path))
|
|
182
|
+
};
|
|
183
|
+
query = r.query;
|
|
184
|
+
}
|
|
185
|
+
authority = b.authority;
|
|
186
|
+
}
|
|
187
|
+
if b.scheme.is_some() {
|
|
188
|
+
normalize_parts(b.scheme, authority, &path, query)
|
|
189
|
+
} else {
|
|
190
|
+
format(None, authority, &path, query)
|
|
191
|
+
}
|
|
192
|
+
}
|
|
193
|
+
|
|
194
|
+
/// Percent-decodes a fragment (invalid escapes leave the text unchanged).
|
|
195
|
+
pub(crate) fn decode_fragment(fragment: &str) -> String {
|
|
196
|
+
if !fragment.contains('%') {
|
|
197
|
+
return fragment.to_string();
|
|
198
|
+
}
|
|
199
|
+
let bytes = fragment.as_bytes();
|
|
200
|
+
let mut out = Vec::with_capacity(bytes.len());
|
|
201
|
+
let mut i = 0;
|
|
202
|
+
while i < bytes.len() {
|
|
203
|
+
if bytes[i] == b'%' {
|
|
204
|
+
let hex = |c: u8| (c as char).to_digit(16);
|
|
205
|
+
match (bytes.get(i + 1).and_then(|&c| hex(c)), bytes.get(i + 2).and_then(|&c| hex(c))) {
|
|
206
|
+
(Some(h), Some(l)) => {
|
|
207
|
+
out.push((h * 16 + l) as u8);
|
|
208
|
+
i += 3;
|
|
209
|
+
continue;
|
|
210
|
+
}
|
|
211
|
+
_ => return fragment.to_string(),
|
|
212
|
+
}
|
|
213
|
+
}
|
|
214
|
+
out.push(bytes[i]);
|
|
215
|
+
i += 1;
|
|
216
|
+
}
|
|
217
|
+
String::from_utf8(out).unwrap_or_else(|_| fragment.to_string())
|
|
218
|
+
}
|
|
219
|
+
|
|
220
|
+
/// Escapes a JSON pointer token (`~` as `~0`, `/` as `~1`).
|
|
221
|
+
pub(crate) fn escape_pointer_token(token: &str) -> std::borrow::Cow<'_, str> {
|
|
222
|
+
if !token.contains(['~', '/']) {
|
|
223
|
+
return std::borrow::Cow::Borrowed(token);
|
|
224
|
+
}
|
|
225
|
+
std::borrow::Cow::Owned(token.replace('~', "~0").replace('/', "~1"))
|
|
226
|
+
}
|
|
227
|
+
|
|
228
|
+
/// Resolves an RFC 6901 JSON pointer against a JSON value, returning the value and its normalised pointer.
|
|
229
|
+
pub(crate) fn resolve_pointer<'a>(root: &'a Value, pointer: &str) -> Option<(&'a Value, String)> {
|
|
230
|
+
if pointer.is_empty() {
|
|
231
|
+
return Some((root, String::new()));
|
|
232
|
+
}
|
|
233
|
+
let rest = pointer.strip_prefix('/')?;
|
|
234
|
+
let mut current = root;
|
|
235
|
+
let mut path = String::with_capacity(pointer.len());
|
|
236
|
+
for raw in rest.split('/') {
|
|
237
|
+
let token = raw.replace("~1", "/").replace("~0", "~");
|
|
238
|
+
match current {
|
|
239
|
+
Value::Array(items) => {
|
|
240
|
+
let valid = token == "0"
|
|
241
|
+
|| (!token.is_empty() && !token.starts_with('0') && token.bytes().all(|b| b.is_ascii_digit()));
|
|
242
|
+
if !valid {
|
|
243
|
+
return None;
|
|
244
|
+
}
|
|
245
|
+
let i: usize = token.parse().ok()?;
|
|
246
|
+
current = items.get(i)?;
|
|
247
|
+
path.push('/');
|
|
248
|
+
path.push_str(&token);
|
|
249
|
+
}
|
|
250
|
+
Value::Object(map) => {
|
|
251
|
+
current = map.get(&token)?;
|
|
252
|
+
path.push('/');
|
|
253
|
+
path.push_str(&escape_pointer_token(&token));
|
|
254
|
+
}
|
|
255
|
+
_ => return None,
|
|
256
|
+
}
|
|
257
|
+
}
|
|
258
|
+
Some((current, path))
|
|
259
|
+
}
|
|
260
|
+
|
|
261
|
+
#[cfg(test)]
|
|
262
|
+
mod tests {
|
|
263
|
+
use super::*;
|
|
264
|
+
|
|
265
|
+
#[test]
|
|
266
|
+
fn resolves_relative_references() {
|
|
267
|
+
assert_eq!(resolve("http://example.com/a/b.json", "c.json"), "http://example.com/a/c.json");
|
|
268
|
+
assert_eq!(resolve("http://example.com/a/b.json", "../c.json"), "http://example.com/c.json");
|
|
269
|
+
assert_eq!(resolve("http://Example.com:80", ""), "http://Example.com:80");
|
|
270
|
+
assert_eq!(normalize("HTTP://Example.COM:80#frag"), "http://example.com/");
|
|
271
|
+
assert_eq!(resolve("urn:uuid:deadbeef-1234", ""), "urn:uuid:deadbeef-1234");
|
|
272
|
+
assert_eq!(resolve("tag:example.com,2021:a", "b"), "tag:b");
|
|
273
|
+
}
|
|
274
|
+
}
|
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
# A high-performance JSON Schema evaluator (draft 4, 6, 7, 2019-09 and 2020-12) for Ruby, backed by the
|
|
4
|
+
# corvus-json-schema Rust crate, with results collection and annotations.
|
|
5
|
+
#
|
|
6
|
+
# validator = CorvusJsonSchema.compile({ "type" => "object", "required" => ["id"] })
|
|
7
|
+
# validator.valid?({ "id" => 3 }) # => true
|
|
8
|
+
# validator.valid_json?('{"id": 3}') # => true, parsed in Rust
|
|
9
|
+
#
|
|
10
|
+
# Values are read in place: Hash (String or Symbol keys), Array, String, Symbol (as its name), Integer, Float, true,
|
|
11
|
+
# false and nil. Integers beyond 64 bits become the nearest double. A value the schema examines that is none of these
|
|
12
|
+
# raises (TypeError; ArgumentError for NaN and infinities; EncodingError for a String that is not valid UTF-8); values
|
|
13
|
+
# are read only as the schema examines them.
|
|
14
|
+
require_relative "corvus_json_schema/version"
|
|
15
|
+
|
|
16
|
+
begin
|
|
17
|
+
RUBY_VERSION =~ /(\d+\.\d+)/
|
|
18
|
+
require_relative "corvus_json_schema/#{Regexp.last_match(1)}/corvus_json_schema"
|
|
19
|
+
rescue LoadError
|
|
20
|
+
require_relative "corvus_json_schema/corvus_json_schema"
|
|
21
|
+
end
|
|
22
|
+
|
|
23
|
+
module CorvusJsonSchema
|
|
24
|
+
DIALECTS = {
|
|
25
|
+
draft4: 4, draft6: 6, draft7: 7, draft201909: 2019, draft202012: 2020
|
|
26
|
+
}.freeze
|
|
27
|
+
|
|
28
|
+
RESULTS_LEVELS = { basic: 0, detailed: 1, verbose: 2 }.freeze
|
|
29
|
+
|
|
30
|
+
# Compiles a schema (a Hash or Array, true or false, or JSON text) into a Validator.
|
|
31
|
+
#
|
|
32
|
+
# Options: default_dialect (:draft4, :draft6, :draft7, :draft201909, :draft202012; the dialect of schemas without
|
|
33
|
+
# $schema), assert_format (nil follows the vocabularies), assert_format_in_legacy_drafts, assert_content, formats (a
|
|
34
|
+
# Hash of format name to a callable returning whether a string is valid), resolver (a callable from an absolute URI
|
|
35
|
+
# to the document, as a Hash or JSON text, or nil when unknown), base_uri, entry_point (a reference such as
|
|
36
|
+
# "#/$defs/item") and max_depth (of in-place recursion, default 128).
|
|
37
|
+
def self.compile(schema, default_dialect: :draft202012, assert_format: nil, assert_format_in_legacy_drafts: false,
|
|
38
|
+
assert_content: true, formats: nil, resolver: nil, base_uri: nil, entry_point: nil, max_depth: 128)
|
|
39
|
+
dialect = DIALECTS.fetch(default_dialect) { raise ArgumentError, "unknown dialect #{default_dialect.inspect}" }
|
|
40
|
+
Native.compile(schema, dialect, assert_format, assert_format_in_legacy_drafts, assert_content, formats, resolver,
|
|
41
|
+
base_uri, entry_point, Integer(max_depth))
|
|
42
|
+
end
|
|
43
|
+
|
|
44
|
+
# The version of the corvus-json-schema crate the extension was built from.
|
|
45
|
+
def self.crate_version
|
|
46
|
+
Native.crate_version
|
|
47
|
+
end
|
|
48
|
+
|
|
49
|
+
# A compiled schema: valid?(value), valid_json?(text) and evaluate(value, collector).
|
|
50
|
+
class Validator
|
|
51
|
+
# Evaluates the value, replacing the collector's results with this evaluation's (every keyword is evaluated and
|
|
52
|
+
# reported at the collector's level). Returns whether the value is valid.
|
|
53
|
+
def evaluate(value, collector)
|
|
54
|
+
native_evaluate(value, collector)
|
|
55
|
+
end
|
|
56
|
+
end
|
|
57
|
+
|
|
58
|
+
# The results of the latest evaluation into it: results (rows as Hashes) and annotations (from a :verbose
|
|
59
|
+
# evaluation, grouped by instance location, keyword and schema location).
|
|
60
|
+
class Collector
|
|
61
|
+
def self.new(level = :basic)
|
|
62
|
+
native_new(RESULTS_LEVELS.fetch(level) { raise ArgumentError, "unknown results level #{level.inspect}" })
|
|
63
|
+
end
|
|
64
|
+
end
|
|
65
|
+
end
|