corvus_json_schema 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (33) hide show
  1. checksums.yaml +7 -0
  2. data/Cargo.lock +390 -0
  3. data/Cargo.toml +8 -0
  4. data/LICENSE +201 -0
  5. data/README.md +111 -0
  6. data/VERSIONHISTORY.md +7 -0
  7. data/ext/corvus_json_schema/Cargo.toml +18 -0
  8. data/ext/corvus_json_schema/extconf.rb +6 -0
  9. data/ext/corvus_json_schema/rustfmt.toml +2 -0
  10. data/ext/corvus_json_schema/src/lib.rs +643 -0
  11. data/ext/corvus_json_schema/vendor/corvus-json-schema/Cargo.toml +35 -0
  12. data/ext/corvus_json_schema/vendor/corvus-json-schema/LICENSE +201 -0
  13. data/ext/corvus_json_schema/vendor/corvus-json-schema/README.md +139 -0
  14. data/ext/corvus_json_schema/vendor/corvus-json-schema/src/compiler.rs +1118 -0
  15. data/ext/corvus_json_schema/vendor/corvus-json-schema/src/dialect.rs +108 -0
  16. data/ext/corvus_json_schema/vendor/corvus-json-schema/src/document.rs +905 -0
  17. data/ext/corvus_json_schema/vendor/corvus-json-schema/src/eval/plan/fused.rs +1187 -0
  18. data/ext/corvus_json_schema/vendor/corvus-json-schema/src/eval/plan.rs +1728 -0
  19. data/ext/corvus_json_schema/vendor/corvus-json-schema/src/eval.rs +1451 -0
  20. data/ext/corvus_json_schema/vendor/corvus-json-schema/src/formats.rs +1021 -0
  21. data/ext/corvus_json_schema/vendor/corvus-json-schema/src/instance.rs +228 -0
  22. data/ext/corvus_json_schema/vendor/corvus-json-schema/src/lib.rs +189 -0
  23. data/ext/corvus_json_schema/vendor/corvus-json-schema/src/loader.rs +403 -0
  24. data/ext/corvus_json_schema/vendor/corvus-json-schema/src/metaschemas.rs +29 -0
  25. data/ext/corvus_json_schema/vendor/corvus-json-schema/src/node.rs +417 -0
  26. data/ext/corvus_json_schema/vendor/corvus-json-schema/src/numbers.rs +244 -0
  27. data/ext/corvus_json_schema/vendor/corvus-json-schema/src/options.rs +108 -0
  28. data/ext/corvus_json_schema/vendor/corvus-json-schema/src/pattern.rs +1591 -0
  29. data/ext/corvus_json_schema/vendor/corvus-json-schema/src/results.rs +316 -0
  30. data/ext/corvus_json_schema/vendor/corvus-json-schema/src/uri.rs +274 -0
  31. data/lib/corvus_json_schema/version.rb +5 -0
  32. data/lib/corvus_json_schema.rb +65 -0
  33. metadata +94 -0
@@ -0,0 +1,316 @@
1
+ //! Results collection: a port of Corvus.Text.Json's `JsonSchemaResultsCollector` and `JsonSchemaAnnotationProducer`.
2
+ //!
3
+ //! The evaluator opens a context per subschema application and writes keyword rows into the open context. Closing a
4
+ //! context either commits it (a summary row, then its own rows newest first, after its committed descendants) or
5
+ //! pops it (everything it and its descendants wrote is discarded). Levels decide which rows exist and which carry
6
+ //! message text: Basic has failures without text, Detailed adds text to failures, Verbose keeps every row with text,
7
+ //! annotations included.
8
+
9
+ use std::collections::BTreeMap;
10
+
11
+ use serde_json::Value;
12
+
13
+ /// How much a results collector records.
14
+ #[derive(Clone, Copy, Debug, PartialEq, Eq, PartialOrd, Ord)]
15
+ pub enum ResultsLevel {
16
+ /// Failures only, without message text (the lowest overhead).
17
+ Basic,
18
+ /// Failures only, with message text.
19
+ Detailed,
20
+ /// Every evaluation, passing and failing, with message text, including annotations.
21
+ Verbose,
22
+ }
23
+
24
+ /// One result row.
25
+ #[derive(Clone, Debug, PartialEq, Eq)]
26
+ pub struct SchemaResult {
27
+ pub is_match: bool,
28
+ /// The message, or `""` when the level records none or the keyword has none. Annotation rows carry raw JSON.
29
+ pub message: String,
30
+ /// The path of keywords from the root schema (e.g. `/properties/name/type`).
31
+ pub evaluation_location: String,
32
+ /// The JSON pointer of the evaluated schema (or keyword) within its document (e.g. `/properties/name/type`).
33
+ pub schema_evaluation_location: String,
34
+ /// The JSON pointer of the instance location (e.g. `/name`).
35
+ pub document_evaluation_location: String,
36
+ }
37
+
38
+ /// A message for a result row, produced only when the level records message text.
39
+ pub(crate) enum Message<'a> {
40
+ None,
41
+ Static(&'static str),
42
+ Lazy(&'a dyn Fn() -> String),
43
+ }
44
+
45
+ struct Frame {
46
+ eval_len: usize,
47
+ schema_path: String,
48
+ doc_len: usize,
49
+ commit_index: usize,
50
+ rows_start: usize,
51
+ }
52
+
53
+ /// Collects the results of an evaluation (`JsonSchemaResultsCollector`).
54
+ pub struct JsonSchemaResultsCollector {
55
+ level: ResultsLevel,
56
+ committed: Vec<SchemaResult>,
57
+ frames: Vec<Frame>,
58
+ /// The rows written into open frames (each frame owns the tail from its `rows_start`).
59
+ pending: Vec<SchemaResult>,
60
+ eval_path: String,
61
+ schema_path: String,
62
+ doc_path: String,
63
+ }
64
+
65
+ /// Encodes a JSON pointer segment (`~` as `~0`, `/` as `~1`).
66
+ pub fn encode_pointer_segment(segment: &str) -> std::borrow::Cow<'_, str> {
67
+ crate::uri::escape_pointer_token(segment)
68
+ }
69
+
70
+ impl JsonSchemaResultsCollector {
71
+ /// Creates a collector at the given level.
72
+ pub fn new(level: ResultsLevel) -> Self {
73
+ JsonSchemaResultsCollector {
74
+ level,
75
+ committed: Vec::new(),
76
+ frames: Vec::new(),
77
+ pending: Vec::new(),
78
+ eval_path: String::new(),
79
+ schema_path: String::new(),
80
+ doc_path: String::new(),
81
+ }
82
+ }
83
+
84
+ pub fn level(&self) -> ResultsLevel {
85
+ self.level
86
+ }
87
+
88
+ /// The results, in commit order.
89
+ pub fn results(&self) -> &[SchemaResult] {
90
+ &self.committed
91
+ }
92
+
93
+ /// Removes and returns the results.
94
+ pub fn take_results(&mut self) -> Vec<SchemaResult> {
95
+ std::mem::take(&mut self.committed)
96
+ }
97
+
98
+ // ------------------------------------------------------------------------------------------------------------
99
+ // The evaluator's side
100
+
101
+ /// Opens a child context. The evaluation path is extended by `eval_segment` (verbatim), the schema path is
102
+ /// replaced by `schema_location`, and the document path is extended by `doc_segment` (already pointer-encoded).
103
+ pub(crate) fn begin_child_context(
104
+ &mut self,
105
+ eval_segment: Option<&str>,
106
+ schema_location: Option<&str>,
107
+ doc_segment: Option<&str>,
108
+ ) {
109
+ self.frames.push(Frame {
110
+ eval_len: self.eval_path.len(),
111
+ schema_path: self.schema_path.clone(),
112
+ doc_len: self.doc_path.len(),
113
+ commit_index: self.committed.len(),
114
+ rows_start: self.pending.len(),
115
+ });
116
+ if let Some(s) = eval_segment {
117
+ self.eval_path.push('/');
118
+ self.eval_path.push_str(s);
119
+ }
120
+ if let Some(s) = schema_location {
121
+ self.schema_path.clear();
122
+ self.schema_path.push_str(s);
123
+ }
124
+ if let Some(s) = doc_segment {
125
+ self.doc_path.push('/');
126
+ self.doc_path.push_str(s);
127
+ }
128
+ }
129
+
130
+ /// Closes a child context. When the parent does not need the child's results (`parent_is_match`) they are
131
+ /// discarded below Verbose; otherwise the context's summary row is written and its rows are committed.
132
+ pub(crate) fn commit_child_context(&mut self, parent_is_match: bool, child_is_match: bool, message: Message<'_>) {
133
+ if parent_is_match && self.level != ResultsLevel::Verbose {
134
+ self.pop_child_context();
135
+ return;
136
+ }
137
+ let row =
138
+ self.row(child_is_match, message, self.eval_path.clone(), self.schema_path.clone(), self.doc_path.clone());
139
+ self.pending.push(row);
140
+ let frame = self.frames.pop().unwrap();
141
+ let rows = self.pending.split_off(frame.rows_start);
142
+ self.committed.extend(rows.into_iter().rev());
143
+ self.restore(frame);
144
+ }
145
+
146
+ /// Closes a child context and discards everything it and its descendants wrote.
147
+ pub(crate) fn pop_child_context(&mut self) {
148
+ let frame = self.frames.pop().unwrap();
149
+ self.committed.truncate(frame.commit_index);
150
+ self.pending.truncate(frame.rows_start);
151
+ self.restore(frame);
152
+ }
153
+
154
+ pub(crate) fn evaluated_keyword(&mut self, is_match: bool, message: Message<'_>, keyword: &str) {
155
+ if !is_match || self.level == ResultsLevel::Verbose {
156
+ let k = encode_pointer_segment(keyword);
157
+ let row = self.row(
158
+ is_match,
159
+ message,
160
+ format!("{}/{k}", self.eval_path),
161
+ format!("{}/{k}", self.schema_path),
162
+ self.doc_path.clone(),
163
+ );
164
+ self.pending.push(row);
165
+ }
166
+ }
167
+
168
+ pub(crate) fn evaluated_keyword_for_property(
169
+ &mut self,
170
+ is_match: bool,
171
+ message: Message<'_>,
172
+ property_name: &str,
173
+ keyword: &str,
174
+ ) {
175
+ if !is_match || self.level == ResultsLevel::Verbose {
176
+ let k = encode_pointer_segment(keyword);
177
+ let row = self.row(
178
+ is_match,
179
+ message,
180
+ format!("{}/{k}", self.eval_path),
181
+ format!("{}/{k}", self.schema_path),
182
+ format!("{}/{}", self.doc_path, encode_pointer_segment(property_name)),
183
+ );
184
+ self.pending.push(row);
185
+ }
186
+ }
187
+
188
+ /// An annotation: Verbose only; the keyword extends the evaluation path but not the schema path.
189
+ pub(crate) fn ignored_keyword(&mut self, message: Message<'_>, keyword: &str) {
190
+ if self.level == ResultsLevel::Verbose {
191
+ let row = self.row(
192
+ true,
193
+ message,
194
+ format!("{}/{}", self.eval_path, encode_pointer_segment(keyword)),
195
+ self.schema_path.clone(),
196
+ self.doc_path.clone(),
197
+ );
198
+ self.pending.push(row);
199
+ }
200
+ }
201
+
202
+ pub(crate) fn evaluated_boolean_schema(&mut self, is_match: bool) {
203
+ if !is_match || self.level == ResultsLevel::Verbose {
204
+ let row = self.row(
205
+ is_match,
206
+ Message::None,
207
+ self.eval_path.clone(),
208
+ self.schema_path.clone(),
209
+ self.doc_path.clone(),
210
+ );
211
+ self.pending.push(row);
212
+ }
213
+ }
214
+
215
+ fn row(
216
+ &self,
217
+ is_match: bool,
218
+ message: Message<'_>,
219
+ evaluation_location: String,
220
+ schema_evaluation_location: String,
221
+ document_evaluation_location: String,
222
+ ) -> SchemaResult {
223
+ let with_text = self.level == ResultsLevel::Verbose || (!is_match && self.level >= ResultsLevel::Detailed);
224
+ let message = if with_text {
225
+ match message {
226
+ Message::None => String::new(),
227
+ Message::Static(s) => s.to_string(),
228
+ Message::Lazy(f) => f(),
229
+ }
230
+ } else {
231
+ String::new()
232
+ };
233
+ SchemaResult {
234
+ is_match,
235
+ message,
236
+ evaluation_location,
237
+ schema_evaluation_location,
238
+ document_evaluation_location,
239
+ }
240
+ }
241
+
242
+ fn restore(&mut self, frame: Frame) {
243
+ self.eval_path.truncate(frame.eval_len);
244
+ self.schema_path = frame.schema_path;
245
+ self.doc_path.truncate(frame.doc_len);
246
+ }
247
+ }
248
+
249
+ /// An annotation extracted from verbose results.
250
+ #[derive(Clone, Debug, PartialEq, Eq)]
251
+ pub struct Annotation {
252
+ /// The instance location (JSON pointer).
253
+ pub instance_location: String,
254
+ pub keyword: String,
255
+ /// The JSON pointer of the schema object that holds the keyword.
256
+ pub schema_location: String,
257
+ /// The annotation value as JSON text.
258
+ pub value: String,
259
+ }
260
+
261
+ /// The annotations in a verbose collector's results (`JsonSchemaAnnotationProducer.EnumerateAnnotations`).
262
+ pub fn enumerate_annotations(collector: &JsonSchemaResultsCollector) -> impl Iterator<Item = Annotation> + '_ {
263
+ collector.results().iter().filter_map(|r| {
264
+ if !r.is_match || r.message.is_empty() {
265
+ return None;
266
+ }
267
+ let slash = r.evaluation_location.rfind('/')?;
268
+ if r.evaluation_location == r.schema_evaluation_location {
269
+ return None;
270
+ }
271
+ let keyword = &r.evaluation_location[slash + 1..];
272
+ let first = r.message.as_bytes()[0];
273
+ if keyword.is_empty()
274
+ || !(matches!(first, b'"' | b'{' | b'[' | b't' | b'f' | b'n' | b'-') || first.is_ascii_digit())
275
+ {
276
+ return None;
277
+ }
278
+ Some(Annotation {
279
+ instance_location: r.document_evaluation_location.clone(),
280
+ keyword: keyword.to_string(),
281
+ schema_location: r.schema_evaluation_location.clone(),
282
+ value: r.message.clone(),
283
+ })
284
+ })
285
+ }
286
+
287
+ /// `#` followed by the schema location, percent-encoded as a URI fragment (upper-case hex, UTF-8).
288
+ pub fn schema_location_fragment(schema_location: &str) -> String {
289
+ let mut out = String::from("#");
290
+ for &byte in schema_location.as_bytes() {
291
+ let c = byte as char;
292
+ if byte < 128 && (c.is_ascii_alphanumeric() || "-._~!$&'()*+,;=:@/?".contains(c)) {
293
+ out.push(c);
294
+ } else {
295
+ out.push_str(&format!("%{byte:02X}"));
296
+ }
297
+ }
298
+ out
299
+ }
300
+
301
+ /// Annotations grouped by instance location, then keyword, then schema location fragment, with parsed values
302
+ /// (`JsonSchemaAnnotationProducer.WriteAnnotationsTo`): `{ "/name": { "title": { "#/properties/name": "Name" } } }`.
303
+ pub fn collect_annotations(
304
+ collector: &JsonSchemaResultsCollector,
305
+ ) -> BTreeMap<String, BTreeMap<String, BTreeMap<String, Value>>> {
306
+ let mut out: BTreeMap<String, BTreeMap<String, BTreeMap<String, Value>>> = BTreeMap::new();
307
+ for a in enumerate_annotations(collector) {
308
+ let value = serde_json::from_str(&a.value).unwrap_or(Value::String(a.value.clone()));
309
+ out.entry(a.instance_location)
310
+ .or_default()
311
+ .entry(a.keyword)
312
+ .or_default()
313
+ .insert(schema_location_fragment(&a.schema_location), value);
314
+ }
315
+ out
316
+ }
@@ -0,0 +1,274 @@
1
+ //! URI handling for schema identification and reference resolution (RFC 3986 section 5).
2
+ //!
3
+ //! Mirrors the TypeScript port's `uri.ts`: reference resolution is implemented directly so that opaque bases
4
+ //! (`urn:`, `tag:`) resolve the same way everywhere.
5
+
6
+ use serde_json::Value;
7
+
8
+ struct UriParts<'a> {
9
+ scheme: Option<&'a str>,
10
+ authority: Option<&'a str>,
11
+ path: &'a str,
12
+ query: Option<&'a str>,
13
+ }
14
+
15
+ /// Splits a URI into scheme, authority, path and query (the fragment must already be removed).
16
+ fn parse(uri: &str) -> UriParts<'_> {
17
+ let mut rest = uri;
18
+ let mut scheme = None;
19
+ if let Some(colon) = scheme_end(rest) {
20
+ scheme = Some(&rest[..colon]);
21
+ rest = &rest[colon + 1..];
22
+ }
23
+ let mut authority = None;
24
+ if let Some(after) = rest.strip_prefix("//") {
25
+ let end = after.find(['/', '?']).unwrap_or(after.len());
26
+ authority = Some(&after[..end]);
27
+ rest = &after[end..];
28
+ }
29
+ let (path, query) = match rest.find('?') {
30
+ Some(q) => (&rest[..q], Some(&rest[q + 1..])),
31
+ None => (rest, None),
32
+ };
33
+ UriParts { scheme, authority, path, query }
34
+ }
35
+
36
+ /// The index of the ':' ending a URI scheme, if the text starts with one.
37
+ fn scheme_end(s: &str) -> Option<usize> {
38
+ let bytes = s.as_bytes();
39
+ if bytes.is_empty() || !bytes[0].is_ascii_alphabetic() {
40
+ return None;
41
+ }
42
+ for (i, &b) in bytes.iter().enumerate().skip(1) {
43
+ if b == b':' {
44
+ return Some(i);
45
+ }
46
+ if !(b.is_ascii_alphanumeric() || b == b'+' || b == b'-' || b == b'.') {
47
+ return None;
48
+ }
49
+ }
50
+ None
51
+ }
52
+
53
+ /// True when the reference starts with a URI scheme.
54
+ pub(crate) fn has_scheme(reference: &str) -> bool {
55
+ scheme_end(reference).is_some()
56
+ }
57
+
58
+ /// Splits a reference at its first '#'.
59
+ pub(crate) fn split(reference: &str) -> (&str, &str) {
60
+ match reference.find('#') {
61
+ Some(i) => (&reference[..i], &reference[i + 1..]),
62
+ None => (reference, ""),
63
+ }
64
+ }
65
+
66
+ fn format(scheme: Option<&str>, authority: Option<&str>, path: &str, query: Option<&str>) -> String {
67
+ let mut s = String::with_capacity(path.len() + 32);
68
+ if let Some(sc) = scheme {
69
+ s.push_str(sc);
70
+ s.push(':');
71
+ }
72
+ if let Some(a) = authority {
73
+ s.push_str("//");
74
+ s.push_str(a);
75
+ }
76
+ s.push_str(path);
77
+ if let Some(q) = query {
78
+ s.push('?');
79
+ s.push_str(q);
80
+ }
81
+ s
82
+ }
83
+
84
+ fn remove_dot_segments(path: &str) -> String {
85
+ if !path.contains('.') {
86
+ return path.to_string();
87
+ }
88
+ let input: Vec<&str> = path.split('/').collect();
89
+ let mut out: Vec<&str> = Vec::with_capacity(input.len());
90
+ let last = input.len() - 1;
91
+ for (i, seg) in input.iter().enumerate() {
92
+ match *seg {
93
+ "." => {
94
+ if i == last {
95
+ out.push("");
96
+ }
97
+ }
98
+ ".." => {
99
+ if out.len() > 1 || (out.len() == 1 && !out[0].is_empty()) {
100
+ out.pop();
101
+ }
102
+ if i == last {
103
+ out.push("");
104
+ }
105
+ }
106
+ s => out.push(s),
107
+ }
108
+ }
109
+ out.join("/")
110
+ }
111
+
112
+ fn merge(base: &UriParts<'_>, ref_path: &str) -> String {
113
+ if base.authority.is_some() && base.path.is_empty() {
114
+ return format!("/{ref_path}");
115
+ }
116
+ match base.path.rfind('/') {
117
+ Some(i) => format!("{}{}", &base.path[..=i], ref_path),
118
+ None => ref_path.to_string(),
119
+ }
120
+ }
121
+
122
+ fn normalize_parts(scheme: Option<&str>, authority: Option<&str>, path: &str, query: Option<&str>) -> String {
123
+ let scheme = scheme.map(|s| s.to_ascii_lowercase());
124
+ let mut path = remove_dot_segments(path);
125
+ let authority = authority.map(|a| {
126
+ let mut a = a.to_ascii_lowercase();
127
+ let default_port = match scheme.as_deref() {
128
+ Some("http") => Some(":80"),
129
+ Some("https") => Some(":443"),
130
+ _ => None,
131
+ };
132
+ if let Some(port) = default_port {
133
+ if a.ends_with(port) {
134
+ a.truncate(a.len() - port.len());
135
+ }
136
+ }
137
+ if path.is_empty() {
138
+ path = "/".to_string();
139
+ }
140
+ a
141
+ });
142
+ format(scheme.as_deref(), authority.as_deref(), &path, query)
143
+ }
144
+
145
+ /// Normalises an absolute URI (dropping its fragment) so that equivalent spellings compare equal.
146
+ pub(crate) fn normalize(uri: &str) -> String {
147
+ let (uri_part, _) = split(uri);
148
+ if !has_scheme(uri_part) {
149
+ return uri_part.to_string();
150
+ }
151
+ let p = parse(uri_part);
152
+ normalize_parts(p.scheme, p.authority, p.path, p.query)
153
+ }
154
+
155
+ /// Resolves a reference (without fragment) against a base URI, returning the normalised absolute URI.
156
+ pub(crate) fn resolve(base_uri: &str, reference: &str) -> String {
157
+ if reference.is_empty() {
158
+ return base_uri.to_string();
159
+ }
160
+ let r = parse(reference);
161
+ if r.scheme.is_some() {
162
+ return normalize_parts(r.scheme, r.authority, r.path, r.query);
163
+ }
164
+ if base_uri.is_empty() {
165
+ return reference.to_string();
166
+ }
167
+ let b = parse(base_uri);
168
+ let (authority, path, query);
169
+ if r.authority.is_some() {
170
+ authority = r.authority;
171
+ path = remove_dot_segments(r.path);
172
+ query = r.query;
173
+ } else {
174
+ if r.path.is_empty() {
175
+ path = b.path.to_string();
176
+ query = if r.query.is_some() { r.query } else { b.query };
177
+ } else {
178
+ path = if r.path.starts_with('/') {
179
+ remove_dot_segments(r.path)
180
+ } else {
181
+ remove_dot_segments(&merge(&b, r.path))
182
+ };
183
+ query = r.query;
184
+ }
185
+ authority = b.authority;
186
+ }
187
+ if b.scheme.is_some() {
188
+ normalize_parts(b.scheme, authority, &path, query)
189
+ } else {
190
+ format(None, authority, &path, query)
191
+ }
192
+ }
193
+
194
+ /// Percent-decodes a fragment (invalid escapes leave the text unchanged).
195
+ pub(crate) fn decode_fragment(fragment: &str) -> String {
196
+ if !fragment.contains('%') {
197
+ return fragment.to_string();
198
+ }
199
+ let bytes = fragment.as_bytes();
200
+ let mut out = Vec::with_capacity(bytes.len());
201
+ let mut i = 0;
202
+ while i < bytes.len() {
203
+ if bytes[i] == b'%' {
204
+ let hex = |c: u8| (c as char).to_digit(16);
205
+ match (bytes.get(i + 1).and_then(|&c| hex(c)), bytes.get(i + 2).and_then(|&c| hex(c))) {
206
+ (Some(h), Some(l)) => {
207
+ out.push((h * 16 + l) as u8);
208
+ i += 3;
209
+ continue;
210
+ }
211
+ _ => return fragment.to_string(),
212
+ }
213
+ }
214
+ out.push(bytes[i]);
215
+ i += 1;
216
+ }
217
+ String::from_utf8(out).unwrap_or_else(|_| fragment.to_string())
218
+ }
219
+
220
+ /// Escapes a JSON pointer token (`~` as `~0`, `/` as `~1`).
221
+ pub(crate) fn escape_pointer_token(token: &str) -> std::borrow::Cow<'_, str> {
222
+ if !token.contains(['~', '/']) {
223
+ return std::borrow::Cow::Borrowed(token);
224
+ }
225
+ std::borrow::Cow::Owned(token.replace('~', "~0").replace('/', "~1"))
226
+ }
227
+
228
+ /// Resolves an RFC 6901 JSON pointer against a JSON value, returning the value and its normalised pointer.
229
+ pub(crate) fn resolve_pointer<'a>(root: &'a Value, pointer: &str) -> Option<(&'a Value, String)> {
230
+ if pointer.is_empty() {
231
+ return Some((root, String::new()));
232
+ }
233
+ let rest = pointer.strip_prefix('/')?;
234
+ let mut current = root;
235
+ let mut path = String::with_capacity(pointer.len());
236
+ for raw in rest.split('/') {
237
+ let token = raw.replace("~1", "/").replace("~0", "~");
238
+ match current {
239
+ Value::Array(items) => {
240
+ let valid = token == "0"
241
+ || (!token.is_empty() && !token.starts_with('0') && token.bytes().all(|b| b.is_ascii_digit()));
242
+ if !valid {
243
+ return None;
244
+ }
245
+ let i: usize = token.parse().ok()?;
246
+ current = items.get(i)?;
247
+ path.push('/');
248
+ path.push_str(&token);
249
+ }
250
+ Value::Object(map) => {
251
+ current = map.get(&token)?;
252
+ path.push('/');
253
+ path.push_str(&escape_pointer_token(&token));
254
+ }
255
+ _ => return None,
256
+ }
257
+ }
258
+ Some((current, path))
259
+ }
260
+
261
+ #[cfg(test)]
262
+ mod tests {
263
+ use super::*;
264
+
265
+ #[test]
266
+ fn resolves_relative_references() {
267
+ assert_eq!(resolve("http://example.com/a/b.json", "c.json"), "http://example.com/a/c.json");
268
+ assert_eq!(resolve("http://example.com/a/b.json", "../c.json"), "http://example.com/c.json");
269
+ assert_eq!(resolve("http://Example.com:80", ""), "http://Example.com:80");
270
+ assert_eq!(normalize("HTTP://Example.COM:80#frag"), "http://example.com/");
271
+ assert_eq!(resolve("urn:uuid:deadbeef-1234", ""), "urn:uuid:deadbeef-1234");
272
+ assert_eq!(resolve("tag:example.com,2021:a", "b"), "tag:b");
273
+ }
274
+ }
@@ -0,0 +1,5 @@
1
+ # frozen_string_literal: true
2
+
3
+ module CorvusJsonSchema
4
+ VERSION = "0.1.0"
5
+ end
@@ -0,0 +1,65 @@
1
+ # frozen_string_literal: true
2
+
3
+ # A high-performance JSON Schema evaluator (draft 4, 6, 7, 2019-09 and 2020-12) for Ruby, backed by the
4
+ # corvus-json-schema Rust crate, with results collection and annotations.
5
+ #
6
+ # validator = CorvusJsonSchema.compile({ "type" => "object", "required" => ["id"] })
7
+ # validator.valid?({ "id" => 3 }) # => true
8
+ # validator.valid_json?('{"id": 3}') # => true, parsed in Rust
9
+ #
10
+ # Values are read in place: Hash (String or Symbol keys), Array, String, Symbol (as its name), Integer, Float, true,
11
+ # false and nil. Integers beyond 64 bits become the nearest double. A value the schema examines that is none of these
12
+ # raises (TypeError; ArgumentError for NaN and infinities; EncodingError for a String that is not valid UTF-8); values
13
+ # are read only as the schema examines them.
14
+ require_relative "corvus_json_schema/version"
15
+
16
+ begin
17
+ RUBY_VERSION =~ /(\d+\.\d+)/
18
+ require_relative "corvus_json_schema/#{Regexp.last_match(1)}/corvus_json_schema"
19
+ rescue LoadError
20
+ require_relative "corvus_json_schema/corvus_json_schema"
21
+ end
22
+
23
+ module CorvusJsonSchema
24
+ DIALECTS = {
25
+ draft4: 4, draft6: 6, draft7: 7, draft201909: 2019, draft202012: 2020
26
+ }.freeze
27
+
28
+ RESULTS_LEVELS = { basic: 0, detailed: 1, verbose: 2 }.freeze
29
+
30
+ # Compiles a schema (a Hash or Array, true or false, or JSON text) into a Validator.
31
+ #
32
+ # Options: default_dialect (:draft4, :draft6, :draft7, :draft201909, :draft202012; the dialect of schemas without
33
+ # $schema), assert_format (nil follows the vocabularies), assert_format_in_legacy_drafts, assert_content, formats (a
34
+ # Hash of format name to a callable returning whether a string is valid), resolver (a callable from an absolute URI
35
+ # to the document, as a Hash or JSON text, or nil when unknown), base_uri, entry_point (a reference such as
36
+ # "#/$defs/item") and max_depth (of in-place recursion, default 128).
37
+ def self.compile(schema, default_dialect: :draft202012, assert_format: nil, assert_format_in_legacy_drafts: false,
38
+ assert_content: true, formats: nil, resolver: nil, base_uri: nil, entry_point: nil, max_depth: 128)
39
+ dialect = DIALECTS.fetch(default_dialect) { raise ArgumentError, "unknown dialect #{default_dialect.inspect}" }
40
+ Native.compile(schema, dialect, assert_format, assert_format_in_legacy_drafts, assert_content, formats, resolver,
41
+ base_uri, entry_point, Integer(max_depth))
42
+ end
43
+
44
+ # The version of the corvus-json-schema crate the extension was built from.
45
+ def self.crate_version
46
+ Native.crate_version
47
+ end
48
+
49
+ # A compiled schema: valid?(value), valid_json?(text) and evaluate(value, collector).
50
+ class Validator
51
+ # Evaluates the value, replacing the collector's results with this evaluation's (every keyword is evaluated and
52
+ # reported at the collector's level). Returns whether the value is valid.
53
+ def evaluate(value, collector)
54
+ native_evaluate(value, collector)
55
+ end
56
+ end
57
+
58
+ # The results of the latest evaluation into it: results (rows as Hashes) and annotations (from a :verbose
59
+ # evaluation, grouped by instance location, keyword and schema location).
60
+ class Collector
61
+ def self.new(level = :basic)
62
+ native_new(RESULTS_LEVELS.fetch(level) { raise ArgumentError, "unknown results level #{level.inspect}" })
63
+ end
64
+ end
65
+ end