quantdiff 0.1.0rc1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- quantdiff/__init__.py +53 -0
- quantdiff/__main__.py +5 -0
- quantdiff/_http.py +151 -0
- quantdiff/_text.py +13 -0
- quantdiff/_version.py +1 -0
- quantdiff/api.py +340 -0
- quantdiff/backends/__init__.py +28 -0
- quantdiff/backends/_common.py +342 -0
- quantdiff/backends/base.py +91 -0
- quantdiff/backends/llamacpp.py +428 -0
- quantdiff/backends/ollama.py +359 -0
- quantdiff/backends/openai_compat.py +338 -0
- quantdiff/cache.py +240 -0
- quantdiff/card.py +1664 -0
- quantdiff/cli.py +377 -0
- quantdiff/discover.py +488 -0
- quantdiff/errors.py +45 -0
- quantdiff/metrics/__init__.py +36 -0
- quantdiff/metrics/codeexec.py +428 -0
- quantdiff/metrics/jsonschema.py +610 -0
- quantdiff/metrics/logit.py +214 -0
- quantdiff/metrics/tasks.py +114 -0
- quantdiff/metrics/textsim.py +66 -0
- quantdiff/metrics/toolcheck.py +99 -0
- quantdiff/png.py +360 -0
- quantdiff/preflight.py +365 -0
- quantdiff/progress.py +283 -0
- quantdiff/py.typed +0 -0
- quantdiff/report.py +780 -0
- quantdiff/runner.py +492 -0
- quantdiff/spec.py +154 -0
- quantdiff/stats.py +226 -0
- quantdiff/suites/__init__.py +462 -0
- quantdiff/suites/data/chat.jsonl +22 -0
- quantdiff/suites/data/code.jsonl +32 -0
- quantdiff/suites/data/json.jsonl +34 -0
- quantdiff/suites/data/scoring.jsonl +41 -0
- quantdiff/suites/data/tools.jsonl +32 -0
- quantdiff/types.py +322 -0
- quantdiff/verdict.py +1513 -0
- quantdiff-0.1.0rc1.dist-info/METADATA +514 -0
- quantdiff-0.1.0rc1.dist-info/RECORD +45 -0
- quantdiff-0.1.0rc1.dist-info/WHEEL +4 -0
- quantdiff-0.1.0rc1.dist-info/entry_points.txt +2 -0
- quantdiff-0.1.0rc1.dist-info/licenses/LICENSE +202 -0
|
@@ -0,0 +1,610 @@
|
|
|
1
|
+
"""A small JSON Schema validator for the keyword subset quantdiff suites use.
|
|
2
|
+
|
|
3
|
+
Suite loaders call `check_schema` on every schema they read. It rejects any keyword outside
|
|
4
|
+
`SUPPORTED_KEYWORDS`, so a schema is never silently checked only in part, and any malformed
|
|
5
|
+
keyword value, so suite bugs surface at load time. `validate` raises SuiteError for the same
|
|
6
|
+
problems when handed a schema that skipped that check.
|
|
7
|
+
|
|
8
|
+
`pattern` follows ECMA 262 regular expressions where Python's dialect differs in ways a
|
|
9
|
+
schema author would not expect: `\\d`, `\\w` and `\\b` match ASCII only, and `$` matches only
|
|
10
|
+
at the very end of the string (Python's `$` also matches before a trailing newline).
|
|
11
|
+
Patterns that repeat a group which itself contains a quantifier or an alternation, such as
|
|
12
|
+
`(a+)+` or `(a|aa)*`, are refused because they can backtrack for exponential time.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import functools
|
|
18
|
+
import json
|
|
19
|
+
import math
|
|
20
|
+
import re
|
|
21
|
+
from collections.abc import Callable, Iterator, Mapping
|
|
22
|
+
from typing import Final
|
|
23
|
+
|
|
24
|
+
from quantdiff.errors import SuiteError
|
|
25
|
+
from quantdiff.metrics.textsim import strip_reasoning
|
|
26
|
+
from quantdiff.types import JSONValue
|
|
27
|
+
|
|
28
|
+
MAX_PATTERN_INPUT: Final = 10_000
|
|
29
|
+
"""Longer strings are not matched against `pattern`.
|
|
30
|
+
|
|
31
|
+
This bounds the regex engine's input. It does not by itself bound backtracking time; refusing
|
|
32
|
+
nested quantifiers does that for the exponential cases.
|
|
33
|
+
"""
|
|
34
|
+
|
|
35
|
+
_SUBSCHEMA_KEYWORDS: Final = ("additionalProperties", "items")
|
|
36
|
+
_SCHEMA_LIST_KEYWORDS: Final = ("anyOf", "oneOf", "allOf")
|
|
37
|
+
_FENCE: Final = re.compile(r"```[ \t]*([A-Za-z]*)[ \t]*\r?\n(.*?)```", re.DOTALL)
|
|
38
|
+
_JSON_FENCE_TAGS: Final = frozenset({"", "json"})
|
|
39
|
+
|
|
40
|
+
_JSON_TYPE_NAMES: Final = (
|
|
41
|
+
(bool, "boolean"),
|
|
42
|
+
(int, "integer"),
|
|
43
|
+
(float, "number"),
|
|
44
|
+
(str, "string"),
|
|
45
|
+
(list, "array"),
|
|
46
|
+
(dict, "object"),
|
|
47
|
+
)
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def _is_number(value: JSONValue) -> bool:
|
|
51
|
+
return isinstance(value, (int, float)) and not isinstance(value, bool)
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def _is_integer(value: JSONValue) -> bool:
|
|
55
|
+
"""JSON has one number type, so 3.0 is an integer; float() would overflow on big ints."""
|
|
56
|
+
if isinstance(value, float):
|
|
57
|
+
return value.is_integer()
|
|
58
|
+
return isinstance(value, int) and not isinstance(value, bool)
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
_TYPE_CHECKS: Final[Mapping[str, Callable[[JSONValue], bool]]] = {
|
|
62
|
+
"null": lambda value: value is None,
|
|
63
|
+
"boolean": lambda value: isinstance(value, bool),
|
|
64
|
+
"integer": _is_integer,
|
|
65
|
+
"number": _is_number,
|
|
66
|
+
"string": lambda value: isinstance(value, str),
|
|
67
|
+
"array": lambda value: isinstance(value, list),
|
|
68
|
+
"object": lambda value: isinstance(value, dict),
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
# Regular expression tokens, as produced by _regex_tokens.
|
|
72
|
+
_ATOM: Final = "atom"
|
|
73
|
+
_OPEN: Final = "open"
|
|
74
|
+
_CLOSE: Final = "close"
|
|
75
|
+
_ALTERNATION: Final = "alternation"
|
|
76
|
+
_QUANTIFIER: Final = "quantifier"
|
|
77
|
+
_END_ANCHOR: Final = "end"
|
|
78
|
+
_QUANTIFIER_SYNTAX: Final = re.compile(r"(?:[*+?]|\{(?:\d+(?:,\d*)?|,\d*)\})[?+]?")
|
|
79
|
+
_GROUP_START: Final = re.compile(
|
|
80
|
+
r"\((?:\?(?:P<\w+>|<\w+>|<[=!]|[:=!>]|[aiLmsux]*(?:-[imsx]+)?:?))?"
|
|
81
|
+
)
|
|
82
|
+
_GROUP_ATOM: Final = re.compile(r"\(\?(?:P=\w+|#[^)]*)\)")
|
|
83
|
+
_SINGLE_CHAR_TOKENS: Final = {")": _CLOSE, "|": _ALTERNATION, "$": _END_ANCHOR}
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
class _StrictJSONError(ValueError):
|
|
87
|
+
"""Input that Python's json module accepts but JSON does not allow."""
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
class _NestedQuantifierError(Exception):
|
|
91
|
+
"""A pattern repeats a group that can match the same text in more than one way."""
|
|
92
|
+
|
|
93
|
+
def __init__(self, group: str) -> None:
|
|
94
|
+
super().__init__(group)
|
|
95
|
+
self.group = group
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
# Schema checks ---------------------------------------------------------------------------
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def check_schema(schema: JSONValue, where: str = "schema") -> None:
|
|
102
|
+
"""Raise SuiteError unless `schema` uses only supported keywords with valid values.
|
|
103
|
+
|
|
104
|
+
`where` names the schema in error messages; nested locations are appended to it.
|
|
105
|
+
"""
|
|
106
|
+
if isinstance(schema, bool):
|
|
107
|
+
return
|
|
108
|
+
if not isinstance(schema, dict):
|
|
109
|
+
raise SuiteError(f"{where} must be a JSON object or boolean")
|
|
110
|
+
for key, value in schema.items():
|
|
111
|
+
check = _KEYWORD_CHECKS.get(key)
|
|
112
|
+
if check is None:
|
|
113
|
+
raise SuiteError(f"{where} uses unsupported JSON Schema keyword {key!r}")
|
|
114
|
+
check(value, f"{where}.{key}")
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def _schema_properties(value: JSONValue, where: str) -> None:
|
|
118
|
+
if not isinstance(value, dict):
|
|
119
|
+
raise SuiteError(f"{where} must be an object")
|
|
120
|
+
for name, subschema in value.items():
|
|
121
|
+
check_schema(subschema, f"{where}.{name}")
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def _schema_list(value: JSONValue, where: str) -> None:
|
|
125
|
+
if not isinstance(value, list) or not value:
|
|
126
|
+
raise SuiteError(f"{where} must be a non-empty array of schemas")
|
|
127
|
+
for index, subschema in enumerate(value):
|
|
128
|
+
check_schema(subschema, f"{where}[{index}]")
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def _schema_type(value: JSONValue, where: str) -> None:
|
|
132
|
+
names = value if isinstance(value, list) else [value]
|
|
133
|
+
if not names or not all(isinstance(name, str) and name in _TYPE_CHECKS for name in names):
|
|
134
|
+
raise SuiteError(f"{where} must be one of {sorted(_TYPE_CHECKS)} or a list of them")
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
def _schema_required(value: JSONValue, where: str) -> None:
|
|
138
|
+
if not isinstance(value, list) or not all(isinstance(name, str) for name in value):
|
|
139
|
+
raise SuiteError(f"{where} must be an array of strings")
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def _schema_enum(value: JSONValue, where: str) -> None:
|
|
143
|
+
if not isinstance(value, list) or not value:
|
|
144
|
+
raise SuiteError(f"{where} must be a non-empty array")
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def _schema_number(value: JSONValue, where: str) -> None:
|
|
148
|
+
if not _is_number(value):
|
|
149
|
+
raise SuiteError(f"{where} must be a number")
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def _schema_count(value: JSONValue, where: str) -> None:
|
|
153
|
+
if isinstance(value, bool) or not isinstance(value, int) or value < 0:
|
|
154
|
+
raise SuiteError(f"{where} must be a non-negative integer")
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
def _schema_pattern(value: JSONValue, where: str) -> None:
|
|
158
|
+
if not isinstance(value, str):
|
|
159
|
+
raise SuiteError(f"{where} must be a string")
|
|
160
|
+
_compile(value, where)
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
def _schema_string(value: JSONValue, where: str) -> None:
|
|
164
|
+
if not isinstance(value, str):
|
|
165
|
+
raise SuiteError(f"{where} must be a string")
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def _schema_any(value: JSONValue, where: str) -> None:
|
|
169
|
+
return
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
_KEYWORD_CHECKS: Final[Mapping[str, Callable[[JSONValue, str], None]]] = {
|
|
173
|
+
"type": _schema_type,
|
|
174
|
+
"properties": _schema_properties,
|
|
175
|
+
"required": _schema_required,
|
|
176
|
+
"additionalProperties": check_schema,
|
|
177
|
+
"items": check_schema,
|
|
178
|
+
"enum": _schema_enum,
|
|
179
|
+
"const": _schema_any,
|
|
180
|
+
"minLength": _schema_count,
|
|
181
|
+
"maxLength": _schema_count,
|
|
182
|
+
"minimum": _schema_number,
|
|
183
|
+
"maximum": _schema_number,
|
|
184
|
+
"exclusiveMinimum": _schema_number,
|
|
185
|
+
"exclusiveMaximum": _schema_number,
|
|
186
|
+
"minItems": _schema_count,
|
|
187
|
+
"maxItems": _schema_count,
|
|
188
|
+
"pattern": _schema_pattern,
|
|
189
|
+
"anyOf": _schema_list,
|
|
190
|
+
"oneOf": _schema_list,
|
|
191
|
+
"allOf": _schema_list,
|
|
192
|
+
"description": _schema_string,
|
|
193
|
+
"title": _schema_string,
|
|
194
|
+
"default": _schema_any,
|
|
195
|
+
"$schema": _schema_string,
|
|
196
|
+
}
|
|
197
|
+
SUPPORTED_KEYWORDS: Final = frozenset(_KEYWORD_CHECKS)
|
|
198
|
+
"""JSON Schema keywords the validator checks. Anything else is rejected up front."""
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
def unsupported_keywords(schema: JSONValue, path: str = "") -> list[str]:
|
|
202
|
+
"""Return the JSON-pointer path of every keyword this validator cannot check."""
|
|
203
|
+
if isinstance(schema, bool):
|
|
204
|
+
return []
|
|
205
|
+
if not isinstance(schema, dict):
|
|
206
|
+
return [path or "/"]
|
|
207
|
+
found = [_pointer(path, key) for key in schema if key not in SUPPORTED_KEYWORDS]
|
|
208
|
+
properties = schema.get("properties")
|
|
209
|
+
if isinstance(properties, dict):
|
|
210
|
+
for name, subschema in properties.items():
|
|
211
|
+
found += unsupported_keywords(subschema, _pointer(_pointer(path, "properties"), name))
|
|
212
|
+
for keyword in _SUBSCHEMA_KEYWORDS:
|
|
213
|
+
if keyword in schema:
|
|
214
|
+
found += unsupported_keywords(schema[keyword], _pointer(path, keyword))
|
|
215
|
+
for keyword in _SCHEMA_LIST_KEYWORDS:
|
|
216
|
+
branches = schema.get(keyword)
|
|
217
|
+
if isinstance(branches, list):
|
|
218
|
+
for index, branch in enumerate(branches):
|
|
219
|
+
found += unsupported_keywords(branch, _pointer(_pointer(path, keyword), index))
|
|
220
|
+
return found
|
|
221
|
+
|
|
222
|
+
|
|
223
|
+
# Validation ------------------------------------------------------------------------------
|
|
224
|
+
|
|
225
|
+
|
|
226
|
+
def validate(instance: JSONValue, schema: JSONValue) -> list[str]:
|
|
227
|
+
"""Return one readable error per violation, each prefixed with its path; [] if valid."""
|
|
228
|
+
return _validate(instance, schema, "")
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
def extract_json(text: str) -> tuple[JSONValue | None, str | None]:
|
|
232
|
+
"""Parse a model answer as JSON, returning (value, None) or (None, reason).
|
|
233
|
+
|
|
234
|
+
The whole answer must be JSON, or contain exactly one ```json or bare ``` fenced
|
|
235
|
+
block that parses. A leading <think> block is ignored. NaN, Infinity, numbers too large
|
|
236
|
+
for a float and duplicate object keys are rejected: Python's json module accepts them,
|
|
237
|
+
but they are not valid JSON and would let a value slip past numeric bounds.
|
|
238
|
+
"""
|
|
239
|
+
answer = strip_reasoning(text).strip()
|
|
240
|
+
if not answer:
|
|
241
|
+
return None, "response is empty"
|
|
242
|
+
value, error = _parse(answer)
|
|
243
|
+
if error is None:
|
|
244
|
+
return value, None
|
|
245
|
+
|
|
246
|
+
parsed = []
|
|
247
|
+
for match in _FENCE.finditer(answer):
|
|
248
|
+
if match.group(1).lower() not in _JSON_FENCE_TAGS:
|
|
249
|
+
continue
|
|
250
|
+
block_value, block_error = _parse(match.group(2).strip())
|
|
251
|
+
if block_error is None:
|
|
252
|
+
parsed.append(block_value)
|
|
253
|
+
if len(parsed) == 1:
|
|
254
|
+
return parsed[0], None
|
|
255
|
+
if parsed:
|
|
256
|
+
return None, f"found {len(parsed)} fenced JSON blocks, expected exactly one"
|
|
257
|
+
return None, f"invalid JSON: {error}"
|
|
258
|
+
|
|
259
|
+
|
|
260
|
+
def _parse(text: str) -> tuple[JSONValue, str | None]:
|
|
261
|
+
try:
|
|
262
|
+
value = json.loads(
|
|
263
|
+
text,
|
|
264
|
+
object_pairs_hook=_unique_keys,
|
|
265
|
+
parse_constant=_reject_constant,
|
|
266
|
+
parse_float=_finite_float,
|
|
267
|
+
)
|
|
268
|
+
except (json.JSONDecodeError, _StrictJSONError) as exc:
|
|
269
|
+
return None, str(exc)
|
|
270
|
+
except RecursionError:
|
|
271
|
+
return None, "nesting is too deep"
|
|
272
|
+
return value, None
|
|
273
|
+
|
|
274
|
+
|
|
275
|
+
def _unique_keys(pairs: list[tuple[str, JSONValue]]) -> dict[str, JSONValue]:
|
|
276
|
+
record: dict[str, JSONValue] = {}
|
|
277
|
+
for key, value in pairs:
|
|
278
|
+
if key in record:
|
|
279
|
+
raise _StrictJSONError(f"duplicate key {key!r}")
|
|
280
|
+
record[key] = value
|
|
281
|
+
return record
|
|
282
|
+
|
|
283
|
+
|
|
284
|
+
def _reject_constant(name: str) -> JSONValue:
|
|
285
|
+
raise _StrictJSONError(f"{name} is not a JSON value")
|
|
286
|
+
|
|
287
|
+
|
|
288
|
+
def _finite_float(text: str) -> float:
|
|
289
|
+
value = float(text)
|
|
290
|
+
if math.isinf(value):
|
|
291
|
+
raise _StrictJSONError(f"number {text} is too large")
|
|
292
|
+
return value
|
|
293
|
+
|
|
294
|
+
|
|
295
|
+
def _validate(instance: JSONValue, schema: JSONValue, path: str) -> list[str]:
|
|
296
|
+
if schema is True:
|
|
297
|
+
return []
|
|
298
|
+
if schema is False:
|
|
299
|
+
return [_error(path, "no value is allowed here")]
|
|
300
|
+
if not isinstance(schema, dict):
|
|
301
|
+
raise SuiteError(f"schema at {path or '/'} must be an object or boolean")
|
|
302
|
+
|
|
303
|
+
type_errors = _check_type(instance, schema, path)
|
|
304
|
+
if type_errors:
|
|
305
|
+
# Further keywords would only restate the type mismatch.
|
|
306
|
+
return type_errors
|
|
307
|
+
errors = _check_values(instance, schema, path)
|
|
308
|
+
if isinstance(instance, str):
|
|
309
|
+
errors += _check_string(instance, schema, path)
|
|
310
|
+
elif _is_number(instance):
|
|
311
|
+
errors += _check_number(instance, schema, path)
|
|
312
|
+
elif isinstance(instance, list):
|
|
313
|
+
errors += _check_array(instance, schema, path)
|
|
314
|
+
elif isinstance(instance, dict):
|
|
315
|
+
errors += _check_object(instance, schema, path)
|
|
316
|
+
return errors + _check_combinators(instance, schema, path)
|
|
317
|
+
|
|
318
|
+
|
|
319
|
+
def _check_type(instance: JSONValue, schema: dict[str, JSONValue], path: str) -> list[str]:
|
|
320
|
+
if "type" not in schema:
|
|
321
|
+
return []
|
|
322
|
+
declared = schema["type"]
|
|
323
|
+
names = declared if isinstance(declared, list) else [declared]
|
|
324
|
+
for name in names:
|
|
325
|
+
if name not in _TYPE_CHECKS:
|
|
326
|
+
raise SuiteError(f"schema at {path or '/'} has unknown type {name!r}")
|
|
327
|
+
if any(_TYPE_CHECKS[name](instance) for name in names):
|
|
328
|
+
return []
|
|
329
|
+
expected = " or ".join(names)
|
|
330
|
+
return [_error(path, f"expected {expected}, got {_json_type(instance)}")]
|
|
331
|
+
|
|
332
|
+
|
|
333
|
+
def _check_values(instance: JSONValue, schema: dict[str, JSONValue], path: str) -> list[str]:
|
|
334
|
+
errors = []
|
|
335
|
+
if "const" in schema and not _json_equal(instance, schema["const"]):
|
|
336
|
+
errors.append(_error(path, f"must equal {_preview(schema['const'])}"))
|
|
337
|
+
if "enum" in schema:
|
|
338
|
+
options = schema["enum"]
|
|
339
|
+
if not isinstance(options, list):
|
|
340
|
+
raise SuiteError(f"schema at {path or '/'} has a non-array enum")
|
|
341
|
+
if not any(_json_equal(instance, option) for option in options):
|
|
342
|
+
errors.append(_error(path, f"must be one of {_preview(options)}"))
|
|
343
|
+
return errors
|
|
344
|
+
|
|
345
|
+
|
|
346
|
+
def _check_string(instance: str, schema: dict[str, JSONValue], path: str) -> list[str]:
|
|
347
|
+
errors = []
|
|
348
|
+
min_length = _count(schema, "minLength", path)
|
|
349
|
+
max_length = _count(schema, "maxLength", path)
|
|
350
|
+
if min_length is not None and len(instance) < min_length:
|
|
351
|
+
errors.append(_error(path, f"must be at least {min_length} characters"))
|
|
352
|
+
if max_length is not None and len(instance) > max_length:
|
|
353
|
+
errors.append(_error(path, f"must be at most {max_length} characters"))
|
|
354
|
+
if "pattern" in schema:
|
|
355
|
+
pattern = schema["pattern"]
|
|
356
|
+
if not isinstance(pattern, str):
|
|
357
|
+
raise SuiteError(f"schema at {path or '/'} has a non-string pattern")
|
|
358
|
+
compiled = _compile(pattern, f"schema at {path or '/'}")
|
|
359
|
+
if len(instance) > MAX_PATTERN_INPUT:
|
|
360
|
+
errors.append(_error(path, f"is longer than {MAX_PATTERN_INPUT} characters"))
|
|
361
|
+
elif not compiled.search(instance):
|
|
362
|
+
errors.append(_error(path, f"does not match pattern {pattern!r}"))
|
|
363
|
+
return errors
|
|
364
|
+
|
|
365
|
+
|
|
366
|
+
def _check_number(instance: float, schema: dict[str, JSONValue], path: str) -> list[str]:
|
|
367
|
+
errors = []
|
|
368
|
+
bounds: tuple[tuple[str, Callable[[float, float], bool], str], ...] = (
|
|
369
|
+
("minimum", lambda value, limit: value >= limit, ">="),
|
|
370
|
+
("maximum", lambda value, limit: value <= limit, "<="),
|
|
371
|
+
("exclusiveMinimum", lambda value, limit: value > limit, ">"),
|
|
372
|
+
("exclusiveMaximum", lambda value, limit: value < limit, "<"),
|
|
373
|
+
)
|
|
374
|
+
for keyword, holds, symbol in bounds:
|
|
375
|
+
if keyword not in schema:
|
|
376
|
+
continue
|
|
377
|
+
limit = schema[keyword]
|
|
378
|
+
if not _is_number(limit):
|
|
379
|
+
raise SuiteError(f"schema at {path or '/'} has a non-numeric {keyword}")
|
|
380
|
+
if not holds(instance, limit):
|
|
381
|
+
errors.append(_error(path, f"must be {symbol} {limit}"))
|
|
382
|
+
return errors
|
|
383
|
+
|
|
384
|
+
|
|
385
|
+
def _check_array(instance: list[JSONValue], schema: dict[str, JSONValue], path: str) -> list[str]:
|
|
386
|
+
errors = []
|
|
387
|
+
min_items = _count(schema, "minItems", path)
|
|
388
|
+
max_items = _count(schema, "maxItems", path)
|
|
389
|
+
if min_items is not None and len(instance) < min_items:
|
|
390
|
+
errors.append(_error(path, f"must have at least {min_items} items"))
|
|
391
|
+
if max_items is not None and len(instance) > max_items:
|
|
392
|
+
errors.append(_error(path, f"must have at most {max_items} items"))
|
|
393
|
+
if "items" in schema:
|
|
394
|
+
for index, item in enumerate(instance):
|
|
395
|
+
errors += _validate(item, schema["items"], _pointer(path, index))
|
|
396
|
+
return errors
|
|
397
|
+
|
|
398
|
+
|
|
399
|
+
def _check_object(
|
|
400
|
+
instance: dict[str, JSONValue], schema: dict[str, JSONValue], path: str
|
|
401
|
+
) -> list[str]:
|
|
402
|
+
properties = schema.get("properties", {})
|
|
403
|
+
required = schema.get("required", [])
|
|
404
|
+
if not isinstance(properties, dict):
|
|
405
|
+
raise SuiteError(f"schema at {path or '/'} has non-object properties")
|
|
406
|
+
if not isinstance(required, list) or not all(isinstance(name, str) for name in required):
|
|
407
|
+
raise SuiteError(f"schema at {path or '/'} has a required list that is not strings")
|
|
408
|
+
|
|
409
|
+
errors = [
|
|
410
|
+
_error(path, f"missing required property {name!r}")
|
|
411
|
+
for name in required
|
|
412
|
+
if name not in instance
|
|
413
|
+
]
|
|
414
|
+
for name, value in instance.items():
|
|
415
|
+
child = _pointer(path, name)
|
|
416
|
+
if name in properties:
|
|
417
|
+
errors += _validate(value, properties[name], child)
|
|
418
|
+
elif "additionalProperties" in schema:
|
|
419
|
+
if schema["additionalProperties"] is False:
|
|
420
|
+
errors.append(_error(path, f"unexpected property {name!r}"))
|
|
421
|
+
else:
|
|
422
|
+
errors += _validate(value, schema["additionalProperties"], child)
|
|
423
|
+
return errors
|
|
424
|
+
|
|
425
|
+
|
|
426
|
+
def _check_combinators(instance: JSONValue, schema: dict[str, JSONValue], path: str) -> list[str]:
|
|
427
|
+
errors = []
|
|
428
|
+
for branch in _branches(schema, "allOf", path):
|
|
429
|
+
errors += _validate(instance, branch, path)
|
|
430
|
+
if "anyOf" in schema:
|
|
431
|
+
branches = _branches(schema, "anyOf", path)
|
|
432
|
+
if not any(not _validate(instance, branch, path) for branch in branches):
|
|
433
|
+
errors.append(_error(path, "does not match any schema in anyOf"))
|
|
434
|
+
if "oneOf" in schema:
|
|
435
|
+
branches = _branches(schema, "oneOf", path)
|
|
436
|
+
matched = sum(not _validate(instance, branch, path) for branch in branches)
|
|
437
|
+
if matched != 1:
|
|
438
|
+
errors.append(_error(path, f"matches {matched} schemas in oneOf, expected exactly 1"))
|
|
439
|
+
return errors
|
|
440
|
+
|
|
441
|
+
|
|
442
|
+
def _branches(schema: dict[str, JSONValue], keyword: str, path: str) -> list[JSONValue]:
|
|
443
|
+
branches = schema.get(keyword, [])
|
|
444
|
+
if not isinstance(branches, list) or (keyword in schema and not branches):
|
|
445
|
+
raise SuiteError(f"schema at {path or '/'} needs a non-empty array for {keyword}")
|
|
446
|
+
return branches
|
|
447
|
+
|
|
448
|
+
|
|
449
|
+
def _count(schema: dict[str, JSONValue], keyword: str, path: str) -> int | None:
|
|
450
|
+
value = schema.get(keyword)
|
|
451
|
+
if value is None:
|
|
452
|
+
return None
|
|
453
|
+
if isinstance(value, bool) or not isinstance(value, int) or value < 0:
|
|
454
|
+
raise SuiteError(f"schema at {path or '/'} needs a non-negative integer for {keyword}")
|
|
455
|
+
count: int = value
|
|
456
|
+
return count
|
|
457
|
+
|
|
458
|
+
|
|
459
|
+
# Patterns --------------------------------------------------------------------------------
|
|
460
|
+
|
|
461
|
+
|
|
462
|
+
def _compile(pattern: str, where: str) -> re.Pattern[str]:
|
|
463
|
+
try:
|
|
464
|
+
return _compile_cached(pattern)
|
|
465
|
+
except re.error as exc:
|
|
466
|
+
raise SuiteError(f"{where} is not a valid regular expression: {exc}") from None
|
|
467
|
+
except _NestedQuantifierError as exc:
|
|
468
|
+
raise SuiteError(
|
|
469
|
+
f"{where} repeats {exc.group!r}, which itself contains a quantifier or an "
|
|
470
|
+
"alternation; such patterns can take exponential time to fail, so use a "
|
|
471
|
+
"character class or an unrepeated group instead"
|
|
472
|
+
) from None
|
|
473
|
+
|
|
474
|
+
|
|
475
|
+
@functools.lru_cache(maxsize=256)
|
|
476
|
+
def _compile_cached(pattern: str) -> re.Pattern[str]:
|
|
477
|
+
# Compiling first rejects invalid syntax, so the scanner below only sees valid patterns.
|
|
478
|
+
re.compile(pattern)
|
|
479
|
+
group = _nested_quantifier(pattern)
|
|
480
|
+
if group is not None:
|
|
481
|
+
raise _NestedQuantifierError(group)
|
|
482
|
+
ecma = "".join(
|
|
483
|
+
r"\Z" if kind == _END_ANCHOR else pattern[start:end]
|
|
484
|
+
for kind, start, end in _regex_tokens(pattern)
|
|
485
|
+
)
|
|
486
|
+
return re.compile(ecma, re.ASCII)
|
|
487
|
+
|
|
488
|
+
|
|
489
|
+
def _nested_quantifier(pattern: str) -> str | None:
|
|
490
|
+
"""Return the first group repeated more than once that contains a quantifier or `|`.
|
|
491
|
+
|
|
492
|
+
That shape lets the engine split one input many ways, which is what makes failing
|
|
493
|
+
matches take exponential time. Fixed counts such as `{3}` are not quantifiers here
|
|
494
|
+
because they leave only one way to split.
|
|
495
|
+
"""
|
|
496
|
+
enclosing: list[tuple[int, bool]] = []
|
|
497
|
+
ambiguous = False
|
|
498
|
+
closed: tuple[int, bool] | None = None
|
|
499
|
+
for kind, start, end in _regex_tokens(pattern):
|
|
500
|
+
just_closed, closed = closed, None
|
|
501
|
+
if kind == _OPEN:
|
|
502
|
+
enclosing.append((start, ambiguous))
|
|
503
|
+
ambiguous = False
|
|
504
|
+
elif kind == _CLOSE and enclosing:
|
|
505
|
+
group_start, outer = enclosing.pop()
|
|
506
|
+
closed = (group_start, ambiguous)
|
|
507
|
+
ambiguous = outer or ambiguous
|
|
508
|
+
elif kind == _ALTERNATION:
|
|
509
|
+
ambiguous = True
|
|
510
|
+
elif kind == _QUANTIFIER:
|
|
511
|
+
low, high = _repeat_bounds(pattern[start:end])
|
|
512
|
+
if just_closed is not None and just_closed[1] and (high is None or high > 1):
|
|
513
|
+
return pattern[just_closed[0] : end]
|
|
514
|
+
ambiguous = ambiguous or low != high
|
|
515
|
+
return None
|
|
516
|
+
|
|
517
|
+
|
|
518
|
+
def _repeat_bounds(quantifier: str) -> tuple[int, int | None]:
|
|
519
|
+
"""Return (minimum, maximum) repeats for a quantifier token; None means unbounded."""
|
|
520
|
+
symbol = quantifier[0]
|
|
521
|
+
if symbol == "*":
|
|
522
|
+
return 0, None
|
|
523
|
+
if symbol == "+":
|
|
524
|
+
return 1, None
|
|
525
|
+
if symbol == "?":
|
|
526
|
+
return 0, 1
|
|
527
|
+
low, comma, high = quantifier[1 : quantifier.index("}")].partition(",")
|
|
528
|
+
minimum = int(low or "0")
|
|
529
|
+
if not comma:
|
|
530
|
+
return minimum, minimum
|
|
531
|
+
return minimum, int(high) if high else None
|
|
532
|
+
|
|
533
|
+
|
|
534
|
+
def _regex_tokens(pattern: str) -> Iterator[tuple[str, int, int]]:
|
|
535
|
+
"""Split a valid Python regular expression into (kind, start, end) tokens."""
|
|
536
|
+
index = 0
|
|
537
|
+
while index < len(pattern):
|
|
538
|
+
kind, end = _next_token(pattern, index)
|
|
539
|
+
yield kind, index, end
|
|
540
|
+
index = end
|
|
541
|
+
|
|
542
|
+
|
|
543
|
+
def _next_token(pattern: str, index: int) -> tuple[str, int]:
|
|
544
|
+
char = pattern[index]
|
|
545
|
+
if char in _SINGLE_CHAR_TOKENS:
|
|
546
|
+
return _SINGLE_CHAR_TOKENS[char], index + 1
|
|
547
|
+
if char == "\\":
|
|
548
|
+
return _ATOM, index + 2
|
|
549
|
+
if char == "[":
|
|
550
|
+
return _ATOM, _class_end(pattern, index)
|
|
551
|
+
if char == "(":
|
|
552
|
+
atom = _GROUP_ATOM.match(pattern, index)
|
|
553
|
+
if atom:
|
|
554
|
+
return _ATOM, atom.end()
|
|
555
|
+
group = _GROUP_START.match(pattern, index)
|
|
556
|
+
return _OPEN, group.end() if group else index + 1
|
|
557
|
+
quantifier = _QUANTIFIER_SYNTAX.match(pattern, index)
|
|
558
|
+
return (_QUANTIFIER, quantifier.end()) if quantifier else (_ATOM, index + 1)
|
|
559
|
+
|
|
560
|
+
|
|
561
|
+
def _class_end(pattern: str, start: int) -> int:
|
|
562
|
+
index = start + 1
|
|
563
|
+
if pattern.startswith("^", index):
|
|
564
|
+
index += 1
|
|
565
|
+
# A "]" right after the opening bracket (or "[^") is a literal member.
|
|
566
|
+
if pattern.startswith("]", index):
|
|
567
|
+
index += 1
|
|
568
|
+
while index < len(pattern) and pattern[index] != "]":
|
|
569
|
+
index += 2 if pattern[index] == "\\" else 1
|
|
570
|
+
return index + 1
|
|
571
|
+
|
|
572
|
+
|
|
573
|
+
# Helpers ---------------------------------------------------------------------------------
|
|
574
|
+
|
|
575
|
+
|
|
576
|
+
def _json_equal(left: JSONValue, right: JSONValue) -> bool:
|
|
577
|
+
"""Equality under JSON rules: true is not 1, but 1 equals 1.0."""
|
|
578
|
+
if _is_number(left) and _is_number(right):
|
|
579
|
+
return bool(left == right)
|
|
580
|
+
if type(left) is not type(right):
|
|
581
|
+
return False
|
|
582
|
+
if isinstance(left, list):
|
|
583
|
+
return len(left) == len(right) and all(map(_json_equal, left, right))
|
|
584
|
+
if isinstance(left, dict):
|
|
585
|
+
return left.keys() == right.keys() and all(_json_equal(left[k], right[k]) for k in left)
|
|
586
|
+
return bool(left == right)
|
|
587
|
+
|
|
588
|
+
|
|
589
|
+
def _json_type(value: JSONValue) -> str:
|
|
590
|
+
if value is None:
|
|
591
|
+
return "null"
|
|
592
|
+
# bool is checked before int because bool is a subclass of int.
|
|
593
|
+
for python_type, name in _JSON_TYPE_NAMES:
|
|
594
|
+
if isinstance(value, python_type):
|
|
595
|
+
return name
|
|
596
|
+
return type(value).__name__
|
|
597
|
+
|
|
598
|
+
|
|
599
|
+
def _pointer(path: str, token: str | int) -> str:
|
|
600
|
+
escaped = str(token).replace("~", "~0").replace("/", "~1")
|
|
601
|
+
return f"{path}/{escaped}"
|
|
602
|
+
|
|
603
|
+
|
|
604
|
+
def _error(path: str, message: str) -> str:
|
|
605
|
+
return f"{path or '/'}: {message}"
|
|
606
|
+
|
|
607
|
+
|
|
608
|
+
def _preview(value: JSONValue, limit: int = 80) -> str:
|
|
609
|
+
text = json.dumps(value, ensure_ascii=False)
|
|
610
|
+
return text if len(text) <= limit else text[: limit - 3] + "..."
|