divejson 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
divejson/conform.py ADDED
@@ -0,0 +1,450 @@
1
+ """The conformance runner: what `divejson conform` walks, and what it decides.
2
+
3
+ Every implementation of DiveJSON provides this command, and it is what checks the
4
+ format's corpus. The specification repository carries no code of its own, so a pinned
5
+ release of an implementation is what runs against its fixtures; this repository runs the
6
+ same command over its vendored copy of them. One directory shape, walked the same way by
7
+ both:
8
+
9
+ valid/ documents that must validate
10
+ invalid/ documents that must not, or must not even parse
11
+ <format>/ reader pairs: an input, and the document reading it must produce
12
+ write/<format>/ writer pairs: a document, and the file writing it must produce
13
+
14
+ A pair directory is named for a **format id** — what an implementation registers an
15
+ adapter under — and that is the whole coupling between a corpus and an implementation. A
16
+ corpus carrying pairs for a format this implementation does not register is one this
17
+ implementation cannot answer for, and saying so is worth more than passing.
18
+
19
+ Three exit statuses, and the distinction between the last two is the point:
20
+
21
+ * 0 — every case passed.
22
+ * 1 — a case **failed**: a valid fixture that does not validate, an invalid one that
23
+ does, a pair whose produced document differs from the expected one.
24
+ * 2 — the **corpus's shape** is wrong: an empty directory, an input with no expected
25
+ document beside it, an expected document with no input, a pair directory for a format
26
+ this implementation does not register. Nothing was proved either way, which is not the
27
+ same answer as a failure — and a suite that reads "the runner found nothing to run" as
28
+ success is the failure this status exists for. A glob that silently matches nothing is
29
+ how a conformance corpus stops testing anything.
30
+ """
31
+
32
+ from __future__ import annotations
33
+
34
+ import difflib
35
+ import json
36
+ from collections.abc import Callable, Iterable
37
+ from dataclasses import dataclass
38
+ from enum import Enum, auto
39
+ from pathlib import Path
40
+ from typing import Any
41
+
42
+ from .uddf import UddfError, convert_uddf
43
+ from .validate import DuplicateMemberError, parse_document, validate_document
44
+
45
+ __all__ = [
46
+ "IGNORED",
47
+ "READERS",
48
+ "WRITTEN",
49
+ "Finding",
50
+ "Group",
51
+ "Result",
52
+ "compared",
53
+ "known_formats",
54
+ "run",
55
+ ]
56
+
57
+ # The two members a converted document asserts about its own *run* rather than about the
58
+ # input: when it was converted, and what converted it. A port of this converter would
59
+ # write a different `generator` and still be right, and a release moves
60
+ # `generator.version` without moving anything the mapping decided — so neither belongs in
61
+ # a comparison of what the mapping produced. The rule lives in the package rather than in
62
+ # a test helper because a port, an application checking its own determinism and this
63
+ # runner all have to drop exactly these two, and a second copy is a second thing to get
64
+ # out of step.
65
+ IGNORED = ("exported_at", "generator")
66
+
67
+ # How many lines of a mismatch are printed before the rest are counted instead. A pair
68
+ # that differs everywhere would otherwise bury the build log in a document.
69
+ DIFF_LINES = 60
70
+
71
+
72
+ def compared(document: dict[str, Any]) -> dict[str, Any]:
73
+ """A converted document reduced to what a fixture comparison is about."""
74
+ return {member: value for member, value in document.items() if member not in IGNORED}
75
+
76
+
77
+ def _read_uddf(data: bytes) -> dict[str, Any]:
78
+ # `exported_at` is left to default, which is the clock: it is one of the two members
79
+ # `compared` drops, so nothing this runner decides can see it.
80
+ return convert_uddf(data).document
81
+
82
+
83
+ # What this implementation reads, by the format id a corpus directory is named for.
84
+ READERS: dict[str, Callable[[bytes], dict[str, Any]]] = {"uddf": _read_uddf}
85
+
86
+ # What a reader raises for a source it cannot read. An input the reader refuses is a
87
+ # failing case rather than a crash, so that one broken pair does not hide the others.
88
+ _READ_ERRORS: tuple[type[Exception], ...] = (UddfError,)
89
+
90
+ # The formats this implementation can *write*. Empty: it reads other formats into
91
+ # DiveJSON and writes none of them back out, so a `write/<format>/` directory is one this
92
+ # implementation does not register — status 2 rather than a silent pass. What a writer
93
+ # pair compares, canonical XML with `<generator>` ignored, belongs beside the writer that
94
+ # produces one.
95
+ WRITTEN: frozenset[str] = frozenset()
96
+
97
+
98
+ def known_formats() -> frozenset[str]:
99
+ """Every format id this implementation registers, read or written."""
100
+ return frozenset(READERS) | WRITTEN
101
+
102
+
103
+ @dataclass(frozen=True, slots=True)
104
+ class Finding:
105
+ """One thing the runner has to say, with anything long carried in `detail`."""
106
+
107
+ where: str
108
+ message: str
109
+ detail: tuple[str, ...] = ()
110
+
111
+ def __str__(self) -> str:
112
+ return f"{self.where}: {self.message}"
113
+
114
+
115
+ @dataclass(frozen=True, slots=True)
116
+ class Group:
117
+ """One part of the corpus, and how it went: `valid`, `invalid`, or a format id."""
118
+
119
+ name: str
120
+ noun: str # singular: the caller pluralizes
121
+ checked: int
122
+ failed: int
123
+
124
+
125
+ @dataclass(frozen=True, slots=True)
126
+ class Result:
127
+ groups: tuple[Group, ...]
128
+ failures: tuple[Finding, ...]
129
+ shape: tuple[Finding, ...]
130
+ warnings: tuple[Finding, ...]
131
+
132
+ @property
133
+ def checked(self) -> int:
134
+ return sum(group.checked for group in self.groups)
135
+
136
+ @property
137
+ def status(self) -> int:
138
+ """The exit status, with a shape error outranking a failure.
139
+
140
+ Both are non-zero, so the order only matters to somebody reading the number: a
141
+ corpus whose shape is wrong has cases that never ran, and that is the more
142
+ useful thing to be told first.
143
+ """
144
+ if self.shape:
145
+ return 2
146
+ if self.failures:
147
+ return 1
148
+ return 0
149
+
150
+
151
+ class _Outcome(Enum):
152
+ PASSED = auto()
153
+ FAILED = auto()
154
+ UNCHECKED = auto() # the corpus's shape stopped the case from running at all
155
+
156
+
157
+ def run(
158
+ corpus: Path,
159
+ *,
160
+ strict: bool = False,
161
+ only: Iterable[str] = (),
162
+ skip: Iterable[str] = (),
163
+ ) -> Result:
164
+ """Walk a conformance corpus and report what it says about this implementation.
165
+
166
+ `only` and `skip` name **formats**, and neither reaches `valid/` or `invalid/`: those
167
+ are what the validator owes the format whichever adapters are registered. Under
168
+ `strict`, a format this implementation registers and the corpus has no pairs for
169
+ stops being a warning and becomes a corpus-shape error.
170
+ """
171
+ return _Walk(corpus, strict=strict, only=only, skip=skip).result()
172
+
173
+
174
+ class _Walk:
175
+ def __init__(
176
+ self, corpus: Path, *, strict: bool, only: Iterable[str], skip: Iterable[str]
177
+ ) -> None:
178
+ self._corpus = corpus
179
+ self._strict = strict
180
+ self._only = frozenset(only)
181
+ self._skip = frozenset(skip)
182
+ self._groups: list[Group] = []
183
+ self._failures: list[Finding] = []
184
+ self._shape: list[Finding] = []
185
+ self._warnings: list[Finding] = []
186
+
187
+ def result(self) -> Result:
188
+ self._walk()
189
+ return Result(
190
+ groups=tuple(self._groups),
191
+ failures=tuple(self._failures),
192
+ shape=tuple(self._shape),
193
+ warnings=tuple(self._warnings),
194
+ )
195
+
196
+ def _walk(self) -> None:
197
+ if not self._corpus.is_dir():
198
+ self._shape.append(Finding(str(self._corpus), "is not a directory"))
199
+ return
200
+
201
+ self._documents("valid", conforming=True)
202
+ self._documents("invalid", conforming=False)
203
+
204
+ for directory in sorted(
205
+ path
206
+ for path in self._corpus.iterdir()
207
+ if path.is_dir()
208
+ and not path.name.startswith(".")
209
+ and path.name not in ("valid", "invalid")
210
+ ):
211
+ if directory.name == "write":
212
+ self._writer_directories(directory)
213
+ else:
214
+ self._reader_directory(directory)
215
+
216
+ self._formats_with_no_pairs()
217
+
218
+ if not self._groups and not self._shape:
219
+ self._shape.append(
220
+ Finding(str(self._corpus), "holds no cases: no documents and no pairs")
221
+ )
222
+
223
+ # Documents: `valid/` and `invalid/`.
224
+
225
+ def _documents(self, name: str, *, conforming: bool) -> None:
226
+ directory = self._corpus / name
227
+ if not directory.is_dir():
228
+ return
229
+ paths = sorted(path for path in directory.glob("*.divejson") if path.is_file())
230
+ if not paths:
231
+ self._shape.append(Finding(name, "holds no documents"))
232
+ return
233
+ outcomes = [self._document(path, conforming=conforming) for path in paths]
234
+ self._record(name, "document", outcomes)
235
+
236
+ def _document(self, path: Path, *, conforming: bool) -> _Outcome:
237
+ where = self._where(path)
238
+ try:
239
+ document = parse_document(path.read_text(encoding="utf-8"))
240
+ except (UnicodeDecodeError, DuplicateMemberError, json.JSONDecodeError) as error:
241
+ if conforming:
242
+ self._failures.append(Finding(where, f"is not readable as JSON — {error}"))
243
+ return _Outcome.FAILED
244
+ # Refusing at the parse is a refusal: a duplicate member name is one of the
245
+ # things §9 has readers reject, and the document never reaches the schema.
246
+ return _Outcome.PASSED
247
+ except OSError as error:
248
+ self._failures.append(Finding(where, f"is unreadable — {error}"))
249
+ return _Outcome.FAILED
250
+
251
+ issues = validate_document(document)
252
+ if conforming and issues:
253
+ self._failures.append(
254
+ Finding(
255
+ where,
256
+ f"does not conform, in {len(issues)} way{'s' if len(issues) != 1 else ''}",
257
+ tuple(str(issue) for issue in issues),
258
+ )
259
+ )
260
+ return _Outcome.FAILED
261
+ if not conforming and not issues:
262
+ self._failures.append(Finding(where, "conforms, and every document here must not"))
263
+ return _Outcome.FAILED
264
+ return _Outcome.PASSED
265
+
266
+ # Reader pairs: `<format>/`.
267
+
268
+ def _reader_directory(self, directory: Path) -> None:
269
+ fmt = directory.name
270
+ if fmt not in READERS:
271
+ # Asked before `--only`/`--skip` are applied, deliberately. A corpus carrying
272
+ # pairs this implementation cannot run is a fact about the corpus, and no
273
+ # filter may turn it into silence — `--skip <that format>` is refused by the
274
+ # command for the same reason.
275
+ self._shape.append(
276
+ Finding(fmt, f"is a format this implementation does not read ({self._reads()})")
277
+ )
278
+ return
279
+ if not self._selected(fmt):
280
+ return
281
+
282
+ # Everything that is not a `.divejson` is an input to convert, which is what lets
283
+ # a format bring whatever extension it has. Hidden files are not: a corpus copied
284
+ # off a Mac carries `.DS_Store`, and that rule would make it a source file.
285
+ contents = sorted(
286
+ path
287
+ for path in directory.iterdir()
288
+ if path.is_file() and not path.name.startswith(".")
289
+ )
290
+ inputs = [path for path in contents if path.suffix != ".divejson"]
291
+ expectations = {path for path in contents if path.suffix == ".divejson"}
292
+ if not inputs:
293
+ self._shape.append(Finding(fmt, "holds no inputs to convert"))
294
+ return
295
+
296
+ outcomes = []
297
+ claimed = set()
298
+ for source in inputs:
299
+ expected = source.with_suffix(".divejson")
300
+ claimed.add(expected)
301
+ outcomes.append(self._reader_pair(fmt, source, expected))
302
+ for orphan in sorted(expectations - claimed):
303
+ self._shape.append(
304
+ Finding(self._where(orphan), "is an expected document with no input beside it")
305
+ )
306
+ self._record(fmt, "reader pair", outcomes)
307
+
308
+ def _reader_pair(self, fmt: str, source: Path, expected_path: Path) -> _Outcome:
309
+ where = self._where(source)
310
+ if not expected_path.is_file():
311
+ self._shape.append(
312
+ Finding(where, f"has no {expected_path.name} beside it to be checked against")
313
+ )
314
+ return _Outcome.UNCHECKED
315
+
316
+ try:
317
+ expected = parse_document(expected_path.read_text(encoding="utf-8"))
318
+ except (OSError, UnicodeDecodeError, DuplicateMemberError, json.JSONDecodeError) as error:
319
+ self._failures.append(
320
+ Finding(self._where(expected_path), f"is not readable as JSON — {error}")
321
+ )
322
+ return _Outcome.FAILED
323
+ if not isinstance(expected, dict):
324
+ self._failures.append(
325
+ Finding(self._where(expected_path), "is not a JSON object, so it is not a document")
326
+ )
327
+ return _Outcome.FAILED
328
+
329
+ try:
330
+ produced = READERS[fmt](source.read_bytes())
331
+ except _READ_ERRORS as error:
332
+ self._failures.append(Finding(where, f"could not be converted — {error}"))
333
+ return _Outcome.FAILED
334
+ except OSError as error:
335
+ self._failures.append(Finding(where, f"is unreadable — {error}"))
336
+ return _Outcome.FAILED
337
+
338
+ outcome = _Outcome.PASSED
339
+ if compared(produced) != compared(expected):
340
+ self._failures.append(
341
+ Finding(
342
+ where,
343
+ f"converts to a document that differs from {expected_path.name}",
344
+ _diff(expected, produced),
345
+ )
346
+ )
347
+ outcome = _Outcome.FAILED
348
+ issues = validate_document(expected)
349
+ if issues:
350
+ self._failures.append(
351
+ Finding(
352
+ self._where(expected_path),
353
+ f"is expected of a converter and does not itself conform, "
354
+ f"in {len(issues)} way{'s' if len(issues) != 1 else ''}",
355
+ tuple(str(issue) for issue in issues),
356
+ )
357
+ )
358
+ outcome = _Outcome.FAILED
359
+ return outcome
360
+
361
+ # Writer pairs: `write/<format>/`.
362
+
363
+ def _writer_directories(self, directory: Path) -> None:
364
+ """`write/` holds one directory of writer pairs per format an implementation writes.
365
+
366
+ With `WRITTEN` empty, every one of them is a directory this implementation cannot
367
+ answer for, and the walk stops at that. Comparing a writer pair is the writer's
368
+ own question — canonical XML with `<generator>` ignored, for a writer that emits
369
+ XML — and it arrives with the writer rather than being decided for one nobody has
370
+ seen yet.
371
+ """
372
+ children = sorted(
373
+ path
374
+ for path in directory.iterdir()
375
+ if path.is_dir() and not path.name.startswith(".")
376
+ )
377
+ if not children:
378
+ self._shape.append(Finding("write", "holds no writer-pair directories"))
379
+ return
380
+ for child in children:
381
+ if child.name not in WRITTEN:
382
+ self._shape.append(
383
+ Finding(
384
+ f"write/{child.name}",
385
+ f"is a format this implementation does not write ({self._writes()})",
386
+ )
387
+ )
388
+
389
+ # Formats with no directory at all.
390
+
391
+ def _formats_with_no_pairs(self) -> None:
392
+ for fmt in sorted(READERS):
393
+ if self._selected(fmt) and not (self._corpus / fmt).is_dir():
394
+ self._missing(
395
+ fmt,
396
+ "is a format this implementation reads, and the corpus has no pairs for it",
397
+ )
398
+ for fmt in sorted(WRITTEN):
399
+ if self._selected(fmt) and not (self._corpus / "write" / fmt).is_dir():
400
+ self._missing(
401
+ f"write/{fmt}",
402
+ "is a format this implementation writes, and the corpus has no pairs for it",
403
+ )
404
+
405
+ def _missing(self, where: str, message: str) -> None:
406
+ if self._strict:
407
+ self._shape.append(Finding(where, message))
408
+ else:
409
+ self._warnings.append(Finding(where, f"{message} (an error under --strict)"))
410
+
411
+ # Bookkeeping.
412
+
413
+ def _record(self, name: str, noun: str, outcomes: list[_Outcome]) -> None:
414
+ checked = sum(outcome is not _Outcome.UNCHECKED for outcome in outcomes)
415
+ if checked:
416
+ failed = sum(outcome is _Outcome.FAILED for outcome in outcomes)
417
+ self._groups.append(Group(name, noun, checked, failed))
418
+
419
+ def _selected(self, fmt: str) -> bool:
420
+ return (not self._only or fmt in self._only) and fmt not in self._skip
421
+
422
+ def _where(self, path: Path) -> str:
423
+ return path.relative_to(self._corpus).as_posix()
424
+
425
+ def _reads(self) -> str:
426
+ return f"it reads {', '.join(sorted(READERS))}"
427
+
428
+ def _writes(self) -> str:
429
+ return f"it writes {', '.join(sorted(WRITTEN))}" if WRITTEN else "it writes nothing"
430
+
431
+
432
+ def _diff(expected: dict[str, Any], produced: dict[str, Any]) -> tuple[str, ...]:
433
+ lines = list(
434
+ difflib.unified_diff(
435
+ _rendered(expected),
436
+ _rendered(produced),
437
+ fromfile="expected",
438
+ tofile="produced",
439
+ lineterm="",
440
+ n=2,
441
+ )
442
+ )
443
+ if len(lines) > DIFF_LINES:
444
+ return (*lines[:DIFF_LINES], f"... and {len(lines) - DIFF_LINES} more lines of difference")
445
+ return tuple(lines)
446
+
447
+
448
+ def _rendered(document: dict[str, Any]) -> list[str]:
449
+ """The lines a mismatch is diffed over: the document, minus what is not compared."""
450
+ return json.dumps(compared(document), indent=2, ensure_ascii=False).splitlines()
divejson/py.typed ADDED
File without changes