iirds-validate 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- iirds_validate/__init__.py +7 -0
- iirds_validate/__main__.py +5 -0
- iirds_validate/banner.py +45 -0
- iirds_validate/cli.py +255 -0
- iirds_validate/context.py +269 -0
- iirds_validate/data/ontologies/1.3/iirds-core.rdf +2122 -0
- iirds_validate/data/ontologies/1.3/iirds-handover.rdf +100 -0
- iirds_validate/data/ontologies/1.3/iirds-machinery.rdf +380 -0
- iirds_validate/data/ontologies/1.3/iirds-skos.rdf +1342 -0
- iirds_validate/data/ontologies/1.3/iirds-software.rdf +86 -0
- iirds_validate/data/ontologies/README.md +14 -0
- iirds_validate/data/ontologies/sha256sums.txt +5 -0
- iirds_validate/data/rule-catalog.json +3950 -0
- iirds_validate/model.py +270 -0
- iirds_validate/ontology.py +118 -0
- iirds_validate/package.py +191 -0
- iirds_validate/packer.py +117 -0
- iirds_validate/py.typed +0 -0
- iirds_validate/registry.py +101 -0
- iirds_validate/report.py +158 -0
- iirds_validate/resources.py +68 -0
- iirds_validate/rules/__init__.py +11 -0
- iirds_validate/rules/container.py +287 -0
- iirds_validate/rules/content.py +297 -0
- iirds_validate/rules/handover.py +237 -0
- iirds_validate/rules/lint.py +460 -0
- iirds_validate/rules/requirements.py +130 -0
- iirds_validate/rules/schema.py +709 -0
- iirds_validate/rules/schema_tables.py +168 -0
- iirds_validate/rules/system.py +197 -0
- iirds_validate/runner.py +172 -0
- iirds_validate/terms.py +152 -0
- iirds_validate-0.1.0.dist-info/METADATA +450 -0
- iirds_validate-0.1.0.dist-info/RECORD +40 -0
- iirds_validate-0.1.0.dist-info/WHEEL +5 -0
- iirds_validate-0.1.0.dist-info/entry_points.txt +3 -0
- iirds_validate-0.1.0.dist-info/licenses/LICENSE +202 -0
- iirds_validate-0.1.0.dist-info/licenses/NOTICE +76 -0
- iirds_validate-0.1.0.dist-info/licenses/THIRD_PARTY.md +62 -0
- iirds_validate-0.1.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
"""Offline, graph-based validator and interoperability linter for iiRDS packages."""
|
|
2
|
+
|
|
3
|
+
__version__ = "0.1.0"
|
|
4
|
+
|
|
5
|
+
from .model import Finding, Rule, Severity, Violation # noqa: F401
|
|
6
|
+
from .package import Package, PackageError # noqa: F401
|
|
7
|
+
from .runner import check, lint, load # noqa: F401
|
iirds_validate/banner.py
ADDED
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
"""The thing you see when you type `iirdsv` and nothing else.
|
|
2
|
+
|
|
3
|
+
Deliberately not printed by `check`, `lint` or `all`. Those write to a build
|
|
4
|
+
log or into a pipe, and `--format json` writes a document another program
|
|
5
|
+
parses; a banner in front of either is somewhere between noise and corruption.
|
|
6
|
+
|
|
7
|
+
Plain ASCII on purpose. Block-drawing characters look better in a modern
|
|
8
|
+
terminal and turn into rubbish in a Windows console or over a serial link, and
|
|
9
|
+
the machines this tool is built for are exactly the ones with the old fonts.
|
|
10
|
+
"""
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
from . import __version__
|
|
14
|
+
|
|
15
|
+
#: figlet, "slant". 25 columns, which leaves room on an 80-column terminal for
|
|
16
|
+
#: the version to sit on the baseline.
|
|
17
|
+
LOGO = r""" _ _ ____ ____ _____
|
|
18
|
+
(_|_) __ \/ __ \/ ___/
|
|
19
|
+
/ / / /_/ / / / /\__ \
|
|
20
|
+
/ / / _, _/ /_/ /___/ /
|
|
21
|
+
/_/_/_/ |_/_____//____/"""
|
|
22
|
+
|
|
23
|
+
TAGLINE = "conformance and interoperability checking for iiRDS packages, offline"
|
|
24
|
+
|
|
25
|
+
COMMANDS = (
|
|
26
|
+
("<path>", "check and lint it — a package, a directory, either"),
|
|
27
|
+
("check <path>", "does it conform to the specification?"),
|
|
28
|
+
("lint <path>", "will anyone else be able to read it?"),
|
|
29
|
+
("pack <directory>", "write it as a conformant .iirds, then check that"),
|
|
30
|
+
("rules", "every rule this tool checks, and its source"),
|
|
31
|
+
)
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def banner(coverage=None) -> str:
|
|
35
|
+
"""The logo, the version on the baseline, and how to start."""
|
|
36
|
+
lines = LOGO.splitlines()
|
|
37
|
+
lines[-1] = "%-26s validate %s" % (lines[-1], __version__)
|
|
38
|
+
out = ["", "\n".join(lines), "", " " + TAGLINE, ""]
|
|
39
|
+
for command, purpose in COMMANDS:
|
|
40
|
+
out.append(" iirdsv %-18s %s" % (command, purpose))
|
|
41
|
+
out.append("")
|
|
42
|
+
if coverage:
|
|
43
|
+
out.append(" " + coverage)
|
|
44
|
+
out.append("")
|
|
45
|
+
return "\n".join(out)
|
iirds_validate/cli.py
ADDED
|
@@ -0,0 +1,255 @@
|
|
|
1
|
+
"""Command line interface.
|
|
2
|
+
|
|
3
|
+
iirdsv check pkg.iirds conformance (container + graph rules)
|
|
4
|
+
iirdsv lint pkg.iirds interoperability (can a consumer use it?)
|
|
5
|
+
iirdsv all pkg.iirds both
|
|
6
|
+
iirdsv rules --kind lint what this tool knows how to check
|
|
7
|
+
|
|
8
|
+
Exit codes: 0 clean, 1 errors found, 2 could not run.
|
|
9
|
+
"""
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import argparse
|
|
13
|
+
import json
|
|
14
|
+
import os
|
|
15
|
+
import sys
|
|
16
|
+
|
|
17
|
+
from . import __version__, runner
|
|
18
|
+
from .banner import banner
|
|
19
|
+
from .model import VERSIONS, Severity
|
|
20
|
+
from .package import discover
|
|
21
|
+
from .packer import PackError, pack
|
|
22
|
+
from .registry import CATALOG, all_rules, coverage
|
|
23
|
+
from .report import render
|
|
24
|
+
|
|
25
|
+
EXIT_OK, EXIT_FINDINGS, EXIT_ERROR = 0, 1, 2
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def _add_target(parser: argparse.ArgumentParser) -> None:
|
|
29
|
+
parser.add_argument("package", nargs="+",
|
|
30
|
+
help="packages: a .iirds file, an unpacked container directory, "
|
|
31
|
+
"or a directory to search")
|
|
32
|
+
parser.add_argument("-f", "--format", choices=("text", "json"), default="text")
|
|
33
|
+
parser.add_argument("--iirds-version", dest="version", default=None, choices=VERSIONS,
|
|
34
|
+
metavar="{%s}" % ",".join(VERSIONS),
|
|
35
|
+
help="validate against this version instead of the declared one")
|
|
36
|
+
parser.add_argument("-v", "--verbose", action="store_true", help="include spec links")
|
|
37
|
+
parser.add_argument("-q", "--quiet", action="store_true", help="exit code only")
|
|
38
|
+
parser.add_argument("-W", "--warnings-as-errors", action="store_true",
|
|
39
|
+
help="fail the run on warnings too")
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def _targets(paths):
|
|
43
|
+
"""Expand what the user pointed at into packages.
|
|
44
|
+
|
|
45
|
+
A path can be a package, a directory that is one unpacked, or a directory
|
|
46
|
+
with packages somewhere underneath. Pointing at a build output directory
|
|
47
|
+
should do the obvious thing rather than require a shell glob.
|
|
48
|
+
"""
|
|
49
|
+
found, missing, empty = [], [], []
|
|
50
|
+
for path in paths:
|
|
51
|
+
if not os.path.exists(path):
|
|
52
|
+
missing.append(path)
|
|
53
|
+
continue
|
|
54
|
+
expanded = discover(path)
|
|
55
|
+
if expanded:
|
|
56
|
+
found.extend(expanded)
|
|
57
|
+
else:
|
|
58
|
+
empty.append(path)
|
|
59
|
+
return found, missing, empty
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def _run(args, kinds) -> int:
|
|
63
|
+
# A path that is not there is an operator error, not a validation result:
|
|
64
|
+
# exit 2. A file that opens but is not a valid container is a finding about
|
|
65
|
+
# the package, so it goes through the rules and exits 1.
|
|
66
|
+
targets, missing, empty = _targets(args.package)
|
|
67
|
+
for path in missing:
|
|
68
|
+
print("iirds-validate: no such file or directory: %s" % path, file=sys.stderr)
|
|
69
|
+
for path in empty:
|
|
70
|
+
print("iirds-validate: no iiRDS package found under %s" % path, file=sys.stderr)
|
|
71
|
+
if missing or empty:
|
|
72
|
+
return EXIT_ERROR
|
|
73
|
+
|
|
74
|
+
reports = [runner.run(path, kinds, version=args.version) for path in targets]
|
|
75
|
+
|
|
76
|
+
if args.format == "json":
|
|
77
|
+
payload = [r.as_dict() for r in reports]
|
|
78
|
+
json.dump(payload[0] if len(payload) == 1 else payload,
|
|
79
|
+
sys.stdout, ensure_ascii=False, indent=2)
|
|
80
|
+
sys.stdout.write("\n")
|
|
81
|
+
elif not args.quiet:
|
|
82
|
+
for i, report in enumerate(reports):
|
|
83
|
+
if i:
|
|
84
|
+
print()
|
|
85
|
+
render(report, "text", verbose=args.verbose)
|
|
86
|
+
|
|
87
|
+
failed = any(not r.ok for r in reports)
|
|
88
|
+
if args.warnings_as_errors:
|
|
89
|
+
failed = failed or any(r.count(Severity.WARNING) for r in reports)
|
|
90
|
+
|
|
91
|
+
if len(reports) > 1 and args.format == "text" and not args.quiet:
|
|
92
|
+
bad = sum(1 for r in reports if not r.ok)
|
|
93
|
+
print("\n%d packages: %d passed, %d failed"
|
|
94
|
+
% (len(reports), len(reports) - bad, bad))
|
|
95
|
+
return EXIT_FINDINGS if failed else EXIT_OK
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def _cmd_pack(args) -> int:
|
|
99
|
+
"""Write the archive, then validate the archive.
|
|
100
|
+
|
|
101
|
+
Validating what was just written rather than the directory is the point:
|
|
102
|
+
the five requirements about the ZIP that a directory cannot answer are now
|
|
103
|
+
answerable, and answered against the file that will actually be delivered.
|
|
104
|
+
"""
|
|
105
|
+
try:
|
|
106
|
+
output = pack(args.directory, args.output, overwrite=args.overwrite)
|
|
107
|
+
except PackError as exc:
|
|
108
|
+
print("iirds-validate: %s" % exc, file=sys.stderr)
|
|
109
|
+
return EXIT_ERROR
|
|
110
|
+
|
|
111
|
+
if not args.quiet and args.format == "text":
|
|
112
|
+
print("wrote %s (%.0f KB)\n" % (output, output.stat().st_size / 1024))
|
|
113
|
+
|
|
114
|
+
args.package = [str(output)]
|
|
115
|
+
return _run(args, runner.ALL_KINDS)
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def _cmd_rules(args) -> int:
|
|
119
|
+
rules = all_rules()
|
|
120
|
+
if args.kind:
|
|
121
|
+
rules = [r for r in rules if r.kind == args.kind]
|
|
122
|
+
if args.ids:
|
|
123
|
+
wanted = {rid.upper() for rid in args.ids}
|
|
124
|
+
rules = [r for r in rules if r.id.upper() in wanted]
|
|
125
|
+
missing = wanted - {r.id.upper() for r in rules}
|
|
126
|
+
if missing:
|
|
127
|
+
print("no such rule: %s" % ", ".join(sorted(missing)), file=sys.stderr)
|
|
128
|
+
return EXIT_ERROR
|
|
129
|
+
args.verbose = True # asking for one rule means asking about it
|
|
130
|
+
|
|
131
|
+
if args.format == "json":
|
|
132
|
+
json.dump([{"id": r.id, "kind": r.kind, "priority": r.prio, "severity": str(r.severity),
|
|
133
|
+
"versions": list(r.versions), "variants": list(r.variants),
|
|
134
|
+
"title": r.title, "spec": r.spec,
|
|
135
|
+
"source": "catalogue" if r.id in CATALOG else "iirds-validate",
|
|
136
|
+
"covers": list(r.covers), "fix": r.fix} for r in rules],
|
|
137
|
+
sys.stdout, ensure_ascii=False, indent=2)
|
|
138
|
+
sys.stdout.write("\n")
|
|
139
|
+
return EXIT_OK
|
|
140
|
+
|
|
141
|
+
for r in rules:
|
|
142
|
+
variants = ("/" + ",".join(r.variants)) if r.variants else ""
|
|
143
|
+
print("%-9s %-9s %-9s %s" % (r.id, r.kind, r.prio + variants, r.title[:96]))
|
|
144
|
+
if args.verbose:
|
|
145
|
+
source = "catalogue" if r.id in CATALOG else "this project"
|
|
146
|
+
print(" versions: %s source: %s"
|
|
147
|
+
% (", ".join(r.versions) if r.versions else "all", source))
|
|
148
|
+
if r.spec:
|
|
149
|
+
print(" spec: %s" % r.spec)
|
|
150
|
+
if r.covers:
|
|
151
|
+
print(" covers: %s" % ", ".join(r.covers))
|
|
152
|
+
if r.fix:
|
|
153
|
+
print(" fix: %s" % r.fix)
|
|
154
|
+
print()
|
|
155
|
+
|
|
156
|
+
if args.ids:
|
|
157
|
+
return EXIT_OK # asked about specific rules; no summary tail
|
|
158
|
+
|
|
159
|
+
print()
|
|
160
|
+
cov = coverage()
|
|
161
|
+
labels = {"container": "the ZIP and its layout",
|
|
162
|
+
"schema": "the metadata graph",
|
|
163
|
+
"system": "the run itself",
|
|
164
|
+
"content": "iiRDS XHTML5 (Appendix B)",
|
|
165
|
+
"lint": "will a consumer be able to use it"}
|
|
166
|
+
for kind in ("container", "schema", "system", "content", "lint"):
|
|
167
|
+
c = cov.get(kind)
|
|
168
|
+
if not c or not (c["total"] or c["ours"]):
|
|
169
|
+
continue
|
|
170
|
+
catalogued = "%d/%d" % (c["implemented"], c["total"]) if c["total"] else "-"
|
|
171
|
+
ours = (" +%d of its own" % c["ours"]) if c["ours"] else ""
|
|
172
|
+
print("%-10s %-8s %s%s" % (kind, catalogued, labels[kind], ours))
|
|
173
|
+
print()
|
|
174
|
+
print("%d of %d catalogued rules, plus %d of this project's own" % (
|
|
175
|
+
sum(c["implemented"] for c in cov.values()), len(CATALOG),
|
|
176
|
+
sum(c["ours"] for c in cov.values())))
|
|
177
|
+
return EXIT_OK
|
|
178
|
+
|
|
179
|
+
|
|
180
|
+
def main(argv=None) -> int:
|
|
181
|
+
parser = argparse.ArgumentParser(
|
|
182
|
+
prog="iirds-validate",
|
|
183
|
+
description="Offline validator and interoperability linter for iiRDS packages.")
|
|
184
|
+
parser.add_argument("--version", action="version", version="iirds-validate %s" % __version__)
|
|
185
|
+
sub = parser.add_subparsers(dest="command")
|
|
186
|
+
|
|
187
|
+
p_check = sub.add_parser("check", help="conformance: container structure and metadata graph")
|
|
188
|
+
_add_target(p_check)
|
|
189
|
+
|
|
190
|
+
p_lint = sub.add_parser("lint", help="interoperability: can a consumer actually use this?")
|
|
191
|
+
_add_target(p_lint)
|
|
192
|
+
|
|
193
|
+
p_all = sub.add_parser("all", help="check and lint together")
|
|
194
|
+
_add_target(p_all)
|
|
195
|
+
|
|
196
|
+
p_pack = sub.add_parser(
|
|
197
|
+
"pack", help="write a directory as a conformant .iirds and check it")
|
|
198
|
+
p_pack.add_argument("directory")
|
|
199
|
+
p_pack.add_argument("-o", "--output", default=None,
|
|
200
|
+
help="where to write it (default: alongside the directory)")
|
|
201
|
+
p_pack.add_argument("--overwrite", action="store_true")
|
|
202
|
+
p_pack.add_argument("-f", "--format", choices=("text", "json"), default="text")
|
|
203
|
+
p_pack.add_argument("--iirds-version", dest="version", default=None, choices=VERSIONS)
|
|
204
|
+
p_pack.add_argument("-v", "--verbose", action="store_true")
|
|
205
|
+
p_pack.add_argument("-q", "--quiet", action="store_true")
|
|
206
|
+
p_pack.add_argument("-W", "--warnings-as-errors", action="store_true")
|
|
207
|
+
|
|
208
|
+
p_rules = sub.add_parser("rules", help="list the rules this tool implements")
|
|
209
|
+
p_rules.add_argument("ids", nargs="*", metavar="RULE",
|
|
210
|
+
help="show only these rules, in full (e.g. M11 B8 R3)")
|
|
211
|
+
p_rules.add_argument("--kind",
|
|
212
|
+
choices=("container", "schema", "system", "content", "lint"))
|
|
213
|
+
p_rules.add_argument("-v", "--verbose", action="store_true",
|
|
214
|
+
help="also print versions, spec link, source and remedy")
|
|
215
|
+
p_rules.add_argument("-f", "--format", choices=("text", "json"), default="text")
|
|
216
|
+
|
|
217
|
+
# `iirdsv some/path` with no subcommand means `all`. Typing the verb is
|
|
218
|
+
# friction, and "check it" is what anybody pointing at a package wants.
|
|
219
|
+
argv = list(sys.argv[1:] if argv is None else argv)
|
|
220
|
+
known = {"check", "lint", "all", "pack", "rules", "-h", "--help", "--version"}
|
|
221
|
+
if argv and argv[0] not in known and not argv[0].startswith("-"):
|
|
222
|
+
argv.insert(0, "all")
|
|
223
|
+
|
|
224
|
+
args = parser.parse_args(argv)
|
|
225
|
+
|
|
226
|
+
if args.command is None:
|
|
227
|
+
# Bare `iirdsv`. argparse would exit 2 with a usage error, which is a
|
|
228
|
+
# poor answer to someone who has just installed the thing.
|
|
229
|
+
cov = coverage()
|
|
230
|
+
print(banner("%d of %d catalogued rules, plus %d of its own. no network access."
|
|
231
|
+
% (sum(v["implemented"] for v in cov.values()), len(CATALOG),
|
|
232
|
+
sum(v["ours"] for v in cov.values()))))
|
|
233
|
+
return EXIT_OK
|
|
234
|
+
|
|
235
|
+
try:
|
|
236
|
+
if args.command == "check":
|
|
237
|
+
return _run(args, runner.CONFORMANCE_KINDS)
|
|
238
|
+
if args.command == "lint":
|
|
239
|
+
return _run(args, runner.LINT_KINDS)
|
|
240
|
+
if args.command == "all":
|
|
241
|
+
return _run(args, runner.ALL_KINDS)
|
|
242
|
+
if args.command == "pack":
|
|
243
|
+
return _cmd_pack(args)
|
|
244
|
+
if args.command == "rules":
|
|
245
|
+
return _cmd_rules(args)
|
|
246
|
+
except KeyboardInterrupt:
|
|
247
|
+
return EXIT_ERROR
|
|
248
|
+
except OSError as exc:
|
|
249
|
+
print("iirds-validate: %s" % exc, file=sys.stderr)
|
|
250
|
+
return EXIT_ERROR
|
|
251
|
+
return EXIT_ERROR
|
|
252
|
+
|
|
253
|
+
|
|
254
|
+
if __name__ == "__main__":
|
|
255
|
+
sys.exit(main())
|
|
@@ -0,0 +1,269 @@
|
|
|
1
|
+
"""Everything a rule is handed: the container, the graph, and the ontology.
|
|
2
|
+
|
|
3
|
+
The whole point of this project lives in `build_graph`. iiRDS metadata is RDF,
|
|
4
|
+
and RDF says nothing about how it is written down. These are the same fact:
|
|
5
|
+
|
|
6
|
+
<iirds:Document rdf:about="urn:d1"/>
|
|
7
|
+
|
|
8
|
+
<rdf:Description rdf:about="urn:d1">
|
|
9
|
+
<rdf:type rdf:resource="http://iirds.tekom.de/iirds#Document"/>
|
|
10
|
+
</rdf:Description>
|
|
11
|
+
|
|
12
|
+
and so are these:
|
|
13
|
+
|
|
14
|
+
<iirds:relates-to-event><iirds:Event rdf:about="urn:e1"/></iirds:relates-to-event>
|
|
15
|
+
<iirds:relates-to-event rdf:resource="urn:e1"/>
|
|
16
|
+
|
|
17
|
+
A validator that walks the XML tree sees one form and misses the other, which is
|
|
18
|
+
how a package can be perfectly conformant and still unreadable to the tool that
|
|
19
|
+
is supposed to bless it. Parsing into a graph makes the distinction disappear
|
|
20
|
+
before any rule runs — and lets the same rules apply to metadata.jsonld.
|
|
21
|
+
"""
|
|
22
|
+
from __future__ import annotations
|
|
23
|
+
|
|
24
|
+
import hashlib
|
|
25
|
+
import json
|
|
26
|
+
import re
|
|
27
|
+
from dataclasses import dataclass, field
|
|
28
|
+
from typing import List, Optional, Set
|
|
29
|
+
|
|
30
|
+
from rdflib import BNode, Graph, URIRef
|
|
31
|
+
from rdflib.namespace import RDF, RDFS
|
|
32
|
+
|
|
33
|
+
from . import ontology as ontology_mod
|
|
34
|
+
from . import terms as T
|
|
35
|
+
from .model import LATEST_VERSION, METADATA_JSONLD, METADATA_RDF, PACKAGE_BASE, VERSIONS
|
|
36
|
+
from .package import Package
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
@dataclass
|
|
40
|
+
class Context:
|
|
41
|
+
package: Package
|
|
42
|
+
graph: Graph
|
|
43
|
+
ontology: ontology_mod.Ontology
|
|
44
|
+
version: str
|
|
45
|
+
variant: str
|
|
46
|
+
declared_version: Optional[str] = None # None when the package omits it
|
|
47
|
+
requested_version: Optional[str] = None # set when the caller overrode it
|
|
48
|
+
parse_errors: List[str] = field(default_factory=list)
|
|
49
|
+
sources: List[str] = field(default_factory=list)
|
|
50
|
+
#: One graph per metadata file, kept alongside the merged `graph` so a rule
|
|
51
|
+
#: can ask whether the serialisations agree. Merging is right for every
|
|
52
|
+
#: other rule and is exactly what hides a disagreement.
|
|
53
|
+
per_source: dict = field(default_factory=dict)
|
|
54
|
+
|
|
55
|
+
# -- graph helpers ------------------------------------------------------
|
|
56
|
+
def instances_of(self, cls: URIRef, include_subclasses: bool = True) -> List:
|
|
57
|
+
"""Subjects typed `cls`, or any class beneath it."""
|
|
58
|
+
classes = self.ontology.subclasses_of(cls) if include_subclasses else {cls}
|
|
59
|
+
out, seen = [], set()
|
|
60
|
+
for c in classes:
|
|
61
|
+
for s in self.graph.subjects(RDF.type, c):
|
|
62
|
+
if s not in seen:
|
|
63
|
+
seen.add(s)
|
|
64
|
+
out.append(s)
|
|
65
|
+
return out
|
|
66
|
+
|
|
67
|
+
def typed_exactly(self, cls: URIRef) -> List:
|
|
68
|
+
"""Subjects carrying `cls` itself as an rdf:type (no subclasses)."""
|
|
69
|
+
return list(self.graph.subjects(RDF.type, cls))
|
|
70
|
+
|
|
71
|
+
def values(self, subject, prop: URIRef) -> List:
|
|
72
|
+
return list(self.graph.objects(subject, prop))
|
|
73
|
+
|
|
74
|
+
def one(self, subject, prop: URIRef):
|
|
75
|
+
for o in self.graph.objects(subject, prop):
|
|
76
|
+
return o
|
|
77
|
+
return None
|
|
78
|
+
|
|
79
|
+
def has(self, subject, prop: URIRef) -> bool:
|
|
80
|
+
return (subject, prop, None) in self.graph
|
|
81
|
+
|
|
82
|
+
def information_units(self) -> List:
|
|
83
|
+
return self.instances_of(T.InformationUnit)
|
|
84
|
+
|
|
85
|
+
def iirds_subjects(self) -> Set:
|
|
86
|
+
"""Every subject that carries at least one iiRDS type."""
|
|
87
|
+
out = set()
|
|
88
|
+
for s, o in self.graph.subject_objects(RDF.type):
|
|
89
|
+
if self.ontology.is_iirds_term(o):
|
|
90
|
+
out.add(s)
|
|
91
|
+
return out
|
|
92
|
+
|
|
93
|
+
def ref(self, node) -> str:
|
|
94
|
+
"""A name for a node that is the same on every run.
|
|
95
|
+
|
|
96
|
+
rdflib mints a fresh identifier for every blank node on every parse, so
|
|
97
|
+
a finding that reported `str(node)` gave `N8892b8d9…` one run and
|
|
98
|
+
`N39e7e968…` the next. Two effects, both bad: a JSON report could not be
|
|
99
|
+
diffed between runs, and the same package written as RDF/XML and as
|
|
100
|
+
JSON-LD produced different findings — which is the one property this
|
|
101
|
+
project claims above all others.
|
|
102
|
+
|
|
103
|
+
A blank node is named by how you reach it instead: the nearest named
|
|
104
|
+
subject and the property that points at it, which is also far more use
|
|
105
|
+
to somebody reading the report than an opaque identifier. Where that
|
|
106
|
+
fails, a hash of the statements about the node, which is stable because
|
|
107
|
+
the statements are.
|
|
108
|
+
"""
|
|
109
|
+
if not isinstance(node, BNode):
|
|
110
|
+
return str(node)
|
|
111
|
+
for subject, predicate in sorted(self.graph.subject_predicates(node), key=str):
|
|
112
|
+
if not isinstance(subject, BNode):
|
|
113
|
+
return "%s %s" % (subject, str(predicate).split("#")[-1].split("/")[-1])
|
|
114
|
+
digest = hashlib.sha256()
|
|
115
|
+
for predicate, obj in sorted(self.graph.predicate_objects(node), key=str):
|
|
116
|
+
digest.update(("%s %s\n" % (predicate, obj)).encode("utf-8"))
|
|
117
|
+
return "_:%s" % digest.hexdigest()[:12]
|
|
118
|
+
|
|
119
|
+
def label_of(self, node) -> str:
|
|
120
|
+
for p in (RDFS.label, T.title):
|
|
121
|
+
v = self.one(node, p)
|
|
122
|
+
if v is not None:
|
|
123
|
+
return str(v)
|
|
124
|
+
return str(node)
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
def _detect(graph: Graph):
|
|
128
|
+
"""Read iirds:iiRDSVersion / iirds:formatRestriction off the package node."""
|
|
129
|
+
declared, variant = None, None
|
|
130
|
+
for pkg in graph.subjects(RDF.type, T.Package):
|
|
131
|
+
v = graph.value(pkg, T.iiRDSVersion)
|
|
132
|
+
if v is not None and declared is None:
|
|
133
|
+
declared = str(v).strip()
|
|
134
|
+
r = graph.value(pkg, T.formatRestriction)
|
|
135
|
+
if r is not None and variant is None:
|
|
136
|
+
variant = str(r).strip()
|
|
137
|
+
return declared, (variant or "unrestricted")
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
#: An .iirds package arrives from a supplier, so its metadata is untrusted
|
|
141
|
+
#: input. Two cheap guards, applied before the parser sees anything.
|
|
142
|
+
MAX_METADATA_BYTES = 64 * 1024 * 1024
|
|
143
|
+
_ENTITY_DECL = re.compile(rb"<!ENTITY", re.IGNORECASE)
|
|
144
|
+
_HAS_SCHEME = re.compile(r"^[A-Za-z][A-Za-z0-9+.-]*:")
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
#: rdflib decodes a bytes payload as UTF-8 unconditionally, so a document that
|
|
148
|
+
#: declares — and marks with a byte order mark — any other encoding fails to
|
|
149
|
+
#: parse at all. XML says the BOM decides, so it is honoured here and the
|
|
150
|
+
#: payload handed on as UTF-8.
|
|
151
|
+
_BOMS = ((b"\xff\xfe\x00\x00", "utf-32-le"), (b"\x00\x00\xfe\xff", "utf-32-be"),
|
|
152
|
+
(b"\xff\xfe", "utf-16-le"), (b"\xfe\xff", "utf-16-be"),
|
|
153
|
+
(b"\xef\xbb\xbf", "utf-8-sig"))
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
def _decode_by_bom(raw: bytes) -> bytes:
|
|
157
|
+
for bom, encoding in _BOMS:
|
|
158
|
+
if raw.startswith(bom):
|
|
159
|
+
text = raw.decode(encoding)
|
|
160
|
+
# The declaration would now contradict the bytes.
|
|
161
|
+
text = re.sub(r'(<\?xml[^>]*?)\s+encoding\s*=\s*(["\'])[^"\']*\2',
|
|
162
|
+
r"\1", text, count=1)
|
|
163
|
+
return text.encode("utf-8")
|
|
164
|
+
return raw
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
def _remote_contexts(node, found=None):
|
|
168
|
+
"""Every `@context` in the document that names a location to go and fetch.
|
|
169
|
+
|
|
170
|
+
JSON-LD lets a context be a URL, and the parser will dereference it. In a
|
|
171
|
+
package that arrived from a supplier that is two separate problems: it
|
|
172
|
+
breaks the promise that validation touches no network, and it lets the
|
|
173
|
+
sender choose a host for a machine inside the plant to connect to.
|
|
174
|
+
|
|
175
|
+
Contexts nest, and a context can be an array mixing inline objects with
|
|
176
|
+
URLs, so the whole document is walked rather than just the top level.
|
|
177
|
+
"""
|
|
178
|
+
found = [] if found is None else found
|
|
179
|
+
if isinstance(node, dict):
|
|
180
|
+
for key, value in node.items():
|
|
181
|
+
if key == "@context":
|
|
182
|
+
for candidate in (value if isinstance(value, list) else [value]):
|
|
183
|
+
if isinstance(candidate, str) and _HAS_SCHEME.match(candidate):
|
|
184
|
+
found.append(candidate)
|
|
185
|
+
else:
|
|
186
|
+
_remote_contexts(candidate, found)
|
|
187
|
+
else:
|
|
188
|
+
_remote_contexts(value, found)
|
|
189
|
+
elif isinstance(node, list):
|
|
190
|
+
for item in node:
|
|
191
|
+
_remote_contexts(item, found)
|
|
192
|
+
return found
|
|
193
|
+
|
|
194
|
+
|
|
195
|
+
def build_graph(package: Package):
|
|
196
|
+
"""Parse every metadata serialisation in the container into one graph."""
|
|
197
|
+
graph = Graph()
|
|
198
|
+
errors: List[str] = []
|
|
199
|
+
sources: List[str] = []
|
|
200
|
+
per_source = {}
|
|
201
|
+
|
|
202
|
+
for name, fmt in ((METADATA_RDF, "xml"), (METADATA_JSONLD, "json-ld")):
|
|
203
|
+
if not package.has(name):
|
|
204
|
+
continue
|
|
205
|
+
|
|
206
|
+
info = package.info(name)
|
|
207
|
+
if info is not None and info.file_size > MAX_METADATA_BYTES:
|
|
208
|
+
errors.append("%s: refused: %d bytes uncompressed, above the %d byte limit"
|
|
209
|
+
% (name, info.file_size, MAX_METADATA_BYTES))
|
|
210
|
+
continue
|
|
211
|
+
|
|
212
|
+
raw = _decode_by_bom(package.read(name))
|
|
213
|
+
|
|
214
|
+
# Nested internal entities expand geometrically: a few hundred bytes of
|
|
215
|
+
# declarations can occupy the parser indefinitely. iiRDS metadata has no
|
|
216
|
+
# legitimate use for them, so refuse rather than try to bound the damage.
|
|
217
|
+
if fmt == "xml" and _ENTITY_DECL.search(raw):
|
|
218
|
+
errors.append("%s: refused: the document declares XML entities" % name)
|
|
219
|
+
continue
|
|
220
|
+
|
|
221
|
+
if fmt == "json-ld":
|
|
222
|
+
try:
|
|
223
|
+
document = json.loads(raw.decode("utf-8"))
|
|
224
|
+
except Exception as exc:
|
|
225
|
+
errors.append("%s: %s: %s" % (name, type(exc).__name__, exc))
|
|
226
|
+
continue
|
|
227
|
+
remote = _remote_contexts(document)
|
|
228
|
+
if remote:
|
|
229
|
+
errors.append("%s: refused: @context must be inline, not fetched from %s"
|
|
230
|
+
% (name, ", ".join(sorted(set(remote))[:3])))
|
|
231
|
+
continue
|
|
232
|
+
|
|
233
|
+
try:
|
|
234
|
+
single = Graph()
|
|
235
|
+
single.parse(data=raw, format=fmt, publicID=PACKAGE_BASE)
|
|
236
|
+
except Exception as exc:
|
|
237
|
+
errors.append("%s: %s: %s" % (name, type(exc).__name__, exc))
|
|
238
|
+
continue
|
|
239
|
+
per_source[name] = single
|
|
240
|
+
graph += single
|
|
241
|
+
sources.append(name)
|
|
242
|
+
|
|
243
|
+
return graph, errors, sources, per_source
|
|
244
|
+
|
|
245
|
+
|
|
246
|
+
def load_context(package: Package, version: Optional[str] = None) -> Context:
|
|
247
|
+
graph, errors, sources, per_source = build_graph(package)
|
|
248
|
+
declared, variant = _detect(graph)
|
|
249
|
+
|
|
250
|
+
# plusmeta's tool filters its rules by the declared version string, so a
|
|
251
|
+
# package that omits iirds:iiRDSVersion runs zero schema rules and reports
|
|
252
|
+
# "no violations". Here a missing or unknown version falls back to the
|
|
253
|
+
# newest one and is recorded as a note, so nothing passes by silence.
|
|
254
|
+
effective = version or declared or LATEST_VERSION
|
|
255
|
+
if effective not in VERSIONS:
|
|
256
|
+
effective = LATEST_VERSION
|
|
257
|
+
|
|
258
|
+
return Context(
|
|
259
|
+
package=package,
|
|
260
|
+
graph=graph,
|
|
261
|
+
ontology=ontology_mod.load(effective if effective in VERSIONS else LATEST_VERSION),
|
|
262
|
+
version=effective,
|
|
263
|
+
variant=variant,
|
|
264
|
+
declared_version=declared,
|
|
265
|
+
requested_version=version,
|
|
266
|
+
parse_errors=errors,
|
|
267
|
+
sources=sources,
|
|
268
|
+
per_source=per_source,
|
|
269
|
+
)
|