entitylinkage 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- entitylinkage/__init__.py +5 -0
- entitylinkage/audit.py +26 -0
- entitylinkage/cli.py +90 -0
- entitylinkage/config.py +412 -0
- entitylinkage/errors.py +6 -0
- entitylinkage/matcher.py +249 -0
- entitylinkage/model.py +143 -0
- entitylinkage/normalize.py +20 -0
- entitylinkage/output.py +144 -0
- entitylinkage/overrides.py +50 -0
- entitylinkage-0.1.0.dist-info/METADATA +114 -0
- entitylinkage-0.1.0.dist-info/RECORD +15 -0
- entitylinkage-0.1.0.dist-info/WHEEL +4 -0
- entitylinkage-0.1.0.dist-info/entry_points.txt +2 -0
- entitylinkage-0.1.0.dist-info/licenses/LICENSE +201 -0
entitylinkage/audit.py
ADDED
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
from collections import Counter
|
|
2
|
+
from collections.abc import Sequence
|
|
3
|
+
|
|
4
|
+
from entitylinkage.model import AuditSummary, LinkageResult, LinkageStatus
|
|
5
|
+
|
|
6
|
+
_STATUSES: tuple[LinkageStatus, ...] = (
|
|
7
|
+
"ambiguous",
|
|
8
|
+
"not_applicable",
|
|
9
|
+
"resolved",
|
|
10
|
+
"unresolved",
|
|
11
|
+
)
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def audit_results(results: Sequence[LinkageResult]) -> AuditSummary:
|
|
15
|
+
"""Count statuses and the records carrying each stable reason code."""
|
|
16
|
+
status_counts = dict.fromkeys(_STATUSES, 0)
|
|
17
|
+
reason_code_counts: Counter[str] = Counter()
|
|
18
|
+
for result in results:
|
|
19
|
+
if result.status not in status_counts:
|
|
20
|
+
raise ValueError(f"unsupported linkage status: {result.status!r}")
|
|
21
|
+
status_counts[result.status] += 1
|
|
22
|
+
reason_code_counts.update(set(result.reason_codes))
|
|
23
|
+
return AuditSummary(
|
|
24
|
+
status_counts=status_counts,
|
|
25
|
+
reason_code_counts=dict(reason_code_counts),
|
|
26
|
+
)
|
entitylinkage/cli.py
ADDED
|
@@ -0,0 +1,90 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import argparse
|
|
4
|
+
import json
|
|
5
|
+
import sys
|
|
6
|
+
from collections.abc import Sequence
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
|
|
9
|
+
from entitylinkage.config import load_config
|
|
10
|
+
from entitylinkage.errors import EntityLinkageError
|
|
11
|
+
from entitylinkage.matcher import Linker
|
|
12
|
+
from entitylinkage.output import (
|
|
13
|
+
inspection_payload,
|
|
14
|
+
serialize_summary_json,
|
|
15
|
+
write_artifacts,
|
|
16
|
+
)
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def main(argv: Sequence[str] | None = None) -> int:
|
|
20
|
+
parser = _argument_parser()
|
|
21
|
+
try:
|
|
22
|
+
arguments = parser.parse_args(argv)
|
|
23
|
+
except SystemExit as error:
|
|
24
|
+
return int(error.code or 0)
|
|
25
|
+
|
|
26
|
+
try:
|
|
27
|
+
return _dispatch(arguments)
|
|
28
|
+
except (EntityLinkageError, OSError, ValueError) as error:
|
|
29
|
+
print(f"entitylinkage: {error}", file=sys.stderr)
|
|
30
|
+
return 2
|
|
31
|
+
except Exception as error:
|
|
32
|
+
print(f"entitylinkage: execution failed: {error}", file=sys.stderr)
|
|
33
|
+
return 2
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def _argument_parser() -> argparse.ArgumentParser:
|
|
37
|
+
parser = argparse.ArgumentParser(prog="entitylinkage")
|
|
38
|
+
commands = parser.add_subparsers(dest="command", required=True)
|
|
39
|
+
|
|
40
|
+
validate = commands.add_parser("validate", help="validate an EntityLinkage YAML configuration")
|
|
41
|
+
validate.add_argument("path", type=Path)
|
|
42
|
+
|
|
43
|
+
link = commands.add_parser("link", help="link records and write deterministic reports")
|
|
44
|
+
link.add_argument("path", type=Path)
|
|
45
|
+
link.add_argument("--output", type=Path, default=Path("entitylinkage-output"))
|
|
46
|
+
|
|
47
|
+
audit = commands.add_parser("audit", help="summarize linkage status and reason codes")
|
|
48
|
+
audit.add_argument("path", type=Path)
|
|
49
|
+
|
|
50
|
+
inspect = commands.add_parser("inspect", help="show one record's linkage evidence")
|
|
51
|
+
inspect.add_argument("record_id")
|
|
52
|
+
inspect.add_argument("path", type=Path)
|
|
53
|
+
return parser
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def _dispatch(arguments: argparse.Namespace) -> int:
|
|
57
|
+
loaded = load_config(arguments.path)
|
|
58
|
+
if arguments.command == "validate":
|
|
59
|
+
report = {
|
|
60
|
+
"valid": True,
|
|
61
|
+
"entities": len(loaded.entities),
|
|
62
|
+
"records": len(loaded.records),
|
|
63
|
+
}
|
|
64
|
+
print(json.dumps(report, sort_keys=True, separators=(",", ":"), ensure_ascii=False))
|
|
65
|
+
return 0
|
|
66
|
+
|
|
67
|
+
results = Linker(loaded.config).link(loaded.entities, loaded.records)
|
|
68
|
+
if arguments.command == "link":
|
|
69
|
+
write_artifacts(results, arguments.output)
|
|
70
|
+
print(serialize_summary_json(results))
|
|
71
|
+
return 0
|
|
72
|
+
if arguments.command == "audit":
|
|
73
|
+
print(serialize_summary_json(results))
|
|
74
|
+
return int(any(result.status in {"ambiguous", "unresolved"} for result in results))
|
|
75
|
+
if arguments.command == "inspect":
|
|
76
|
+
record = next((item for item in loaded.records if item.id == arguments.record_id), None)
|
|
77
|
+
if record is None:
|
|
78
|
+
raise EntityLinkageError(f"unknown record ID: {arguments.record_id!r}")
|
|
79
|
+
result = next(item for item in results if item.record_id == record.id)
|
|
80
|
+
print(
|
|
81
|
+
json.dumps(
|
|
82
|
+
inspection_payload(record, result),
|
|
83
|
+
sort_keys=True,
|
|
84
|
+
separators=(",", ":"),
|
|
85
|
+
ensure_ascii=False,
|
|
86
|
+
allow_nan=False,
|
|
87
|
+
)
|
|
88
|
+
)
|
|
89
|
+
return 0
|
|
90
|
+
raise EntityLinkageError(f"unknown command: {arguments.command!r}")
|
entitylinkage/config.py
ADDED
|
@@ -0,0 +1,412 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import math
|
|
4
|
+
import re
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
import yaml
|
|
9
|
+
from yaml.constructor import ConstructorError
|
|
10
|
+
from yaml.nodes import MappingNode
|
|
11
|
+
from yaml.resolver import BaseResolver
|
|
12
|
+
|
|
13
|
+
from entitylinkage.errors import ConfigError
|
|
14
|
+
from entitylinkage.model import (
|
|
15
|
+
CandidateRule,
|
|
16
|
+
Entity,
|
|
17
|
+
EqualConstraint,
|
|
18
|
+
LinkageConfig,
|
|
19
|
+
LinkageInput,
|
|
20
|
+
NormalizationConfig,
|
|
21
|
+
Override,
|
|
22
|
+
Record,
|
|
23
|
+
Scalar,
|
|
24
|
+
)
|
|
25
|
+
from entitylinkage.normalize import normalize_text
|
|
26
|
+
|
|
27
|
+
_ATTRIBUTE_KEY = re.compile(r"[^.\x00-\x1f\x7f]+\Z")
|
|
28
|
+
_REASON_CODE = re.compile(r"[a-z][a-z0-9_]*\Z")
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
class _UniqueKeySafeLoader(yaml.SafeLoader):
|
|
32
|
+
pass
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _construct_unique_mapping(
|
|
36
|
+
loader: _UniqueKeySafeLoader, node: MappingNode, deep: bool = False
|
|
37
|
+
) -> dict[Any, Any]:
|
|
38
|
+
loader.flatten_mapping(node)
|
|
39
|
+
result: dict[Any, Any] = {}
|
|
40
|
+
for key_node, value_node in node.value:
|
|
41
|
+
key = loader.construct_object(key_node, deep=deep)
|
|
42
|
+
try:
|
|
43
|
+
duplicate = key in result
|
|
44
|
+
except TypeError as error:
|
|
45
|
+
raise ConstructorError(
|
|
46
|
+
"while constructing a mapping",
|
|
47
|
+
node.start_mark,
|
|
48
|
+
"found an unhashable mapping key",
|
|
49
|
+
key_node.start_mark,
|
|
50
|
+
) from error
|
|
51
|
+
if duplicate:
|
|
52
|
+
raise ConstructorError(
|
|
53
|
+
"while constructing a mapping",
|
|
54
|
+
node.start_mark,
|
|
55
|
+
f"found duplicate YAML key {key!r}",
|
|
56
|
+
key_node.start_mark,
|
|
57
|
+
)
|
|
58
|
+
result[key] = loader.construct_object(value_node, deep=deep)
|
|
59
|
+
return result
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
_UniqueKeySafeLoader.add_constructor(BaseResolver.DEFAULT_MAPPING_TAG, _construct_unique_mapping)
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def load_config(path: Path) -> LinkageInput:
|
|
66
|
+
"""Read and strictly validate an EntityLinkage YAML file or example directory."""
|
|
67
|
+
source = path / "entitylinkage.yaml" if path.is_dir() else path
|
|
68
|
+
try:
|
|
69
|
+
with source.open(encoding="utf-8") as stream:
|
|
70
|
+
document = yaml.load(stream, Loader=_UniqueKeySafeLoader)
|
|
71
|
+
except OSError as error:
|
|
72
|
+
raise ConfigError(
|
|
73
|
+
f"cannot read configuration {source}: {error.strerror or error}"
|
|
74
|
+
) from error
|
|
75
|
+
except yaml.YAMLError as error:
|
|
76
|
+
message = str(error)
|
|
77
|
+
if "duplicate YAML key" in message:
|
|
78
|
+
raise ConfigError(message) from error
|
|
79
|
+
raise ConfigError(f"invalid YAML: {message}") from error
|
|
80
|
+
|
|
81
|
+
root = _mapping(
|
|
82
|
+
document,
|
|
83
|
+
"configuration",
|
|
84
|
+
allowed={"version", "entities", "records", "matching", "overrides"},
|
|
85
|
+
required={"version", "entities", "records", "matching"},
|
|
86
|
+
)
|
|
87
|
+
version = root["version"]
|
|
88
|
+
if type(version) is not int or version != 1:
|
|
89
|
+
raise ConfigError("version must be the integer 1")
|
|
90
|
+
|
|
91
|
+
entities = tuple(
|
|
92
|
+
_parse_entity(item, index) for index, item in enumerate(_list(root["entities"], "entities"))
|
|
93
|
+
)
|
|
94
|
+
records = tuple(
|
|
95
|
+
_parse_record(item, index) for index, item in enumerate(_list(root["records"], "records"))
|
|
96
|
+
)
|
|
97
|
+
_require_unique((entity.id for entity in entities), "entity ID")
|
|
98
|
+
_require_unique((record.id for record in records), "record ID")
|
|
99
|
+
|
|
100
|
+
matching = _mapping(
|
|
101
|
+
root["matching"],
|
|
102
|
+
"matching",
|
|
103
|
+
allowed={"normalization", "candidate_rules", "constraints", "apply_overrides"},
|
|
104
|
+
required={"normalization", "candidate_rules", "apply_overrides"},
|
|
105
|
+
)
|
|
106
|
+
normalization = _parse_normalization(matching["normalization"])
|
|
107
|
+
candidate_rules = tuple(
|
|
108
|
+
_parse_candidate_rule(item, index)
|
|
109
|
+
for index, item in enumerate(_list(matching["candidate_rules"], "matching.candidate_rules"))
|
|
110
|
+
)
|
|
111
|
+
if not candidate_rules:
|
|
112
|
+
raise ConfigError("matching.candidate_rules must not be empty")
|
|
113
|
+
constraints = tuple(
|
|
114
|
+
_parse_constraint(item, index)
|
|
115
|
+
for index, item in enumerate(_list(matching.get("constraints", []), "matching.constraints"))
|
|
116
|
+
)
|
|
117
|
+
rule_ids = [rule.id for rule in candidate_rules] + [constraint.id for constraint in constraints]
|
|
118
|
+
_require_unique(rule_ids, "rule ID")
|
|
119
|
+
|
|
120
|
+
apply_overrides = matching["apply_overrides"]
|
|
121
|
+
if type(apply_overrides) is not bool:
|
|
122
|
+
raise ConfigError("matching.apply_overrides must be a boolean")
|
|
123
|
+
|
|
124
|
+
overrides = tuple(
|
|
125
|
+
_parse_override(item, index)
|
|
126
|
+
for index, item in enumerate(_list(root.get("overrides", []), "overrides"))
|
|
127
|
+
)
|
|
128
|
+
_require_unique((override.record_id for override in overrides), "override record ID")
|
|
129
|
+
if overrides and not apply_overrides:
|
|
130
|
+
raise ConfigError("overrides require matching.apply_overrides: true")
|
|
131
|
+
|
|
132
|
+
_validate_entity_labels(entities, normalization)
|
|
133
|
+
_validate_references(candidate_rules, constraints)
|
|
134
|
+
_validate_overrides(entities, records, overrides)
|
|
135
|
+
|
|
136
|
+
return LinkageInput(
|
|
137
|
+
entities=entities,
|
|
138
|
+
records=records,
|
|
139
|
+
config=LinkageConfig(
|
|
140
|
+
normalization=normalization,
|
|
141
|
+
candidate_rules=candidate_rules,
|
|
142
|
+
apply_overrides=apply_overrides,
|
|
143
|
+
constraints=constraints,
|
|
144
|
+
overrides=overrides,
|
|
145
|
+
),
|
|
146
|
+
)
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
def _mapping(
|
|
150
|
+
value: object,
|
|
151
|
+
context: str,
|
|
152
|
+
*,
|
|
153
|
+
allowed: set[str],
|
|
154
|
+
required: set[str] = frozenset(),
|
|
155
|
+
) -> dict[str, object]:
|
|
156
|
+
if not isinstance(value, dict):
|
|
157
|
+
raise ConfigError(f"{context} must be a mapping")
|
|
158
|
+
if any(type(key) is not str for key in value):
|
|
159
|
+
raise ConfigError(f"{context} keys must be strings")
|
|
160
|
+
keys = set(value)
|
|
161
|
+
unknown = sorted(keys - allowed)
|
|
162
|
+
missing = sorted(required - keys)
|
|
163
|
+
if unknown:
|
|
164
|
+
raise ConfigError(f"{context} contains unknown keys: {', '.join(unknown)}")
|
|
165
|
+
if missing:
|
|
166
|
+
raise ConfigError(f"{context} is missing required keys: {', '.join(missing)}")
|
|
167
|
+
return value
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
def _list(value: object, context: str) -> list[object]:
|
|
171
|
+
if not isinstance(value, list):
|
|
172
|
+
raise ConfigError(f"{context} must be a list")
|
|
173
|
+
return value
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
def _string(value: object, context: str, *, nonempty: bool = False) -> str:
|
|
177
|
+
if type(value) is not str or (nonempty and not value.strip()):
|
|
178
|
+
suffix = " nonempty" if nonempty else ""
|
|
179
|
+
raise ConfigError(f"{context} must be a{suffix} string")
|
|
180
|
+
return value
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
def _boolean(value: object, context: str) -> bool:
|
|
184
|
+
if type(value) is not bool:
|
|
185
|
+
raise ConfigError(f"{context} must be a boolean")
|
|
186
|
+
return value
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
def _scalar(value: object, context: str) -> Scalar:
|
|
190
|
+
if value is None or type(value) in (str, int, bool):
|
|
191
|
+
return value
|
|
192
|
+
if type(value) is float:
|
|
193
|
+
if not math.isfinite(value):
|
|
194
|
+
raise ConfigError(f"{context} must be a finite scalar")
|
|
195
|
+
return value
|
|
196
|
+
raise ConfigError(f"{context} must be a flat scalar (string, integer, float, boolean, or null)")
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
def _attributes(value: object, context: str) -> dict[str, Scalar]:
|
|
200
|
+
if not isinstance(value, dict):
|
|
201
|
+
raise ConfigError(f"{context} must be a flat mapping")
|
|
202
|
+
attributes: dict[str, Scalar] = {}
|
|
203
|
+
for key, item in value.items():
|
|
204
|
+
attribute_key = _string(key, f"{context} key", nonempty=True)
|
|
205
|
+
if _ATTRIBUTE_KEY.fullmatch(attribute_key) is None:
|
|
206
|
+
raise ConfigError(f"{context} key {attribute_key!r} must be one flat field name")
|
|
207
|
+
attributes[attribute_key] = _scalar(item, f"{context}.{attribute_key}")
|
|
208
|
+
return attributes
|
|
209
|
+
|
|
210
|
+
|
|
211
|
+
def _parse_entity(value: object, index: int) -> Entity:
|
|
212
|
+
context = f"entities[{index}]"
|
|
213
|
+
item = _mapping(
|
|
214
|
+
value,
|
|
215
|
+
context,
|
|
216
|
+
allowed={"id", "name", "aliases", "attributes"},
|
|
217
|
+
required={"id", "name"},
|
|
218
|
+
)
|
|
219
|
+
aliases = tuple(
|
|
220
|
+
_string(alias, f"{context}.aliases[{alias_index}]", nonempty=True)
|
|
221
|
+
for alias_index, alias in enumerate(_list(item.get("aliases", []), f"{context}.aliases"))
|
|
222
|
+
)
|
|
223
|
+
return Entity(
|
|
224
|
+
id=_string(item["id"], f"{context}.id", nonempty=True),
|
|
225
|
+
name=_string(item["name"], f"{context}.name", nonempty=True),
|
|
226
|
+
aliases=aliases,
|
|
227
|
+
attributes=_attributes(item.get("attributes", {}), f"{context}.attributes"),
|
|
228
|
+
)
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
def _parse_record(value: object, index: int) -> Record:
|
|
232
|
+
context = f"records[{index}]"
|
|
233
|
+
item = _mapping(
|
|
234
|
+
value,
|
|
235
|
+
context,
|
|
236
|
+
allowed={"id", "name", "attributes", "applicable", "not_applicable_reason"},
|
|
237
|
+
required={"id"},
|
|
238
|
+
)
|
|
239
|
+
name = item.get("name")
|
|
240
|
+
if name is not None:
|
|
241
|
+
name = _string(name, f"{context}.name")
|
|
242
|
+
applicable = _boolean(item.get("applicable", True), f"{context}.applicable")
|
|
243
|
+
not_applicable_reason = item.get("not_applicable_reason")
|
|
244
|
+
if not_applicable_reason is not None:
|
|
245
|
+
not_applicable_reason = _reason_code(
|
|
246
|
+
not_applicable_reason, f"{context}.not_applicable_reason"
|
|
247
|
+
)
|
|
248
|
+
if not applicable and not_applicable_reason is None:
|
|
249
|
+
raise ConfigError(f"{context} with applicable: false requires not_applicable_reason")
|
|
250
|
+
return Record(
|
|
251
|
+
id=_string(item["id"], f"{context}.id", nonempty=True),
|
|
252
|
+
name=name,
|
|
253
|
+
attributes=_attributes(item.get("attributes", {}), f"{context}.attributes"),
|
|
254
|
+
applicable=applicable,
|
|
255
|
+
not_applicable_reason=not_applicable_reason,
|
|
256
|
+
)
|
|
257
|
+
|
|
258
|
+
|
|
259
|
+
def _parse_normalization(value: object) -> NormalizationConfig:
|
|
260
|
+
item = _mapping(
|
|
261
|
+
value,
|
|
262
|
+
"matching.normalization",
|
|
263
|
+
allowed={"case_fold", "punctuation"},
|
|
264
|
+
required={"case_fold", "punctuation"},
|
|
265
|
+
)
|
|
266
|
+
case_fold = _boolean(item["case_fold"], "matching.normalization.case_fold")
|
|
267
|
+
punctuation = _string(item["punctuation"], "matching.normalization.punctuation")
|
|
268
|
+
if punctuation not in {"preserve", "remove", "space"}:
|
|
269
|
+
raise ConfigError("matching.normalization.punctuation must be preserve, remove, or space")
|
|
270
|
+
return NormalizationConfig(case_fold=case_fold, punctuation=punctuation) # type: ignore[arg-type]
|
|
271
|
+
|
|
272
|
+
|
|
273
|
+
def _parse_candidate_rule(value: object, index: int) -> CandidateRule:
|
|
274
|
+
context = f"matching.candidate_rules[{index}]"
|
|
275
|
+
item = _mapping(
|
|
276
|
+
value,
|
|
277
|
+
context,
|
|
278
|
+
allowed={"id", "type", "entity_field", "record_field"},
|
|
279
|
+
required={"id", "type"},
|
|
280
|
+
)
|
|
281
|
+
rule_id = _string(item["id"], f"{context}.id", nonempty=True)
|
|
282
|
+
rule_type = _string(item["type"], f"{context}.type")
|
|
283
|
+
if rule_type == "normalized_name":
|
|
284
|
+
if "entity_field" in item or "record_field" in item:
|
|
285
|
+
raise ConfigError(f"{context} normalized_name rule does not accept field references")
|
|
286
|
+
return CandidateRule(id=rule_id, type="normalized_name")
|
|
287
|
+
if rule_type == "exact_value":
|
|
288
|
+
if "entity_field" not in item or "record_field" not in item:
|
|
289
|
+
raise ConfigError(f"{context} exact_value rule requires entity_field and record_field")
|
|
290
|
+
return CandidateRule(
|
|
291
|
+
id=rule_id,
|
|
292
|
+
type="exact_value",
|
|
293
|
+
entity_field=_field_reference(item["entity_field"], f"{context}.entity_field"),
|
|
294
|
+
record_field=_field_reference(item["record_field"], f"{context}.record_field"),
|
|
295
|
+
)
|
|
296
|
+
raise ConfigError(f"{context} has unsupported rule type {rule_type!r}")
|
|
297
|
+
|
|
298
|
+
|
|
299
|
+
def _parse_constraint(value: object, index: int) -> EqualConstraint:
|
|
300
|
+
context = f"matching.constraints[{index}]"
|
|
301
|
+
item = _mapping(
|
|
302
|
+
value,
|
|
303
|
+
context,
|
|
304
|
+
allowed={"id", "type", "entity_field", "record_field", "missing"},
|
|
305
|
+
required={"id", "type", "entity_field", "record_field", "missing"},
|
|
306
|
+
)
|
|
307
|
+
constraint_type = _string(item["type"], f"{context}.type")
|
|
308
|
+
if constraint_type != "equal":
|
|
309
|
+
raise ConfigError(f"{context} has unsupported constraint type {constraint_type!r}")
|
|
310
|
+
missing = _string(item["missing"], f"{context}.missing")
|
|
311
|
+
if missing not in {"ignore", "reject"}:
|
|
312
|
+
raise ConfigError(f"{context}.missing must be ignore or reject")
|
|
313
|
+
return EqualConstraint(
|
|
314
|
+
id=_string(item["id"], f"{context}.id", nonempty=True),
|
|
315
|
+
entity_field=_field_reference(item["entity_field"], f"{context}.entity_field"),
|
|
316
|
+
record_field=_field_reference(item["record_field"], f"{context}.record_field"),
|
|
317
|
+
missing=missing, # type: ignore[arg-type]
|
|
318
|
+
)
|
|
319
|
+
|
|
320
|
+
|
|
321
|
+
def _field_reference(value: object, context: str) -> str:
|
|
322
|
+
reference = _string(value, context, nonempty=True)
|
|
323
|
+
if reference == "name":
|
|
324
|
+
return reference
|
|
325
|
+
if reference.startswith("attributes."):
|
|
326
|
+
key = reference.removeprefix("attributes.")
|
|
327
|
+
if _ATTRIBUTE_KEY.fullmatch(key):
|
|
328
|
+
return reference
|
|
329
|
+
raise ConfigError(f"{context} is an invalid field reference: {reference!r}")
|
|
330
|
+
|
|
331
|
+
|
|
332
|
+
def _reason_code(value: object, context: str) -> str:
|
|
333
|
+
reason = _string(value, context)
|
|
334
|
+
if _REASON_CODE.fullmatch(reason) is None:
|
|
335
|
+
raise ConfigError(f"{context} must be a reason code matching [a-z][a-z0-9_]*")
|
|
336
|
+
return reason
|
|
337
|
+
|
|
338
|
+
|
|
339
|
+
def _parse_override(value: object, index: int) -> Override:
|
|
340
|
+
context = f"overrides[{index}]"
|
|
341
|
+
item = _mapping(
|
|
342
|
+
value,
|
|
343
|
+
context,
|
|
344
|
+
allowed={"record_id", "status", "entity_id", "reason"},
|
|
345
|
+
required={"record_id", "status", "reason"},
|
|
346
|
+
)
|
|
347
|
+
status = _string(item["status"], f"{context}.status")
|
|
348
|
+
if status == "resolved":
|
|
349
|
+
if "entity_id" not in item:
|
|
350
|
+
raise ConfigError(f"{context} resolved override requires entity_id")
|
|
351
|
+
entity_id = _string(item["entity_id"], f"{context}.entity_id", nonempty=True)
|
|
352
|
+
elif status == "not_applicable":
|
|
353
|
+
if "entity_id" in item:
|
|
354
|
+
raise ConfigError(f"{context} not_applicable override must omit entity_id")
|
|
355
|
+
entity_id = None
|
|
356
|
+
else:
|
|
357
|
+
raise ConfigError(f"{context} has unsupported override status {status!r}")
|
|
358
|
+
return Override(
|
|
359
|
+
record_id=_string(item["record_id"], f"{context}.record_id", nonempty=True),
|
|
360
|
+
status=status, # type: ignore[arg-type]
|
|
361
|
+
reason=_reason_code(item["reason"], f"{context}.reason"),
|
|
362
|
+
entity_id=entity_id,
|
|
363
|
+
)
|
|
364
|
+
|
|
365
|
+
|
|
366
|
+
def _require_unique(values, context: str) -> None:
|
|
367
|
+
seen: set[str] = set()
|
|
368
|
+
for value in values:
|
|
369
|
+
if value in seen:
|
|
370
|
+
raise ConfigError(f"duplicate {context}: {value!r}")
|
|
371
|
+
seen.add(value)
|
|
372
|
+
|
|
373
|
+
|
|
374
|
+
def _validate_entity_labels(
|
|
375
|
+
entities: tuple[Entity, ...], normalization: NormalizationConfig
|
|
376
|
+
) -> None:
|
|
377
|
+
for entity in entities:
|
|
378
|
+
labels = [entity.name, *entity.aliases]
|
|
379
|
+
normalized = [normalize_text(label, normalization) for label in labels]
|
|
380
|
+
if len(normalized) != len(set(normalized)):
|
|
381
|
+
raise ConfigError(f"entity {entity.id!r} has a duplicate normalized label")
|
|
382
|
+
|
|
383
|
+
|
|
384
|
+
def _validate_references(
|
|
385
|
+
candidate_rules: tuple[CandidateRule, ...], constraints: tuple[EqualConstraint, ...]
|
|
386
|
+
) -> None:
|
|
387
|
+
for rule in candidate_rules:
|
|
388
|
+
if rule.type == "exact_value":
|
|
389
|
+
assert rule.entity_field is not None and rule.record_field is not None
|
|
390
|
+
_field_reference(rule.entity_field, f"candidate rule {rule.id!r}.entity_field")
|
|
391
|
+
_field_reference(rule.record_field, f"candidate rule {rule.id!r}.record_field")
|
|
392
|
+
for constraint in constraints:
|
|
393
|
+
_field_reference(constraint.entity_field, f"constraint {constraint.id!r}.entity_field")
|
|
394
|
+
_field_reference(constraint.record_field, f"constraint {constraint.id!r}.record_field")
|
|
395
|
+
|
|
396
|
+
|
|
397
|
+
def _validate_overrides(
|
|
398
|
+
entities: tuple[Entity, ...], records: tuple[Record, ...], overrides: tuple[Override, ...]
|
|
399
|
+
) -> None:
|
|
400
|
+
entities_by_id = {entity.id: entity for entity in entities}
|
|
401
|
+
records_by_id = {record.id: record for record in records}
|
|
402
|
+
for override in overrides:
|
|
403
|
+
if override.record_id not in records_by_id:
|
|
404
|
+
raise ConfigError(f"override refers to unknown record {override.record_id!r}")
|
|
405
|
+
record = records_by_id[override.record_id]
|
|
406
|
+
if not record.applicable:
|
|
407
|
+
raise ConfigError(
|
|
408
|
+
f"override for record {override.record_id!r} contradicts "
|
|
409
|
+
"an explicitly not applicable record"
|
|
410
|
+
)
|
|
411
|
+
if override.entity_id is not None and override.entity_id not in entities_by_id:
|
|
412
|
+
raise ConfigError(f"override refers to unknown entity {override.entity_id!r}")
|