rag-wright 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (184) hide show
  1. rag_wright/__init__.py +13 -0
  2. rag_wright/api/__init__.py +33 -0
  3. rag_wright/api/config.py +59 -0
  4. rag_wright/api/discover.py +70 -0
  5. rag_wright/api/documents.py +39 -0
  6. rag_wright/api/ids.py +31 -0
  7. rag_wright/api/invoke.py +99 -0
  8. rag_wright/api/kg.py +61 -0
  9. rag_wright/api/mcp.py +94 -0
  10. rag_wright/api/usage.py +30 -0
  11. rag_wright/api/workspace.py +85 -0
  12. rag_wright/capabilities/__init__.py +8 -0
  13. rag_wright/capabilities/answer_generator.py +427 -0
  14. rag_wright/capabilities/ard.py +286 -0
  15. rag_wright/capabilities/assertion_extraction.py +79 -0
  16. rag_wright/capabilities/chunk_read.py +58 -0
  17. rag_wright/capabilities/chunk_write.py +163 -0
  18. rag_wright/capabilities/claim_extraction.py +153 -0
  19. rag_wright/capabilities/clause_exception_linking.py +117 -0
  20. rag_wright/capabilities/compliance_judgment.py +322 -0
  21. rag_wright/capabilities/compliance_store.py +87 -0
  22. rag_wright/capabilities/contract_kg_serve.py +156 -0
  23. rag_wright/capabilities/contract_kg_store.py +251 -0
  24. rag_wright/capabilities/dg_extraction.py +585 -0
  25. rag_wright/capabilities/disambiguation.py +163 -0
  26. rag_wright/capabilities/document_parse.py +87 -0
  27. rag_wright/capabilities/document_scope.py +49 -0
  28. rag_wright/capabilities/embedding.py +164 -0
  29. rag_wright/capabilities/embedding_profiles.py +43 -0
  30. rag_wright/capabilities/entity_resolution.py +154 -0
  31. rag_wright/capabilities/fusion.py +64 -0
  32. rag_wright/capabilities/graph_extraction.py +243 -0
  33. rag_wright/capabilities/graph_query.py +73 -0
  34. rag_wright/capabilities/graph_storage.py +111 -0
  35. rag_wright/capabilities/highlight_serve.py +142 -0
  36. rag_wright/capabilities/hybrid_search.py +65 -0
  37. rag_wright/capabilities/invoke.py +31 -0
  38. rag_wright/capabilities/jev_decision.py +38 -0
  39. rag_wright/capabilities/manifests.py +872 -0
  40. rag_wright/capabilities/okf_navigate.py +456 -0
  41. rag_wright/capabilities/parsing.py +286 -0
  42. rag_wright/capabilities/property_boosted_retrieval.py +125 -0
  43. rag_wright/capabilities/query_function_classifier.py +94 -0
  44. rag_wright/capabilities/query_understanding.py +109 -0
  45. rag_wright/capabilities/registry.py +262 -0
  46. rag_wright/capabilities/remote_encoders.py +94 -0
  47. rag_wright/capabilities/requirement_extraction.py +247 -0
  48. rag_wright/capabilities/reranking.py +123 -0
  49. rag_wright/capabilities/retrieval_core.py +126 -0
  50. rag_wright/capabilities/rlm_chunking.py +808 -0
  51. rag_wright/capabilities/rlm_synthesis.py +316 -0
  52. rag_wright/capabilities/scan_quality.py +136 -0
  53. rag_wright/capabilities/span_relevance_judgment.py +191 -0
  54. rag_wright/capabilities/vision_to_text.py +85 -0
  55. rag_wright/capabilities/vlm_ocr.py +85 -0
  56. rag_wright/contracts/__init__.py +6 -0
  57. rag_wright/contracts/chunk.py +79 -0
  58. rag_wright/contracts/compliance.py +303 -0
  59. rag_wright/contracts/contract_meta.py +27 -0
  60. rag_wright/contracts/extraction.py +130 -0
  61. rag_wright/contracts/function.py +167 -0
  62. rag_wright/contracts/function_routing.py +91 -0
  63. rag_wright/contracts/highlight.py +74 -0
  64. rag_wright/contracts/identifiers.py +153 -0
  65. rag_wright/contracts/jurisdiction.py +96 -0
  66. rag_wright/contracts/ontology.py +142 -0
  67. rag_wright/contracts/property.py +201 -0
  68. rag_wright/contracts/provenance.py +78 -0
  69. rag_wright/contracts/query_intent.py +53 -0
  70. rag_wright/contracts/span.py +76 -0
  71. rag_wright/contracts/value_match.py +84 -0
  72. rag_wright/corpus/__init__.py +0 -0
  73. rag_wright/corpus/canonicalize.py +116 -0
  74. rag_wright/corpus/cuad.py +153 -0
  75. rag_wright/corpus/cuad_ingestion.py +72 -0
  76. rag_wright/corpus/document_parser.py +299 -0
  77. rag_wright/corpus/edgar.py +231 -0
  78. rag_wright/corpus/gcs_ingestion.py +120 -0
  79. rag_wright/corpus/http.py +110 -0
  80. rag_wright/corpus/selection.py +152 -0
  81. rag_wright/mcp/__init__.py +11 -0
  82. rag_wright/mcp/compliance_server.py +299 -0
  83. rag_wright/mcp/intra_document_qa_server.py +170 -0
  84. rag_wright/mcp/relational_qa_server.py +171 -0
  85. rag_wright/mcp/session_store.py +64 -0
  86. rag_wright/mcp/typed_property_retrieval_server.py +191 -0
  87. rag_wright/models/__init__.py +8 -0
  88. rag_wright/models/profiles.py +331 -0
  89. rag_wright/models/seam.py +497 -0
  90. rag_wright/models/tag_structured.py +285 -0
  91. rag_wright/models/tracing.py +179 -0
  92. rag_wright/models/usage.py +102 -0
  93. rag_wright/okf/__init__.py +11 -0
  94. rag_wright/okf/compile.py +292 -0
  95. rag_wright/okf/document.py +47 -0
  96. rag_wright/okf/enrich.py +176 -0
  97. rag_wright/okf/links.py +190 -0
  98. rag_wright/okf/lint.py +105 -0
  99. rag_wright/ontology/__init__.py +6 -0
  100. rag_wright/ontology/_generated_template_meta.py +60 -0
  101. rag_wright/ontology/_generated_vocab.py +52 -0
  102. rag_wright/ontology/clause_template.py +964 -0
  103. rag_wright/ontology/codegen.py +84 -0
  104. rag_wright/ontology/compliance_bridge.ttl +186 -0
  105. rag_wright/ontology/contract_bridge.ttl +2685 -0
  106. rag_wright/ontology/contract_taxonomy.py +24 -0
  107. rag_wright/ontology/derive.py +58 -0
  108. rag_wright/ontology/loader.py +435 -0
  109. rag_wright/ontology/packs/ftc_16cfr255.ttl +29 -0
  110. rag_wright/ontology/registry.py +87 -0
  111. rag_wright/ontology/template_introspect.py +100 -0
  112. rag_wright/py.typed +0 -0
  113. rag_wright/reference/__init__.py +2 -0
  114. rag_wright/reference/compliance.py +41 -0
  115. rag_wright/reference/contract_seam.py +123 -0
  116. rag_wright/skills/__init__.py +7 -0
  117. rag_wright/skills/claim_extraction/SKILL.md +47 -0
  118. rag_wright/skills/claim_extraction/__init__.py +1 -0
  119. rag_wright/skills/claim_extraction/template.py +50 -0
  120. rag_wright/skills/compliance_judgment/SKILL.md +59 -0
  121. rag_wright/skills/corpus_ingest/SKILL.md +106 -0
  122. rag_wright/skills/extraction_semantic_judge/SKILL.md +51 -0
  123. rag_wright/skills/extraction_semantic_judge/__init__.py +1 -0
  124. rag_wright/skills/generation/SKILL.md +64 -0
  125. rag_wright/skills/generation/__init__.py +1 -0
  126. rag_wright/skills/generic_compliance_judgment/SKILL.md +58 -0
  127. rag_wright/skills/okf_navigate/SKILL.md +137 -0
  128. rag_wright/skills/requirement_extraction/SKILL.md +47 -0
  129. rag_wright/skills/requirement_extraction/__init__.py +1 -0
  130. rag_wright/skills/requirement_extraction/template.py +50 -0
  131. rag_wright/skills/rlm/SKILL.md +186 -0
  132. rag_wright/skills/rlm/__init__.py +31 -0
  133. rag_wright/skills/rlm/agent.py +292 -0
  134. rag_wright/skills/span_relevance_judgment/SKILL.md +67 -0
  135. rag_wright/skills/vision_to_text/SKILL.md +36 -0
  136. rag_wright/skills/vision_to_text/__init__.py +1 -0
  137. rag_wright/spans/__init__.py +1 -0
  138. rag_wright/spans/boundary.py +78 -0
  139. rag_wright/spans/clause_function_classifier.py +490 -0
  140. rag_wright/spans/clause_kg_extractor.py +337 -0
  141. rag_wright/spans/cuad_labels.py +81 -0
  142. rag_wright/spans/dim_classifier.py +158 -0
  143. rag_wright/spans/dim_fleet.json +411 -0
  144. rag_wright/spans/function_classifier.py +77 -0
  145. rag_wright/spans/function_families.py +62 -0
  146. rag_wright/spans/hybrid_classifier.py +103 -0
  147. rag_wright/spans/legalbert_classifier.py +83 -0
  148. rag_wright/spans/model_capabilities.py +107 -0
  149. rag_wright/spans/new_function_labels.py +111 -0
  150. rag_wright/spans/page_map.py +68 -0
  151. rag_wright/spans/property_extractor.py +365 -0
  152. rag_wright/spans/property_grounding.py +182 -0
  153. rag_wright/spans/reclassify.py +77 -0
  154. rag_wright/spans/scarce_function_labels.py +105 -0
  155. rag_wright/spans/segment.py +341 -0
  156. rag_wright/spans/semantic_judge.py +197 -0
  157. rag_wright/spans/symbolic_validation.py +131 -0
  158. rag_wright/spans/tag_clause_extractor.py +182 -0
  159. rag_wright/store/__init__.py +6 -0
  160. rag_wright/store/arcadedb.py +1135 -0
  161. rag_wright/store/chunk_text.py +66 -0
  162. rag_wright/store/seam.py +213 -0
  163. rag_wright/subgraphs/__init__.py +0 -0
  164. rag_wright/subgraphs/async_ingestion.py +204 -0
  165. rag_wright/subgraphs/compliance_check.py +1042 -0
  166. rag_wright/subgraphs/compliance_ingestion.py +306 -0
  167. rag_wright/subgraphs/contract_ingestion_pipeline.py +999 -0
  168. rag_wright/subgraphs/graph_extraction.py +102 -0
  169. rag_wright/subgraphs/intra_document_qa.py +328 -0
  170. rag_wright/subgraphs/observability.py +140 -0
  171. rag_wright/subgraphs/query_constraint_extraction.py +73 -0
  172. rag_wright/subgraphs/relational_qa.py +165 -0
  173. rag_wright/subgraphs/requirement_extraction.py +137 -0
  174. rag_wright/subgraphs/scaffold.py +65 -0
  175. rag_wright/subgraphs/semantic_chunking.py +183 -0
  176. rag_wright/subgraphs/typed_clause_extraction.py +172 -0
  177. rag_wright/subgraphs/typed_property_retrieval.py +278 -0
  178. rag_wright/util/__init__.py +1 -0
  179. rag_wright/util/concurrent.py +153 -0
  180. rag_wright/util/spacy_model.py +45 -0
  181. rag_wright-0.1.0.dist-info/METADATA +168 -0
  182. rag_wright-0.1.0.dist-info/RECORD +184 -0
  183. rag_wright-0.1.0.dist-info/WHEEL +4 -0
  184. rag_wright-0.1.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,24 @@
1
+ """The reference CONTRACT domain pack's entity-node + entity-relationship taxonomy (DD-5/DD-7, ADR-0066/0117).
2
+
3
+ De-domaining (DD-5): the engine's generic primitives + contracts are taxonomy-free -- `entity_type` and
4
+ `relationship_type` are plain strings the CALLER (a domain graph) names. The closed value sets below are DOMAIN
5
+ knowledge.
6
+
7
+ ADR-0066 end-state (DD-7): those values are now declared in `contract_bridge.ttl` (the single source of truth) and
8
+ rendered into `_generated_vocab.py` by codegen, CI-diff-enforced (`tests/ontology/test_generated_vocab_in_sync.py`).
9
+ This module is the stable import surface the domain builders use -- it RE-EXPORTS the generated constants, so no
10
+ code hardcodes the values. To change the taxonomy, edit the ttl and regenerate; never edit the generated file or
11
+ re-add literals here. A new domain declares its own entity/edge types in its own pack the same way.
12
+ """
13
+ from __future__ import annotations
14
+
15
+ from rag_wright.ontology._generated_vocab import ( # generated FROM contract_bridge.ttl (DD-7); do not hardcode here
16
+ AFFILIATE_OF,
17
+ CONTRACTS_WITH,
18
+ ENTITY_TYPES,
19
+ ORGANIZATION,
20
+ PERSON,
21
+ RELATIONSHIP_TYPES,
22
+ )
23
+
24
+ __all__ = ["ORGANIZATION", "PERSON", "CONTRACTS_WITH", "AFFILIATE_OF", "ENTITY_TYPES", "RELATIONSHIP_TYPES"]
@@ -0,0 +1,58 @@
1
+ """Ontology derivation (T8, FR-C.8): reconcile the T4 ontology against the real CUAD data.
2
+
3
+ T4 declared the 41 `ClauseCategory` values from the published CUAD label set. This module confirms
4
+ they match the actual `master_clauses.csv` columns and produces a CSV-column -> canonical-category
5
+ mapping. The CSV headers carry artifacts the reconciliation absorbs so downstream annotation reading
6
+ (T9) binds to the canonical categories regardless: paired answer columns use inconsistent spacing
7
+ (`-Answer` and `- Answer`), and some names are cased differently (`Ip Ownership Assignment` vs the
8
+ ontology's canonical `IP Ownership Assignment`). Matching is spacing-tolerant on answer columns and
9
+ case-insensitive on category names; the ontology keeps its canonical casing.
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ import re
15
+
16
+ from pydantic import BaseModel
17
+
18
+ from rag_wright.contracts.ontology import ClauseCategory
19
+
20
+ _ANSWER_SUFFIX = re.compile(r"-\s*answer$", re.IGNORECASE)
21
+
22
+
23
+ def clause_category_columns(header: list[str]) -> list[str]:
24
+ """The category columns of `master_clauses.csv`: everything but `Filename` and answer columns."""
25
+ return [
26
+ col
27
+ for col in header
28
+ if col.strip().lower() != "filename" and not _ANSWER_SUFFIX.search(col.strip())
29
+ ]
30
+
31
+
32
+ class Reconciliation(BaseModel):
33
+ """The result of reconciling CSV category columns against the `ClauseCategory` ontology."""
34
+
35
+ matched: dict[str, ClauseCategory] # CSV column -> canonical category
36
+ missing: list[ClauseCategory] # ontology categories with no CSV column
37
+ extra: list[str] # CSV columns matching no ontology category
38
+
39
+ @property
40
+ def ok(self) -> bool:
41
+ return not self.missing and not self.extra
42
+
43
+
44
+ def reconcile_clause_categories(csv_columns: list[str]) -> Reconciliation:
45
+ """Match CSV category columns to `ClauseCategory` (case-insensitive), reporting gaps both ways."""
46
+ by_norm = {category.value.lower(): category for category in ClauseCategory}
47
+ matched: dict[str, ClauseCategory] = {}
48
+ extra: list[str] = []
49
+ seen: set[ClauseCategory] = set()
50
+ for column in csv_columns:
51
+ category = by_norm.get(column.strip().lower())
52
+ if category is None:
53
+ extra.append(column)
54
+ else:
55
+ matched[column] = category
56
+ seen.add(category)
57
+ missing = [category for category in ClauseCategory if category not in seen]
58
+ return Reconciliation(matched=matched, missing=missing, extra=extra)
@@ -0,0 +1,435 @@
1
+ """ADR-0066: the runtime loader for the contract ontology `.ttl` -- the seed of the ontology-as-source-of-truth
2
+ substrate. Parses `contract_bridge.ttl` into the domain knowledge structures the engine consumes: the closed
3
+ vocabularies, scalar/list cardinality, the function -> applicable-dimensions applicability, the deontic polarity
4
+ (+ restrictive functions), and the value rollups.
5
+
6
+ Phase 0 uses this only to PROVE the ttl reproduces today's Python constants (the equivalence gate). Phase 1
7
+ generates the Python vocab/enums FROM this loader; Phase 2 hands the SHACL shapes straight to pyshacl. The `.ttl`
8
+ uses a uniform, label-keyed layer (a `cbr:PropertyDimension` node per dimension with `owl:oneOf` + a
9
+ `cbr:cardinality` tag; a `sh:NodeShape` per clause function; `skos:broader` for rollups), so this loader is
10
+ robust to IRI encoding -- it reads `rdfs:label`, never decodes an IRI.
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ import re
16
+ from dataclasses import dataclass, field
17
+ from functools import lru_cache
18
+ from pathlib import Path
19
+
20
+ from rdflib import Graph
21
+ from rdflib.collection import Collection
22
+ from rdflib.namespace import OWL, RDF, RDFS, SH, SKOS
23
+
24
+ _TTL_PATH = Path(__file__).with_name("contract_bridge.ttl")
25
+ _COMPLIANCE_TTL_PATH = Path(__file__).with_name("compliance_bridge.ttl")
26
+
27
+
28
+ def load_compliance_vocab(path: Path | str = _COMPLIANCE_TTL_PATH) -> dict[str, set[str]]:
29
+ """ADR-0066 P3b: the closed vocabularies declared in compliance_bridge.ttl, keyed by class local-name
30
+ (`DeonticType`, `ClaimType`, `Severity`, `RuleScope`, `Verdict`) -> the set of `owl:oneOf` value local-names.
31
+ The Python enums in contracts/compliance.py are drift-locked to this (the ttl is the source of truth)."""
32
+ g = Graph()
33
+ g.parse(str(path), format="turtle")
34
+ out: dict[str, set[str]] = {}
35
+ for cls in g.subjects(OWL.oneOf, None):
36
+ local = str(cls).rsplit("#", 1)[-1]
37
+ members = {str(m).rsplit("#", 1)[-1] for m in Collection(g, g.value(cls, OWL.oneOf))}
38
+ out[local] = members
39
+ return out
40
+
41
+
42
+ _CMP = "https://ragwright.local/ontology/compliance-bridge#"
43
+
44
+
45
+ @lru_cache(maxsize=4)
46
+ def load_deontic_cues(path: str = str(_COMPLIANCE_TTL_PATH)) -> frozenset[str]:
47
+ """ADR-0066 P3c (Gap 1): the deontic CUES declared in compliance_bridge.ttl (`cmp:cue` on each deontic type) --
48
+ the lexical markers of operative normative force. The requirement-ingestion validity gate uses them: a section
49
+ with none of these cues is non-operative and is skipped. Cached per path."""
50
+ from rdflib import URIRef
51
+
52
+ g = Graph()
53
+ g.parse(path, format="turtle")
54
+ return frozenset(str(v).strip().lower() for v in g.objects(None, URIRef(_CMP + "cue")) if str(v).strip())
55
+
56
+
57
+ @lru_cache(maxsize=4)
58
+ def load_deontic_cue_map(path: str = str(_COMPLIANCE_TTL_PATH)) -> dict[str, str]:
59
+ """CIC-0 (ADR-0066): the deontic CUE -> deontic TYPE map authored in compliance_bridge.ttl (`cmp:cue` on each
60
+ `cmp:DeonticType`), e.g. {'must': 'obligation', 'must not': 'prohibition', 'may': 'permission'}. Keys are
61
+ lowercased cue phrases; values are the DeonticType local-names. This is the ttl-driven source for the ingest
62
+ cue-RULE (`deontic_type_of`): a rule's deontic_type is derived deterministically from its text's cue instead of
63
+ from an LLM field. Shares its cue set with `load_deontic_cues` (the operative gate). Cached per path."""
64
+ from rdflib import URIRef
65
+
66
+ g = Graph()
67
+ g.parse(path, format="turtle")
68
+ cue = URIRef(_CMP + "cue")
69
+ out: dict[str, str] = {}
70
+ for subj, obj in g.subject_objects(cue):
71
+ value = str(obj).strip().lower()
72
+ if value:
73
+ out[value] = str(subj).rsplit("#", 1)[-1]
74
+ return out
75
+
76
+
77
+ @lru_cache(maxsize=1)
78
+ def _deontic_cue_type_pattern() -> tuple[re.Pattern, dict[str, str]]:
79
+ """The compiled cue regex (longest cue first, so 'must not' is tried before 'must') + the cue->type map."""
80
+ cue_map = load_deontic_cue_map()
81
+ cues = sorted(cue_map, key=len, reverse=True)
82
+ return re.compile(r"\b(?:" + "|".join(re.escape(c) for c in cues) + r")\b", re.IGNORECASE), cue_map
83
+
84
+
85
+ def deontic_type_of(text: str) -> str | None:
86
+ """CIC-0 (ADR-0066): the deontic TYPE of a rule span, from its FIRST deontic cue (ttl `cmp:cue`), longest cue
87
+ first so 'must not' / 'may not' (prohibition) win over 'must' / 'may'. Returns the DeonticType local-name
88
+ (obligation / prohibition / permission) or None when the text carries no cue (non-operative). The deterministic
89
+ cue-rule that replaces the LLM's deontic_type field at ingest; a non-None result also means the span is
90
+ operative (same cue basis as `is_operative`)."""
91
+ if not text:
92
+ return None
93
+ pattern, cue_map = _deontic_cue_type_pattern()
94
+ m = pattern.search(text)
95
+ return cue_map[m.group(0).lower()] if m else None
96
+
97
+
98
+ @lru_cache(maxsize=4)
99
+ def load_actor_synonyms(path: str = str(_COMPLIANCE_TTL_PATH)) -> dict[str, str]:
100
+ """ADR-0066 P4a: the actor-role synonyms from compliance_bridge.ttl -- `{synonym -> canonical role}` built from
101
+ each `cmp:ActorRole`'s `skos:altLabel` (synonym) -> `skos:prefLabel` (canonical). The query-side actor gate
102
+ (`canonical_actor`) normalizes with this. Cached per path."""
103
+ from rdflib import URIRef
104
+
105
+ g = Graph()
106
+ g.parse(path, format="turtle")
107
+ out: dict[str, str] = {}
108
+ for role in g.subjects(RDF.type, URIRef(_CMP + "ActorRole")):
109
+ pref = str(g.value(role, SKOS.prefLabel) or "").strip().lower()
110
+ if not pref:
111
+ continue
112
+ for alt in g.objects(role, SKOS.altLabel):
113
+ out[str(alt).strip().lower()] = pref
114
+ return out
115
+
116
+
117
+ @lru_cache(maxsize=4)
118
+ def load_claim_type_criteria(path: str = str(_COMPLIANCE_TTL_PATH)) -> dict[str, str]:
119
+ """ADR-0119: `{ClaimType local-name -> cmp:decisionCriterion}` -- the one-line criterion each claim type uses
120
+ as its typed-decision (noul) question. Authored in compliance_bridge.ttl, not in capability code (ADR-0066)."""
121
+ from rdflib import URIRef
122
+
123
+ g = Graph()
124
+ g.parse(path, format="turtle")
125
+ crit = URIRef(_CMP + "decisionCriterion")
126
+ out: dict[str, str] = {}
127
+ for m in g.subjects(RDF.type, URIRef(_CMP + "ClaimType")):
128
+ c = g.value(m, crit)
129
+ if c is not None:
130
+ out[str(m).rsplit("#", 1)[-1]] = str(c)
131
+ return out
132
+
133
+
134
+ @lru_cache(maxsize=4)
135
+ def load_actor_role_criteria(path: str = str(_COMPLIANCE_TTL_PATH)) -> dict[str, str]:
136
+ """ADR-0119: `{ActorRole prefLabel -> cmp:decisionCriterion}` -- the actor `choice` options + their criteria,
137
+ from the ttl (includes the workplace-domain `employer`). Authored in the ontology, not capability code."""
138
+ from rdflib import URIRef
139
+
140
+ g = Graph()
141
+ g.parse(path, format="turtle")
142
+ crit = URIRef(_CMP + "decisionCriterion")
143
+ out: dict[str, str] = {}
144
+ for role in g.subjects(RDF.type, URIRef(_CMP + "ActorRole")):
145
+ c = g.value(role, crit)
146
+ label = str(g.value(role, SKOS.prefLabel) or str(role).rsplit("#", 1)[-1]).strip().lower()
147
+ if c is not None and label:
148
+ out[label] = str(c)
149
+ return out
150
+
151
+
152
+ @lru_cache(maxsize=4)
153
+ def load_operative_rubric(path: str = str(_COMPLIANCE_TTL_PATH)) -> dict[str, str]:
154
+ """ADR-0119: the operative-rule binary gate's `{instructions, true, false}` from `cmp:operativeRuleQuestion`
155
+ in compliance_bridge.ttl -- the decision knowledge the `jev_decision` gate asks, authored in the ontology."""
156
+ from rdflib import URIRef
157
+
158
+ g = Graph()
159
+ g.parse(path, format="turtle")
160
+ q = URIRef(_CMP + "operativeRuleQuestion")
161
+
162
+ def _v(prop: str) -> str:
163
+ v = g.value(q, URIRef(_CMP + prop))
164
+ return str(v) if v is not None else ""
165
+
166
+ return {"instructions": _v("questionInstructions"), "true": _v("criterionTrue"), "false": _v("criterionFalse")}
167
+
168
+
169
+ @lru_cache(maxsize=4)
170
+ def load_role_domains(path: str = str(_COMPLIANCE_TTL_PATH)) -> dict[str, str]:
171
+ """ADR-0068 (engine issue 0013): the DISJOINTNESS knowledge for the actor gate -- `{canonical role -> domain}`
172
+ built from each `cmp:ActorRole`'s `skos:prefLabel` (canonical) -> `cmp:roleDomain`. Two roles are disjoint iff
173
+ both appear here with DIFFERENT domains; the recall-first gate excludes only disjoint pairs (a role absent
174
+ here, or two roles in the same domain, are compatible). A customer domain declares its roles' roleDomain in
175
+ its own pack to get cross-domain narrowing. Cached per path."""
176
+ from rdflib import URIRef
177
+
178
+ g = Graph()
179
+ g.parse(path, format="turtle")
180
+ out: dict[str, str] = {}
181
+ for role in g.subjects(RDF.type, URIRef(_CMP + "ActorRole")):
182
+ pref = str(g.value(role, SKOS.prefLabel) or "").strip().lower()
183
+ domain = str(g.value(role, URIRef(_CMP + "roleDomain")) or "").strip().lower()
184
+ if pref and domain:
185
+ out[pref] = domain
186
+ return out
187
+
188
+
189
+ _FTC_PACK_PATH = Path(__file__).parent / "packs" / "ftc_16cfr255.ttl"
190
+
191
+
192
+ @lru_cache(maxsize=4)
193
+ def load_section_overrides(path: str = str(_FTC_PACK_PATH)) -> tuple[dict[str, str], dict[str, frozenset[str]]]:
194
+ """ADR-0066 P4b: a domain pack's per-section overrides from `cmp:SectionOverride` instances. Returns
195
+ `(rule_scope, claim_types)`: `{section -> 'content'|'context'}` (only sections that pin a scope) and
196
+ `{section -> {claim type value}}` (all claim types when `cmp:appliesToAllClaimTypes` is true, else the explicit
197
+ `cmp:appliesToClaimType` set -- empty for a definitions section). Cached per path."""
198
+ from rdflib import URIRef
199
+
200
+ g = Graph()
201
+ g.parse(path, format="turtle")
202
+ all_claim_types = frozenset(load_compliance_vocab().get("ClaimType", set()))
203
+ rule_scope: dict[str, str] = {}
204
+ claim_types: dict[str, frozenset[str]] = {}
205
+ for so in g.subjects(RDF.type, URIRef(_CMP + "SectionOverride")):
206
+ section = str(g.value(so, URIRef(_CMP + "section")) or "").strip()
207
+ if not section:
208
+ continue
209
+ all_flag = g.value(so, URIRef(_CMP + "appliesToAllClaimTypes"))
210
+ if all_flag is not None and bool(all_flag.toPython()):
211
+ claim_types[section] = all_claim_types
212
+ else:
213
+ claim_types[section] = frozenset(
214
+ str(ct).rsplit("#", 1)[-1] for ct in g.objects(so, URIRef(_CMP + "appliesToClaimType")))
215
+ rs = g.value(so, URIRef(_CMP + "overrideRuleScope"))
216
+ if rs is not None:
217
+ rule_scope[section] = str(rs).rsplit("#", 1)[-1]
218
+ return rule_scope, claim_types
219
+
220
+
221
+ @lru_cache(maxsize=4)
222
+ def load_shapes_graph(path: str = str(_TTL_PATH)) -> Graph:
223
+ """ADR-0066 P2: the ttl parsed as an rdflib Graph -- its `sh:NodeShape`s ARE the SHACL shapes handed to pyshacl
224
+ at runtime, so the symbolic layer reads the symbolic artifact directly (no Python-built shapes). pyshacl uses
225
+ the shapes and ignores the ttl's non-SHACL triples. Cached per path."""
226
+ g = Graph()
227
+ g.parse(path, format="turtle")
228
+ return g
229
+ _CBR = "https://ragwright.local/ontology/contract-bridge#"
230
+ # The engine's pack-schema DECLARATION meta-vocabulary (DD-8): the classes/predicates a pack `.ttl` uses to declare
231
+ # its KG node/edge types + entity taxonomy (KgVertexType/vertexName/kgProperty/uniqueIndexOn/KgStructuralEdge/
232
+ # edgeName, EntityNodeType/EntityRelationshipType). Engine-namespaced (not contract-namespaced), so a NON-contract
233
+ # domain declares its schema in this shared language without borrowing the contract pack's namespace (AC-journey).
234
+ _ENG = "https://ragwright.local/ontology/engine#"
235
+ _DIMENSION_CLASS = _CBR + "PropertyDimension"
236
+
237
+
238
+ @dataclass(frozen=True)
239
+ class ContractOntologyView:
240
+ """The contract domain knowledge parsed out of `contract_bridge.ttl` (all string-keyed by label)."""
241
+
242
+ closed_vocab: dict[str, set[str]] = field(default_factory=dict)
243
+ multivalued: set[str] = field(default_factory=set)
244
+ function_applicable_dims: dict[str, set[str]] = field(default_factory=dict)
245
+ permission_polarity: dict[str, set[str]] = field(default_factory=dict)
246
+ restrictive_functions: set[str] = field(default_factory=set)
247
+ value_rollup: dict[str, dict[str, set[str]]] = field(default_factory=dict)
248
+ # issue 0037: ingest synonyms -- {dimension: {normalized surface term: canonical member}}. A skos:broader edge
249
+ # whose BROADER is a closed-vocab member but whose NARROWER is not (a specific surface term). Used at ingest to
250
+ # canonicalize an out-of-vocab extracted value onto its canonical member (else the value is kept verbatim).
251
+ value_synonyms: dict[str, dict[str, str]] = field(default_factory=dict)
252
+ # DD-7 (ADR-0066): the entity-graph taxonomy -- entity node types (cbr:EntityNodeType) + entity-to-entity
253
+ # relationship types (cbr:EntityRelationshipType), by label. Was the Python EntityType/RelationshipType enum.
254
+ entity_types: set[str] = field(default_factory=set)
255
+ relationship_types: set[str] = field(default_factory=set)
256
+
257
+
258
+ def _label(g: Graph, node) -> str:
259
+ lbl = g.value(node, RDFS.label)
260
+ return str(lbl) if lbl is not None else ""
261
+
262
+
263
+ def load_contract_ontology(path: Path | str = _TTL_PATH) -> ContractOntologyView:
264
+ """Parse the contract bridge ontology into a `ContractOntologyView`."""
265
+ g = Graph()
266
+ g.parse(str(path), format="turtle")
267
+
268
+ closed_vocab: dict[str, set[str]] = {}
269
+ multivalued: set[str] = set()
270
+ dim_label_by_node: dict[str, str] = {} # dimension IRI -> label (for the applicability shapes)
271
+ value_label_by_node: dict[str, str] = {} # value IRI -> label (for rollups + oneOf)
272
+ value_dim_by_node: dict[str, str] = {} # value IRI -> its dimension label (for rollups)
273
+
274
+ for dim in g.subjects(RDF.type, _dim_class()):
275
+ label = _label(g, dim)
276
+ dim_label_by_node[str(dim)] = label
277
+ if str(g.value(dim, _cbr("cardinality")) or "") == "list":
278
+ multivalued.add(label)
279
+ one_of = g.value(dim, OWL.oneOf)
280
+ if one_of is not None: # a CLOSED dimension (open-valued dims omit owl:oneOf)
281
+ members = list(Collection(g, one_of))
282
+ values = set()
283
+ for m in members:
284
+ vlabel = _label(g, m)
285
+ values.add(vlabel)
286
+ value_label_by_node[str(m)] = vlabel
287
+ value_dim_by_node[str(m)] = label
288
+ closed_vocab[label] = values
289
+
290
+ # Applicability + deontic: one sh:NodeShape per clause function.
291
+ function_applicable_dims: dict[str, set[str]] = {}
292
+ permission_polarity: dict[str, set[str]] = {}
293
+ restrictive_functions: set[str] = set()
294
+ for shape in g.subjects(RDF.type, SH.NodeShape):
295
+ target = g.value(shape, SH.targetClass)
296
+ fn = _label(g, target)
297
+ if not fn:
298
+ continue
299
+ dims: set[str] = set()
300
+ for prop in g.objects(shape, SH.property):
301
+ path = g.value(prop, SH.path)
302
+ dlabel = dim_label_by_node.get(str(path), _label(g, path))
303
+ dims.add(dlabel)
304
+ in_list = g.value(prop, SH["in"])
305
+ if in_list is not None: # deontic: restrictive function forbids the permission-polarity values
306
+ restrictive_functions.add(fn)
307
+ allowed = {str(x) for x in Collection(g, in_list)}
308
+ forbidden = closed_vocab.get(dlabel, set()) - allowed
309
+ if forbidden:
310
+ permission_polarity.setdefault(dlabel, set()).update(forbidden)
311
+ function_applicable_dims[fn] = dims
312
+
313
+ # Value rollups: skos:broader between value individuals.
314
+ value_rollup: dict[str, dict[str, set[str]]] = {}
315
+ # issue 0037: ingest synonyms -- {dim: {normalized surface: canonical member}} from a skos:broader edge whose
316
+ # BROADER is a closed member but whose NARROWER is not (a specific surface term); surface = narrower local-name
317
+ # + its skos:altLabels. (A narrower that IS a member is a query-side rollup, handled above.)
318
+ value_synonyms: dict[str, dict[str, str]] = {}
319
+ for narrower, broader in g.subject_objects(SKOS.broader):
320
+ dim = value_dim_by_node.get(str(narrower))
321
+ if dim is not None: # narrower is itself a vocab member -> a query-side value rollup
322
+ value_rollup.setdefault(dim, {}).setdefault(
323
+ value_label_by_node.get(str(narrower), ""), set()).add(
324
+ value_label_by_node.get(str(broader), ""))
325
+ continue
326
+ bdim = value_dim_by_node.get(str(broader)) # narrower is a surface synonym -> map to the broader member
327
+ bval = value_label_by_node.get(str(broader))
328
+ if not bdim or not bval:
329
+ continue
330
+ surfaces = {_label(g, narrower) or str(narrower).rsplit("#", 1)[-1]}
331
+ surfaces |= {str(a) for a in g.objects(narrower, SKOS.altLabel)}
332
+ for s in surfaces:
333
+ key = re.sub(r"[^A-Za-z0-9]+", "", s).lower()
334
+ if key:
335
+ value_synonyms.setdefault(bdim, {})[key] = bval
336
+
337
+ # DD-7: the entity-graph taxonomy (entity node types + entity-to-entity relationship types), by label.
338
+ entity_types = {_label(g, s) for s in g.subjects(RDF.type, _eng("EntityNodeType"))}
339
+ relationship_types = {_label(g, s) for s in g.subjects(RDF.type, _eng("EntityRelationshipType"))}
340
+
341
+ return ContractOntologyView(
342
+ closed_vocab=closed_vocab, multivalued=multivalued,
343
+ function_applicable_dims=function_applicable_dims,
344
+ permission_polarity=permission_polarity, restrictive_functions=restrictive_functions,
345
+ value_synonyms=value_synonyms,
346
+ value_rollup=value_rollup,
347
+ entity_types=entity_types, relationship_types=relationship_types)
348
+
349
+
350
+ @dataclass(frozen=True)
351
+ class KgVertexType:
352
+ """ADR-0067 P5b: a domain KG vertex-type declaration the store creates -- name, its `(property, SQL type)`
353
+ pairs, and the property to build a UNIQUE index on (if any)."""
354
+
355
+ name: str
356
+ properties: tuple[tuple[str, str], ...]
357
+ unique_index: str | None
358
+
359
+
360
+ @lru_cache(maxsize=8)
361
+ def load_kg_schema(path: str | None = None) -> tuple[tuple[KgVertexType, ...], frozenset[str]]:
362
+ """ADR-0067 P5b: the DOMAIN KG node/edge storage schema from the ttl -- `(vertex types, structural edge names)`.
363
+ The engine infra (Chunk/Span/Entity) stays generic in store code; these domain types are pack-declared. Cached.
364
+ `path=None` is the engine's reference CONTRACT pack; a new domain passes its OWN pack `.ttl` (AC-journey)."""
365
+ g = Graph()
366
+ g.parse(str(path or _TTL_PATH), format="turtle")
367
+ vertices = []
368
+ for v in g.subjects(RDF.type, _eng("KgVertexType")):
369
+ props = tuple(sorted((str(p).split(":", 1)[0], str(p).split(":", 1)[1])
370
+ for p in g.objects(v, _eng("kgProperty")) if ":" in str(p)))
371
+ ui = g.value(v, _eng("uniqueIndexOn"))
372
+ vertices.append(KgVertexType(name=str(g.value(v, _eng("vertexName"))), properties=props,
373
+ unique_index=(str(ui) if ui is not None else None)))
374
+ vertices.sort(key=lambda x: x.name)
375
+ edges = frozenset(str(g.value(e, _eng("edgeName")))
376
+ for e in g.subjects(RDF.type, _eng("KgStructuralEdge")))
377
+ return tuple(vertices), edges
378
+
379
+
380
+ @lru_cache(maxsize=4)
381
+ def load_typed_edges(path: str = str(_TTL_PATH)) -> tuple[dict[str, str], dict[str, str]]:
382
+ """ADR-0067 P5a: the KG typed-edge map from contract_bridge.ttl. Returns `(dim_edge, edge_iri)`:
383
+ `{dimension value -> KG edge type}` (from `cbr:kgEdge`) and `{edge type -> predicate IRI}` (from
384
+ `cbr:predicateIri` on each `cbr:KgEdgeType`). The property-graph edge types are ontology-authoritative;
385
+ `store/arcadedb.py` builds `_TYPED_DIMENSION_EDGE` / `_edge_predicate_iri` from this. Cached per path."""
386
+ g = Graph()
387
+ g.parse(str(path), format="turtle")
388
+ edge_iri = {str(g.value(e, RDFS.label)): str(g.value(e, _cbr("predicateIri")))
389
+ for e in g.subjects(RDF.type, _cbr("KgEdgeType"))}
390
+ dim_edge = {str(g.value(dim, RDFS.label)): str(g.value(edge, RDFS.label))
391
+ for dim, edge in g.subject_objects(_cbr("kgEdge"))}
392
+ return dim_edge, edge_iri
393
+
394
+
395
+ def _dim_class():
396
+ from rdflib import URIRef
397
+ return URIRef(_DIMENSION_CLASS)
398
+
399
+
400
+ def _cbr(frag: str):
401
+ from rdflib import URIRef
402
+ return URIRef(_CBR + frag)
403
+
404
+
405
+ def _eng(frag: str):
406
+ from rdflib import URIRef
407
+ return URIRef(_ENG + frag)
408
+
409
+
410
+ def load_template_fields(path: Path | str = _TTL_PATH):
411
+ """ADR-0066 P1b-1: the extraction template's fields as captured in the ttl, as `TemplateFieldSpec`s in the
412
+ same order the template declares them (`cbr:fieldOrder`). Round-trips `bootstrap_template_capture` -- the drift
413
+ test asserts this equals a fresh introspection of `clause_template.py`."""
414
+ from rag_wright.ontology.template_introspect import TemplateFieldSpec
415
+
416
+ g = Graph()
417
+ g.parse(str(path), format="turtle")
418
+ specs: list = []
419
+ for node in g.subjects(RDF.type, _cbr("TemplateField")):
420
+ order = g.value(node, _cbr("fieldOrder"))
421
+ ml = g.value(node, _cbr("maxLength"))
422
+ specs.append((int(order), TemplateFieldSpec(
423
+ model=str(g.value(node, _cbr("onModel"))),
424
+ name=str(g.value(node, RDFS.label)),
425
+ kind=str(g.value(node, _cbr("fieldKind"))),
426
+ default_token=str(g.value(node, _cbr("default"))),
427
+ definition=str(g.value(node, SKOS.definition) or ""),
428
+ enum_class=(str(v) if (v := g.value(node, _cbr("enumClass"))) is not None else None),
429
+ model_ref=(str(v) if (v := g.value(node, _cbr("modelRef"))) is not None else None),
430
+ edge_label=(str(v) if (v := g.value(node, _cbr("edgeLabel"))) is not None else None),
431
+ max_length=(int(ml) if ml is not None else None),
432
+ examples=(tuple(str(e) for e in Collection(g, exlist))
433
+ if (exlist := g.value(node, _cbr("examples"))) is not None else ()),
434
+ )))
435
+ return [spec for _, spec in sorted(specs, key=lambda t: t[0])]
@@ -0,0 +1,29 @@
1
+ # ftc_16cfr255.ttl -- the FTC 16 CFR Part 255 (Endorsement Guides) DOMAIN PACK (ADR-0066 P4b).
2
+ #
3
+ # The REFERENCE domain pack: the curated per-section overrides for the FTC endorsement guides, as
4
+ # cmp:SectionOverride instances (the shape is declared in compliance_bridge.ttl). Loaded by the query side to
5
+ # override DEON-1 rule scope + DEON-8 applicable claim types for FTC-cited requirements. A customer domain ships
6
+ # its OWN pack; nothing FTC-specific is hardcoded in engine code.
7
+ #
8
+ # KEY FINDING (CC-6): the FTC endorsement guides apply by CONTEXT (is the ad an endorsement?), NOT by claim_type,
9
+ # so every operative section applies to ALL claim types; only the definitions section (255.0) applies to none.
10
+ # 255.5 (material-connection disclosure) + 255.4 (organization endorsements) are CONTEXT rules (always included).
11
+
12
+ @prefix ftc: <https://ragwright.local/ontology/packs/ftc-16cfr255#> .
13
+ @prefix cmp: <https://ragwright.local/ontology/compliance-bridge#> .
14
+ @prefix owl: <http://www.w3.org/2002/07/owl#> .
15
+ @prefix rdfs: <http://www.w3.org/2000/01/rdf-schema#> .
16
+
17
+ <https://ragwright.local/ontology/packs/ftc-16cfr255> a owl:Ontology ;
18
+ rdfs:label "FTC 16 CFR 255 domain pack" ;
19
+ rdfs:comment "Curated per-section overrides for the FTC Endorsement Guides (the engine's reference domain pack)." .
20
+
21
+ ftc:s255_0 a cmp:SectionOverride ; cmp:section "255.0" ; cmp:appliesToAllClaimTypes false . # definitions -> none
22
+ ftc:s255_1 a cmp:SectionOverride ; cmp:section "255.1" ; cmp:appliesToAllClaimTypes true . # general considerations
23
+ ftc:s255_2 a cmp:SectionOverride ; cmp:section "255.2" ; cmp:appliesToAllClaimTypes true . # consumer endorsements
24
+ ftc:s255_3 a cmp:SectionOverride ; cmp:section "255.3" ; cmp:appliesToAllClaimTypes true . # expert endorsements
25
+ ftc:s255_4 a cmp:SectionOverride ; cmp:section "255.4" ; cmp:appliesToAllClaimTypes true ; # organization endorsements
26
+ cmp:overrideRuleScope cmp:context .
27
+ ftc:s255_5 a cmp:SectionOverride ; cmp:section "255.5" ; cmp:appliesToAllClaimTypes true ; # material-connection disclosure
28
+ cmp:overrideRuleScope cmp:context .
29
+ ftc:s255_6 a cmp:SectionOverride ; cmp:section "255.6" ; cmp:appliesToAllClaimTypes true . # endorsements to children
@@ -0,0 +1,87 @@
1
+ """The entity registry (T8, FR-C.8 / FR-C.7) -- a DOMAIN-NEUTRAL closed-world registry of canonical entities.
2
+
3
+ ADR-0067: the registry is generic. A domain's canonical `entity_id`s and their surface normalization are the
4
+ domain's concern: the surface-form normalizer is INJECTABLE (`EntityRegistry(normalize=...)`, default = a generic
5
+ name key), and the domain's own builder constructs the registry (the reference builder lives in the corpus
6
+ layer, keyed by that corpus's canonical ids). This module imports nothing corpus-specific.
7
+
8
+ Lookup is **closed-world**: `resolve` returns `None` for an unknown surface form, never a fabricated id. The
9
+ concrete matching strategy (fuzzy / embedding / language-model-assisted, SPEC section 16.3) is T24's; this
10
+ registry is the closed set T24 resolves against, plus an exact normalized-surface-form index.
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ import re
16
+ from typing import Callable, Optional, Protocol, runtime_checkable
17
+
18
+ from pydantic import BaseModel
19
+
20
+ from rag_wright.contracts.identifiers import EntityId
21
+
22
+ _NON_ALNUM = re.compile(r"[^a-z0-9]+")
23
+
24
+
25
+ def default_surface_key(name: str) -> str:
26
+ """The default DOMAIN-NEUTRAL surface-form key: lowercase, non-alphanumeric runs folded to a single space."""
27
+ return _NON_ALNUM.sub(" ", (name or "").lower()).strip()
28
+
29
+
30
+ @runtime_checkable
31
+ class EntityResolver(Protocol):
32
+ """The entity-resolution seam (DD-3, ADR-0067 P5c): map a surface form to a canonical `entity_id`, or
33
+ `None` (closed-world -- never a fabricated id). The RESOLUTION STRATEGY is the domain's concern and is
34
+ injected into `resolve_entities` (FR-C.7): the generic default is the exact-normalized surface-form
35
+ `EntityRegistry` (below); a domain pack injects its own registry built by that corpus's loader in the
36
+ corpus layer (keyed by that domain's canonical ids). A product may bind any strategy (fuzzy / embedding /
37
+ an external service) as long as it honors this signature and the closed-world contract. The generic
38
+ resolution invariants (cluster surface-form sweep, two-channel dedup, self-loop dropping) stay in the
39
+ capability, not the resolver. This module names no specific domain (the SEC-free scope guard enforces it)."""
40
+
41
+ def resolve(self, surface_form: str) -> Optional[EntityId]:
42
+ ...
43
+
44
+
45
+ class RegistryRecord(BaseModel):
46
+ """One registered entity: its canonical id, conformed name, ticker, and known aliases."""
47
+
48
+ entity_id: EntityId
49
+ canonical_name: str
50
+ ticker: Optional[str] = None
51
+ aliases: list[str] = []
52
+
53
+
54
+ class EntityRegistry:
55
+ """A closed-world registry of canonical entities, indexed by normalized surface form."""
56
+
57
+ def __init__(self, *, normalize: Optional[Callable[[str], str]] = None) -> None:
58
+ self._by_id: dict[str, RegistryRecord] = {}
59
+ self._index: dict[str, EntityId] = {} # normalized surface form -> entity_id
60
+ self.skipped_ids: list[str] = [] # raw canonical-id values that failed normalization (builder-specific)
61
+ self._normalize = normalize or default_surface_key # ADR-0067: domain-neutral, injectable
62
+
63
+ def _index_surface(self, surface: str, entity_id: EntityId) -> None:
64
+ key = self._normalize(surface)
65
+ if key:
66
+ self._index.setdefault(key, entity_id)
67
+
68
+ def add(self, record: RegistryRecord) -> None:
69
+ self._by_id[record.entity_id.value] = record
70
+ self._index_surface(record.canonical_name, record.entity_id)
71
+ if record.ticker:
72
+ self._index_surface(record.ticker, record.entity_id)
73
+ for alias in record.aliases:
74
+ self._index_surface(alias, record.entity_id)
75
+
76
+ def get(self, entity_id: EntityId) -> Optional[RegistryRecord]:
77
+ return self._by_id.get(entity_id.value)
78
+
79
+ def resolve(self, surface_form: str) -> Optional[EntityId]:
80
+ """The canonical `entity_id` for a known surface form, or `None` (closed-world)."""
81
+ return self._index.get(self._normalize(surface_form))
82
+
83
+ def __len__(self) -> int:
84
+ return len(self._by_id)
85
+
86
+ def __contains__(self, entity_id: EntityId) -> bool:
87
+ return entity_id.value in self._by_id