rag-wright 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- rag_wright/__init__.py +13 -0
- rag_wright/api/__init__.py +33 -0
- rag_wright/api/config.py +59 -0
- rag_wright/api/discover.py +70 -0
- rag_wright/api/documents.py +39 -0
- rag_wright/api/ids.py +31 -0
- rag_wright/api/invoke.py +99 -0
- rag_wright/api/kg.py +61 -0
- rag_wright/api/mcp.py +94 -0
- rag_wright/api/usage.py +30 -0
- rag_wright/api/workspace.py +85 -0
- rag_wright/capabilities/__init__.py +8 -0
- rag_wright/capabilities/answer_generator.py +427 -0
- rag_wright/capabilities/ard.py +286 -0
- rag_wright/capabilities/assertion_extraction.py +79 -0
- rag_wright/capabilities/chunk_read.py +58 -0
- rag_wright/capabilities/chunk_write.py +163 -0
- rag_wright/capabilities/claim_extraction.py +153 -0
- rag_wright/capabilities/clause_exception_linking.py +117 -0
- rag_wright/capabilities/compliance_judgment.py +322 -0
- rag_wright/capabilities/compliance_store.py +87 -0
- rag_wright/capabilities/contract_kg_serve.py +156 -0
- rag_wright/capabilities/contract_kg_store.py +251 -0
- rag_wright/capabilities/dg_extraction.py +585 -0
- rag_wright/capabilities/disambiguation.py +163 -0
- rag_wright/capabilities/document_parse.py +87 -0
- rag_wright/capabilities/document_scope.py +49 -0
- rag_wright/capabilities/embedding.py +164 -0
- rag_wright/capabilities/embedding_profiles.py +43 -0
- rag_wright/capabilities/entity_resolution.py +154 -0
- rag_wright/capabilities/fusion.py +64 -0
- rag_wright/capabilities/graph_extraction.py +243 -0
- rag_wright/capabilities/graph_query.py +73 -0
- rag_wright/capabilities/graph_storage.py +111 -0
- rag_wright/capabilities/highlight_serve.py +142 -0
- rag_wright/capabilities/hybrid_search.py +65 -0
- rag_wright/capabilities/invoke.py +31 -0
- rag_wright/capabilities/jev_decision.py +38 -0
- rag_wright/capabilities/manifests.py +872 -0
- rag_wright/capabilities/okf_navigate.py +456 -0
- rag_wright/capabilities/parsing.py +286 -0
- rag_wright/capabilities/property_boosted_retrieval.py +125 -0
- rag_wright/capabilities/query_function_classifier.py +94 -0
- rag_wright/capabilities/query_understanding.py +109 -0
- rag_wright/capabilities/registry.py +262 -0
- rag_wright/capabilities/remote_encoders.py +94 -0
- rag_wright/capabilities/requirement_extraction.py +247 -0
- rag_wright/capabilities/reranking.py +123 -0
- rag_wright/capabilities/retrieval_core.py +126 -0
- rag_wright/capabilities/rlm_chunking.py +808 -0
- rag_wright/capabilities/rlm_synthesis.py +316 -0
- rag_wright/capabilities/scan_quality.py +136 -0
- rag_wright/capabilities/span_relevance_judgment.py +191 -0
- rag_wright/capabilities/vision_to_text.py +85 -0
- rag_wright/capabilities/vlm_ocr.py +85 -0
- rag_wright/contracts/__init__.py +6 -0
- rag_wright/contracts/chunk.py +79 -0
- rag_wright/contracts/compliance.py +303 -0
- rag_wright/contracts/contract_meta.py +27 -0
- rag_wright/contracts/extraction.py +130 -0
- rag_wright/contracts/function.py +167 -0
- rag_wright/contracts/function_routing.py +91 -0
- rag_wright/contracts/highlight.py +74 -0
- rag_wright/contracts/identifiers.py +153 -0
- rag_wright/contracts/jurisdiction.py +96 -0
- rag_wright/contracts/ontology.py +142 -0
- rag_wright/contracts/property.py +201 -0
- rag_wright/contracts/provenance.py +78 -0
- rag_wright/contracts/query_intent.py +53 -0
- rag_wright/contracts/span.py +76 -0
- rag_wright/contracts/value_match.py +84 -0
- rag_wright/corpus/__init__.py +0 -0
- rag_wright/corpus/canonicalize.py +116 -0
- rag_wright/corpus/cuad.py +153 -0
- rag_wright/corpus/cuad_ingestion.py +72 -0
- rag_wright/corpus/document_parser.py +299 -0
- rag_wright/corpus/edgar.py +231 -0
- rag_wright/corpus/gcs_ingestion.py +120 -0
- rag_wright/corpus/http.py +110 -0
- rag_wright/corpus/selection.py +152 -0
- rag_wright/mcp/__init__.py +11 -0
- rag_wright/mcp/compliance_server.py +299 -0
- rag_wright/mcp/intra_document_qa_server.py +170 -0
- rag_wright/mcp/relational_qa_server.py +171 -0
- rag_wright/mcp/session_store.py +64 -0
- rag_wright/mcp/typed_property_retrieval_server.py +191 -0
- rag_wright/models/__init__.py +8 -0
- rag_wright/models/profiles.py +331 -0
- rag_wright/models/seam.py +497 -0
- rag_wright/models/tag_structured.py +285 -0
- rag_wright/models/tracing.py +179 -0
- rag_wright/models/usage.py +102 -0
- rag_wright/okf/__init__.py +11 -0
- rag_wright/okf/compile.py +292 -0
- rag_wright/okf/document.py +47 -0
- rag_wright/okf/enrich.py +176 -0
- rag_wright/okf/links.py +190 -0
- rag_wright/okf/lint.py +105 -0
- rag_wright/ontology/__init__.py +6 -0
- rag_wright/ontology/_generated_template_meta.py +60 -0
- rag_wright/ontology/_generated_vocab.py +52 -0
- rag_wright/ontology/clause_template.py +964 -0
- rag_wright/ontology/codegen.py +84 -0
- rag_wright/ontology/compliance_bridge.ttl +186 -0
- rag_wright/ontology/contract_bridge.ttl +2685 -0
- rag_wright/ontology/contract_taxonomy.py +24 -0
- rag_wright/ontology/derive.py +58 -0
- rag_wright/ontology/loader.py +435 -0
- rag_wright/ontology/packs/ftc_16cfr255.ttl +29 -0
- rag_wright/ontology/registry.py +87 -0
- rag_wright/ontology/template_introspect.py +100 -0
- rag_wright/py.typed +0 -0
- rag_wright/reference/__init__.py +2 -0
- rag_wright/reference/compliance.py +41 -0
- rag_wright/reference/contract_seam.py +123 -0
- rag_wright/skills/__init__.py +7 -0
- rag_wright/skills/claim_extraction/SKILL.md +47 -0
- rag_wright/skills/claim_extraction/__init__.py +1 -0
- rag_wright/skills/claim_extraction/template.py +50 -0
- rag_wright/skills/compliance_judgment/SKILL.md +59 -0
- rag_wright/skills/corpus_ingest/SKILL.md +106 -0
- rag_wright/skills/extraction_semantic_judge/SKILL.md +51 -0
- rag_wright/skills/extraction_semantic_judge/__init__.py +1 -0
- rag_wright/skills/generation/SKILL.md +64 -0
- rag_wright/skills/generation/__init__.py +1 -0
- rag_wright/skills/generic_compliance_judgment/SKILL.md +58 -0
- rag_wright/skills/okf_navigate/SKILL.md +137 -0
- rag_wright/skills/requirement_extraction/SKILL.md +47 -0
- rag_wright/skills/requirement_extraction/__init__.py +1 -0
- rag_wright/skills/requirement_extraction/template.py +50 -0
- rag_wright/skills/rlm/SKILL.md +186 -0
- rag_wright/skills/rlm/__init__.py +31 -0
- rag_wright/skills/rlm/agent.py +292 -0
- rag_wright/skills/span_relevance_judgment/SKILL.md +67 -0
- rag_wright/skills/vision_to_text/SKILL.md +36 -0
- rag_wright/skills/vision_to_text/__init__.py +1 -0
- rag_wright/spans/__init__.py +1 -0
- rag_wright/spans/boundary.py +78 -0
- rag_wright/spans/clause_function_classifier.py +490 -0
- rag_wright/spans/clause_kg_extractor.py +337 -0
- rag_wright/spans/cuad_labels.py +81 -0
- rag_wright/spans/dim_classifier.py +158 -0
- rag_wright/spans/dim_fleet.json +411 -0
- rag_wright/spans/function_classifier.py +77 -0
- rag_wright/spans/function_families.py +62 -0
- rag_wright/spans/hybrid_classifier.py +103 -0
- rag_wright/spans/legalbert_classifier.py +83 -0
- rag_wright/spans/model_capabilities.py +107 -0
- rag_wright/spans/new_function_labels.py +111 -0
- rag_wright/spans/page_map.py +68 -0
- rag_wright/spans/property_extractor.py +365 -0
- rag_wright/spans/property_grounding.py +182 -0
- rag_wright/spans/reclassify.py +77 -0
- rag_wright/spans/scarce_function_labels.py +105 -0
- rag_wright/spans/segment.py +341 -0
- rag_wright/spans/semantic_judge.py +197 -0
- rag_wright/spans/symbolic_validation.py +131 -0
- rag_wright/spans/tag_clause_extractor.py +182 -0
- rag_wright/store/__init__.py +6 -0
- rag_wright/store/arcadedb.py +1135 -0
- rag_wright/store/chunk_text.py +66 -0
- rag_wright/store/seam.py +213 -0
- rag_wright/subgraphs/__init__.py +0 -0
- rag_wright/subgraphs/async_ingestion.py +204 -0
- rag_wright/subgraphs/compliance_check.py +1042 -0
- rag_wright/subgraphs/compliance_ingestion.py +306 -0
- rag_wright/subgraphs/contract_ingestion_pipeline.py +999 -0
- rag_wright/subgraphs/graph_extraction.py +102 -0
- rag_wright/subgraphs/intra_document_qa.py +328 -0
- rag_wright/subgraphs/observability.py +140 -0
- rag_wright/subgraphs/query_constraint_extraction.py +73 -0
- rag_wright/subgraphs/relational_qa.py +165 -0
- rag_wright/subgraphs/requirement_extraction.py +137 -0
- rag_wright/subgraphs/scaffold.py +65 -0
- rag_wright/subgraphs/semantic_chunking.py +183 -0
- rag_wright/subgraphs/typed_clause_extraction.py +172 -0
- rag_wright/subgraphs/typed_property_retrieval.py +278 -0
- rag_wright/util/__init__.py +1 -0
- rag_wright/util/concurrent.py +153 -0
- rag_wright/util/spacy_model.py +45 -0
- rag_wright-0.1.0.dist-info/METADATA +168 -0
- rag_wright-0.1.0.dist-info/RECORD +184 -0
- rag_wright-0.1.0.dist-info/WHEEL +4 -0
- rag_wright-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
"""The reference CONTRACT domain pack's entity-node + entity-relationship taxonomy (DD-5/DD-7, ADR-0066/0117).
|
|
2
|
+
|
|
3
|
+
De-domaining (DD-5): the engine's generic primitives + contracts are taxonomy-free -- `entity_type` and
|
|
4
|
+
`relationship_type` are plain strings the CALLER (a domain graph) names. The closed value sets below are DOMAIN
|
|
5
|
+
knowledge.
|
|
6
|
+
|
|
7
|
+
ADR-0066 end-state (DD-7): those values are now declared in `contract_bridge.ttl` (the single source of truth) and
|
|
8
|
+
rendered into `_generated_vocab.py` by codegen, CI-diff-enforced (`tests/ontology/test_generated_vocab_in_sync.py`).
|
|
9
|
+
This module is the stable import surface the domain builders use -- it RE-EXPORTS the generated constants, so no
|
|
10
|
+
code hardcodes the values. To change the taxonomy, edit the ttl and regenerate; never edit the generated file or
|
|
11
|
+
re-add literals here. A new domain declares its own entity/edge types in its own pack the same way.
|
|
12
|
+
"""
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
from rag_wright.ontology._generated_vocab import ( # generated FROM contract_bridge.ttl (DD-7); do not hardcode here
|
|
16
|
+
AFFILIATE_OF,
|
|
17
|
+
CONTRACTS_WITH,
|
|
18
|
+
ENTITY_TYPES,
|
|
19
|
+
ORGANIZATION,
|
|
20
|
+
PERSON,
|
|
21
|
+
RELATIONSHIP_TYPES,
|
|
22
|
+
)
|
|
23
|
+
|
|
24
|
+
__all__ = ["ORGANIZATION", "PERSON", "CONTRACTS_WITH", "AFFILIATE_OF", "ENTITY_TYPES", "RELATIONSHIP_TYPES"]
|
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
"""Ontology derivation (T8, FR-C.8): reconcile the T4 ontology against the real CUAD data.
|
|
2
|
+
|
|
3
|
+
T4 declared the 41 `ClauseCategory` values from the published CUAD label set. This module confirms
|
|
4
|
+
they match the actual `master_clauses.csv` columns and produces a CSV-column -> canonical-category
|
|
5
|
+
mapping. The CSV headers carry artifacts the reconciliation absorbs so downstream annotation reading
|
|
6
|
+
(T9) binds to the canonical categories regardless: paired answer columns use inconsistent spacing
|
|
7
|
+
(`-Answer` and `- Answer`), and some names are cased differently (`Ip Ownership Assignment` vs the
|
|
8
|
+
ontology's canonical `IP Ownership Assignment`). Matching is spacing-tolerant on answer columns and
|
|
9
|
+
case-insensitive on category names; the ontology keeps its canonical casing.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import re
|
|
15
|
+
|
|
16
|
+
from pydantic import BaseModel
|
|
17
|
+
|
|
18
|
+
from rag_wright.contracts.ontology import ClauseCategory
|
|
19
|
+
|
|
20
|
+
_ANSWER_SUFFIX = re.compile(r"-\s*answer$", re.IGNORECASE)
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def clause_category_columns(header: list[str]) -> list[str]:
|
|
24
|
+
"""The category columns of `master_clauses.csv`: everything but `Filename` and answer columns."""
|
|
25
|
+
return [
|
|
26
|
+
col
|
|
27
|
+
for col in header
|
|
28
|
+
if col.strip().lower() != "filename" and not _ANSWER_SUFFIX.search(col.strip())
|
|
29
|
+
]
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
class Reconciliation(BaseModel):
|
|
33
|
+
"""The result of reconciling CSV category columns against the `ClauseCategory` ontology."""
|
|
34
|
+
|
|
35
|
+
matched: dict[str, ClauseCategory] # CSV column -> canonical category
|
|
36
|
+
missing: list[ClauseCategory] # ontology categories with no CSV column
|
|
37
|
+
extra: list[str] # CSV columns matching no ontology category
|
|
38
|
+
|
|
39
|
+
@property
|
|
40
|
+
def ok(self) -> bool:
|
|
41
|
+
return not self.missing and not self.extra
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def reconcile_clause_categories(csv_columns: list[str]) -> Reconciliation:
|
|
45
|
+
"""Match CSV category columns to `ClauseCategory` (case-insensitive), reporting gaps both ways."""
|
|
46
|
+
by_norm = {category.value.lower(): category for category in ClauseCategory}
|
|
47
|
+
matched: dict[str, ClauseCategory] = {}
|
|
48
|
+
extra: list[str] = []
|
|
49
|
+
seen: set[ClauseCategory] = set()
|
|
50
|
+
for column in csv_columns:
|
|
51
|
+
category = by_norm.get(column.strip().lower())
|
|
52
|
+
if category is None:
|
|
53
|
+
extra.append(column)
|
|
54
|
+
else:
|
|
55
|
+
matched[column] = category
|
|
56
|
+
seen.add(category)
|
|
57
|
+
missing = [category for category in ClauseCategory if category not in seen]
|
|
58
|
+
return Reconciliation(matched=matched, missing=missing, extra=extra)
|
|
@@ -0,0 +1,435 @@
|
|
|
1
|
+
"""ADR-0066: the runtime loader for the contract ontology `.ttl` -- the seed of the ontology-as-source-of-truth
|
|
2
|
+
substrate. Parses `contract_bridge.ttl` into the domain knowledge structures the engine consumes: the closed
|
|
3
|
+
vocabularies, scalar/list cardinality, the function -> applicable-dimensions applicability, the deontic polarity
|
|
4
|
+
(+ restrictive functions), and the value rollups.
|
|
5
|
+
|
|
6
|
+
Phase 0 uses this only to PROVE the ttl reproduces today's Python constants (the equivalence gate). Phase 1
|
|
7
|
+
generates the Python vocab/enums FROM this loader; Phase 2 hands the SHACL shapes straight to pyshacl. The `.ttl`
|
|
8
|
+
uses a uniform, label-keyed layer (a `cbr:PropertyDimension` node per dimension with `owl:oneOf` + a
|
|
9
|
+
`cbr:cardinality` tag; a `sh:NodeShape` per clause function; `skos:broader` for rollups), so this loader is
|
|
10
|
+
robust to IRI encoding -- it reads `rdfs:label`, never decodes an IRI.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import re
|
|
16
|
+
from dataclasses import dataclass, field
|
|
17
|
+
from functools import lru_cache
|
|
18
|
+
from pathlib import Path
|
|
19
|
+
|
|
20
|
+
from rdflib import Graph
|
|
21
|
+
from rdflib.collection import Collection
|
|
22
|
+
from rdflib.namespace import OWL, RDF, RDFS, SH, SKOS
|
|
23
|
+
|
|
24
|
+
_TTL_PATH = Path(__file__).with_name("contract_bridge.ttl")
|
|
25
|
+
_COMPLIANCE_TTL_PATH = Path(__file__).with_name("compliance_bridge.ttl")
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def load_compliance_vocab(path: Path | str = _COMPLIANCE_TTL_PATH) -> dict[str, set[str]]:
|
|
29
|
+
"""ADR-0066 P3b: the closed vocabularies declared in compliance_bridge.ttl, keyed by class local-name
|
|
30
|
+
(`DeonticType`, `ClaimType`, `Severity`, `RuleScope`, `Verdict`) -> the set of `owl:oneOf` value local-names.
|
|
31
|
+
The Python enums in contracts/compliance.py are drift-locked to this (the ttl is the source of truth)."""
|
|
32
|
+
g = Graph()
|
|
33
|
+
g.parse(str(path), format="turtle")
|
|
34
|
+
out: dict[str, set[str]] = {}
|
|
35
|
+
for cls in g.subjects(OWL.oneOf, None):
|
|
36
|
+
local = str(cls).rsplit("#", 1)[-1]
|
|
37
|
+
members = {str(m).rsplit("#", 1)[-1] for m in Collection(g, g.value(cls, OWL.oneOf))}
|
|
38
|
+
out[local] = members
|
|
39
|
+
return out
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
_CMP = "https://ragwright.local/ontology/compliance-bridge#"
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
@lru_cache(maxsize=4)
|
|
46
|
+
def load_deontic_cues(path: str = str(_COMPLIANCE_TTL_PATH)) -> frozenset[str]:
|
|
47
|
+
"""ADR-0066 P3c (Gap 1): the deontic CUES declared in compliance_bridge.ttl (`cmp:cue` on each deontic type) --
|
|
48
|
+
the lexical markers of operative normative force. The requirement-ingestion validity gate uses them: a section
|
|
49
|
+
with none of these cues is non-operative and is skipped. Cached per path."""
|
|
50
|
+
from rdflib import URIRef
|
|
51
|
+
|
|
52
|
+
g = Graph()
|
|
53
|
+
g.parse(path, format="turtle")
|
|
54
|
+
return frozenset(str(v).strip().lower() for v in g.objects(None, URIRef(_CMP + "cue")) if str(v).strip())
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
@lru_cache(maxsize=4)
|
|
58
|
+
def load_deontic_cue_map(path: str = str(_COMPLIANCE_TTL_PATH)) -> dict[str, str]:
|
|
59
|
+
"""CIC-0 (ADR-0066): the deontic CUE -> deontic TYPE map authored in compliance_bridge.ttl (`cmp:cue` on each
|
|
60
|
+
`cmp:DeonticType`), e.g. {'must': 'obligation', 'must not': 'prohibition', 'may': 'permission'}. Keys are
|
|
61
|
+
lowercased cue phrases; values are the DeonticType local-names. This is the ttl-driven source for the ingest
|
|
62
|
+
cue-RULE (`deontic_type_of`): a rule's deontic_type is derived deterministically from its text's cue instead of
|
|
63
|
+
from an LLM field. Shares its cue set with `load_deontic_cues` (the operative gate). Cached per path."""
|
|
64
|
+
from rdflib import URIRef
|
|
65
|
+
|
|
66
|
+
g = Graph()
|
|
67
|
+
g.parse(path, format="turtle")
|
|
68
|
+
cue = URIRef(_CMP + "cue")
|
|
69
|
+
out: dict[str, str] = {}
|
|
70
|
+
for subj, obj in g.subject_objects(cue):
|
|
71
|
+
value = str(obj).strip().lower()
|
|
72
|
+
if value:
|
|
73
|
+
out[value] = str(subj).rsplit("#", 1)[-1]
|
|
74
|
+
return out
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
@lru_cache(maxsize=1)
|
|
78
|
+
def _deontic_cue_type_pattern() -> tuple[re.Pattern, dict[str, str]]:
|
|
79
|
+
"""The compiled cue regex (longest cue first, so 'must not' is tried before 'must') + the cue->type map."""
|
|
80
|
+
cue_map = load_deontic_cue_map()
|
|
81
|
+
cues = sorted(cue_map, key=len, reverse=True)
|
|
82
|
+
return re.compile(r"\b(?:" + "|".join(re.escape(c) for c in cues) + r")\b", re.IGNORECASE), cue_map
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def deontic_type_of(text: str) -> str | None:
|
|
86
|
+
"""CIC-0 (ADR-0066): the deontic TYPE of a rule span, from its FIRST deontic cue (ttl `cmp:cue`), longest cue
|
|
87
|
+
first so 'must not' / 'may not' (prohibition) win over 'must' / 'may'. Returns the DeonticType local-name
|
|
88
|
+
(obligation / prohibition / permission) or None when the text carries no cue (non-operative). The deterministic
|
|
89
|
+
cue-rule that replaces the LLM's deontic_type field at ingest; a non-None result also means the span is
|
|
90
|
+
operative (same cue basis as `is_operative`)."""
|
|
91
|
+
if not text:
|
|
92
|
+
return None
|
|
93
|
+
pattern, cue_map = _deontic_cue_type_pattern()
|
|
94
|
+
m = pattern.search(text)
|
|
95
|
+
return cue_map[m.group(0).lower()] if m else None
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
@lru_cache(maxsize=4)
|
|
99
|
+
def load_actor_synonyms(path: str = str(_COMPLIANCE_TTL_PATH)) -> dict[str, str]:
|
|
100
|
+
"""ADR-0066 P4a: the actor-role synonyms from compliance_bridge.ttl -- `{synonym -> canonical role}` built from
|
|
101
|
+
each `cmp:ActorRole`'s `skos:altLabel` (synonym) -> `skos:prefLabel` (canonical). The query-side actor gate
|
|
102
|
+
(`canonical_actor`) normalizes with this. Cached per path."""
|
|
103
|
+
from rdflib import URIRef
|
|
104
|
+
|
|
105
|
+
g = Graph()
|
|
106
|
+
g.parse(path, format="turtle")
|
|
107
|
+
out: dict[str, str] = {}
|
|
108
|
+
for role in g.subjects(RDF.type, URIRef(_CMP + "ActorRole")):
|
|
109
|
+
pref = str(g.value(role, SKOS.prefLabel) or "").strip().lower()
|
|
110
|
+
if not pref:
|
|
111
|
+
continue
|
|
112
|
+
for alt in g.objects(role, SKOS.altLabel):
|
|
113
|
+
out[str(alt).strip().lower()] = pref
|
|
114
|
+
return out
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
@lru_cache(maxsize=4)
|
|
118
|
+
def load_claim_type_criteria(path: str = str(_COMPLIANCE_TTL_PATH)) -> dict[str, str]:
|
|
119
|
+
"""ADR-0119: `{ClaimType local-name -> cmp:decisionCriterion}` -- the one-line criterion each claim type uses
|
|
120
|
+
as its typed-decision (noul) question. Authored in compliance_bridge.ttl, not in capability code (ADR-0066)."""
|
|
121
|
+
from rdflib import URIRef
|
|
122
|
+
|
|
123
|
+
g = Graph()
|
|
124
|
+
g.parse(path, format="turtle")
|
|
125
|
+
crit = URIRef(_CMP + "decisionCriterion")
|
|
126
|
+
out: dict[str, str] = {}
|
|
127
|
+
for m in g.subjects(RDF.type, URIRef(_CMP + "ClaimType")):
|
|
128
|
+
c = g.value(m, crit)
|
|
129
|
+
if c is not None:
|
|
130
|
+
out[str(m).rsplit("#", 1)[-1]] = str(c)
|
|
131
|
+
return out
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
@lru_cache(maxsize=4)
|
|
135
|
+
def load_actor_role_criteria(path: str = str(_COMPLIANCE_TTL_PATH)) -> dict[str, str]:
|
|
136
|
+
"""ADR-0119: `{ActorRole prefLabel -> cmp:decisionCriterion}` -- the actor `choice` options + their criteria,
|
|
137
|
+
from the ttl (includes the workplace-domain `employer`). Authored in the ontology, not capability code."""
|
|
138
|
+
from rdflib import URIRef
|
|
139
|
+
|
|
140
|
+
g = Graph()
|
|
141
|
+
g.parse(path, format="turtle")
|
|
142
|
+
crit = URIRef(_CMP + "decisionCriterion")
|
|
143
|
+
out: dict[str, str] = {}
|
|
144
|
+
for role in g.subjects(RDF.type, URIRef(_CMP + "ActorRole")):
|
|
145
|
+
c = g.value(role, crit)
|
|
146
|
+
label = str(g.value(role, SKOS.prefLabel) or str(role).rsplit("#", 1)[-1]).strip().lower()
|
|
147
|
+
if c is not None and label:
|
|
148
|
+
out[label] = str(c)
|
|
149
|
+
return out
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
@lru_cache(maxsize=4)
|
|
153
|
+
def load_operative_rubric(path: str = str(_COMPLIANCE_TTL_PATH)) -> dict[str, str]:
|
|
154
|
+
"""ADR-0119: the operative-rule binary gate's `{instructions, true, false}` from `cmp:operativeRuleQuestion`
|
|
155
|
+
in compliance_bridge.ttl -- the decision knowledge the `jev_decision` gate asks, authored in the ontology."""
|
|
156
|
+
from rdflib import URIRef
|
|
157
|
+
|
|
158
|
+
g = Graph()
|
|
159
|
+
g.parse(path, format="turtle")
|
|
160
|
+
q = URIRef(_CMP + "operativeRuleQuestion")
|
|
161
|
+
|
|
162
|
+
def _v(prop: str) -> str:
|
|
163
|
+
v = g.value(q, URIRef(_CMP + prop))
|
|
164
|
+
return str(v) if v is not None else ""
|
|
165
|
+
|
|
166
|
+
return {"instructions": _v("questionInstructions"), "true": _v("criterionTrue"), "false": _v("criterionFalse")}
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
@lru_cache(maxsize=4)
|
|
170
|
+
def load_role_domains(path: str = str(_COMPLIANCE_TTL_PATH)) -> dict[str, str]:
|
|
171
|
+
"""ADR-0068 (engine issue 0013): the DISJOINTNESS knowledge for the actor gate -- `{canonical role -> domain}`
|
|
172
|
+
built from each `cmp:ActorRole`'s `skos:prefLabel` (canonical) -> `cmp:roleDomain`. Two roles are disjoint iff
|
|
173
|
+
both appear here with DIFFERENT domains; the recall-first gate excludes only disjoint pairs (a role absent
|
|
174
|
+
here, or two roles in the same domain, are compatible). A customer domain declares its roles' roleDomain in
|
|
175
|
+
its own pack to get cross-domain narrowing. Cached per path."""
|
|
176
|
+
from rdflib import URIRef
|
|
177
|
+
|
|
178
|
+
g = Graph()
|
|
179
|
+
g.parse(path, format="turtle")
|
|
180
|
+
out: dict[str, str] = {}
|
|
181
|
+
for role in g.subjects(RDF.type, URIRef(_CMP + "ActorRole")):
|
|
182
|
+
pref = str(g.value(role, SKOS.prefLabel) or "").strip().lower()
|
|
183
|
+
domain = str(g.value(role, URIRef(_CMP + "roleDomain")) or "").strip().lower()
|
|
184
|
+
if pref and domain:
|
|
185
|
+
out[pref] = domain
|
|
186
|
+
return out
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
_FTC_PACK_PATH = Path(__file__).parent / "packs" / "ftc_16cfr255.ttl"
|
|
190
|
+
|
|
191
|
+
|
|
192
|
+
@lru_cache(maxsize=4)
|
|
193
|
+
def load_section_overrides(path: str = str(_FTC_PACK_PATH)) -> tuple[dict[str, str], dict[str, frozenset[str]]]:
|
|
194
|
+
"""ADR-0066 P4b: a domain pack's per-section overrides from `cmp:SectionOverride` instances. Returns
|
|
195
|
+
`(rule_scope, claim_types)`: `{section -> 'content'|'context'}` (only sections that pin a scope) and
|
|
196
|
+
`{section -> {claim type value}}` (all claim types when `cmp:appliesToAllClaimTypes` is true, else the explicit
|
|
197
|
+
`cmp:appliesToClaimType` set -- empty for a definitions section). Cached per path."""
|
|
198
|
+
from rdflib import URIRef
|
|
199
|
+
|
|
200
|
+
g = Graph()
|
|
201
|
+
g.parse(path, format="turtle")
|
|
202
|
+
all_claim_types = frozenset(load_compliance_vocab().get("ClaimType", set()))
|
|
203
|
+
rule_scope: dict[str, str] = {}
|
|
204
|
+
claim_types: dict[str, frozenset[str]] = {}
|
|
205
|
+
for so in g.subjects(RDF.type, URIRef(_CMP + "SectionOverride")):
|
|
206
|
+
section = str(g.value(so, URIRef(_CMP + "section")) or "").strip()
|
|
207
|
+
if not section:
|
|
208
|
+
continue
|
|
209
|
+
all_flag = g.value(so, URIRef(_CMP + "appliesToAllClaimTypes"))
|
|
210
|
+
if all_flag is not None and bool(all_flag.toPython()):
|
|
211
|
+
claim_types[section] = all_claim_types
|
|
212
|
+
else:
|
|
213
|
+
claim_types[section] = frozenset(
|
|
214
|
+
str(ct).rsplit("#", 1)[-1] for ct in g.objects(so, URIRef(_CMP + "appliesToClaimType")))
|
|
215
|
+
rs = g.value(so, URIRef(_CMP + "overrideRuleScope"))
|
|
216
|
+
if rs is not None:
|
|
217
|
+
rule_scope[section] = str(rs).rsplit("#", 1)[-1]
|
|
218
|
+
return rule_scope, claim_types
|
|
219
|
+
|
|
220
|
+
|
|
221
|
+
@lru_cache(maxsize=4)
|
|
222
|
+
def load_shapes_graph(path: str = str(_TTL_PATH)) -> Graph:
|
|
223
|
+
"""ADR-0066 P2: the ttl parsed as an rdflib Graph -- its `sh:NodeShape`s ARE the SHACL shapes handed to pyshacl
|
|
224
|
+
at runtime, so the symbolic layer reads the symbolic artifact directly (no Python-built shapes). pyshacl uses
|
|
225
|
+
the shapes and ignores the ttl's non-SHACL triples. Cached per path."""
|
|
226
|
+
g = Graph()
|
|
227
|
+
g.parse(path, format="turtle")
|
|
228
|
+
return g
|
|
229
|
+
_CBR = "https://ragwright.local/ontology/contract-bridge#"
|
|
230
|
+
# The engine's pack-schema DECLARATION meta-vocabulary (DD-8): the classes/predicates a pack `.ttl` uses to declare
|
|
231
|
+
# its KG node/edge types + entity taxonomy (KgVertexType/vertexName/kgProperty/uniqueIndexOn/KgStructuralEdge/
|
|
232
|
+
# edgeName, EntityNodeType/EntityRelationshipType). Engine-namespaced (not contract-namespaced), so a NON-contract
|
|
233
|
+
# domain declares its schema in this shared language without borrowing the contract pack's namespace (AC-journey).
|
|
234
|
+
_ENG = "https://ragwright.local/ontology/engine#"
|
|
235
|
+
_DIMENSION_CLASS = _CBR + "PropertyDimension"
|
|
236
|
+
|
|
237
|
+
|
|
238
|
+
@dataclass(frozen=True)
|
|
239
|
+
class ContractOntologyView:
|
|
240
|
+
"""The contract domain knowledge parsed out of `contract_bridge.ttl` (all string-keyed by label)."""
|
|
241
|
+
|
|
242
|
+
closed_vocab: dict[str, set[str]] = field(default_factory=dict)
|
|
243
|
+
multivalued: set[str] = field(default_factory=set)
|
|
244
|
+
function_applicable_dims: dict[str, set[str]] = field(default_factory=dict)
|
|
245
|
+
permission_polarity: dict[str, set[str]] = field(default_factory=dict)
|
|
246
|
+
restrictive_functions: set[str] = field(default_factory=set)
|
|
247
|
+
value_rollup: dict[str, dict[str, set[str]]] = field(default_factory=dict)
|
|
248
|
+
# issue 0037: ingest synonyms -- {dimension: {normalized surface term: canonical member}}. A skos:broader edge
|
|
249
|
+
# whose BROADER is a closed-vocab member but whose NARROWER is not (a specific surface term). Used at ingest to
|
|
250
|
+
# canonicalize an out-of-vocab extracted value onto its canonical member (else the value is kept verbatim).
|
|
251
|
+
value_synonyms: dict[str, dict[str, str]] = field(default_factory=dict)
|
|
252
|
+
# DD-7 (ADR-0066): the entity-graph taxonomy -- entity node types (cbr:EntityNodeType) + entity-to-entity
|
|
253
|
+
# relationship types (cbr:EntityRelationshipType), by label. Was the Python EntityType/RelationshipType enum.
|
|
254
|
+
entity_types: set[str] = field(default_factory=set)
|
|
255
|
+
relationship_types: set[str] = field(default_factory=set)
|
|
256
|
+
|
|
257
|
+
|
|
258
|
+
def _label(g: Graph, node) -> str:
|
|
259
|
+
lbl = g.value(node, RDFS.label)
|
|
260
|
+
return str(lbl) if lbl is not None else ""
|
|
261
|
+
|
|
262
|
+
|
|
263
|
+
def load_contract_ontology(path: Path | str = _TTL_PATH) -> ContractOntologyView:
|
|
264
|
+
"""Parse the contract bridge ontology into a `ContractOntologyView`."""
|
|
265
|
+
g = Graph()
|
|
266
|
+
g.parse(str(path), format="turtle")
|
|
267
|
+
|
|
268
|
+
closed_vocab: dict[str, set[str]] = {}
|
|
269
|
+
multivalued: set[str] = set()
|
|
270
|
+
dim_label_by_node: dict[str, str] = {} # dimension IRI -> label (for the applicability shapes)
|
|
271
|
+
value_label_by_node: dict[str, str] = {} # value IRI -> label (for rollups + oneOf)
|
|
272
|
+
value_dim_by_node: dict[str, str] = {} # value IRI -> its dimension label (for rollups)
|
|
273
|
+
|
|
274
|
+
for dim in g.subjects(RDF.type, _dim_class()):
|
|
275
|
+
label = _label(g, dim)
|
|
276
|
+
dim_label_by_node[str(dim)] = label
|
|
277
|
+
if str(g.value(dim, _cbr("cardinality")) or "") == "list":
|
|
278
|
+
multivalued.add(label)
|
|
279
|
+
one_of = g.value(dim, OWL.oneOf)
|
|
280
|
+
if one_of is not None: # a CLOSED dimension (open-valued dims omit owl:oneOf)
|
|
281
|
+
members = list(Collection(g, one_of))
|
|
282
|
+
values = set()
|
|
283
|
+
for m in members:
|
|
284
|
+
vlabel = _label(g, m)
|
|
285
|
+
values.add(vlabel)
|
|
286
|
+
value_label_by_node[str(m)] = vlabel
|
|
287
|
+
value_dim_by_node[str(m)] = label
|
|
288
|
+
closed_vocab[label] = values
|
|
289
|
+
|
|
290
|
+
# Applicability + deontic: one sh:NodeShape per clause function.
|
|
291
|
+
function_applicable_dims: dict[str, set[str]] = {}
|
|
292
|
+
permission_polarity: dict[str, set[str]] = {}
|
|
293
|
+
restrictive_functions: set[str] = set()
|
|
294
|
+
for shape in g.subjects(RDF.type, SH.NodeShape):
|
|
295
|
+
target = g.value(shape, SH.targetClass)
|
|
296
|
+
fn = _label(g, target)
|
|
297
|
+
if not fn:
|
|
298
|
+
continue
|
|
299
|
+
dims: set[str] = set()
|
|
300
|
+
for prop in g.objects(shape, SH.property):
|
|
301
|
+
path = g.value(prop, SH.path)
|
|
302
|
+
dlabel = dim_label_by_node.get(str(path), _label(g, path))
|
|
303
|
+
dims.add(dlabel)
|
|
304
|
+
in_list = g.value(prop, SH["in"])
|
|
305
|
+
if in_list is not None: # deontic: restrictive function forbids the permission-polarity values
|
|
306
|
+
restrictive_functions.add(fn)
|
|
307
|
+
allowed = {str(x) for x in Collection(g, in_list)}
|
|
308
|
+
forbidden = closed_vocab.get(dlabel, set()) - allowed
|
|
309
|
+
if forbidden:
|
|
310
|
+
permission_polarity.setdefault(dlabel, set()).update(forbidden)
|
|
311
|
+
function_applicable_dims[fn] = dims
|
|
312
|
+
|
|
313
|
+
# Value rollups: skos:broader between value individuals.
|
|
314
|
+
value_rollup: dict[str, dict[str, set[str]]] = {}
|
|
315
|
+
# issue 0037: ingest synonyms -- {dim: {normalized surface: canonical member}} from a skos:broader edge whose
|
|
316
|
+
# BROADER is a closed member but whose NARROWER is not (a specific surface term); surface = narrower local-name
|
|
317
|
+
# + its skos:altLabels. (A narrower that IS a member is a query-side rollup, handled above.)
|
|
318
|
+
value_synonyms: dict[str, dict[str, str]] = {}
|
|
319
|
+
for narrower, broader in g.subject_objects(SKOS.broader):
|
|
320
|
+
dim = value_dim_by_node.get(str(narrower))
|
|
321
|
+
if dim is not None: # narrower is itself a vocab member -> a query-side value rollup
|
|
322
|
+
value_rollup.setdefault(dim, {}).setdefault(
|
|
323
|
+
value_label_by_node.get(str(narrower), ""), set()).add(
|
|
324
|
+
value_label_by_node.get(str(broader), ""))
|
|
325
|
+
continue
|
|
326
|
+
bdim = value_dim_by_node.get(str(broader)) # narrower is a surface synonym -> map to the broader member
|
|
327
|
+
bval = value_label_by_node.get(str(broader))
|
|
328
|
+
if not bdim or not bval:
|
|
329
|
+
continue
|
|
330
|
+
surfaces = {_label(g, narrower) or str(narrower).rsplit("#", 1)[-1]}
|
|
331
|
+
surfaces |= {str(a) for a in g.objects(narrower, SKOS.altLabel)}
|
|
332
|
+
for s in surfaces:
|
|
333
|
+
key = re.sub(r"[^A-Za-z0-9]+", "", s).lower()
|
|
334
|
+
if key:
|
|
335
|
+
value_synonyms.setdefault(bdim, {})[key] = bval
|
|
336
|
+
|
|
337
|
+
# DD-7: the entity-graph taxonomy (entity node types + entity-to-entity relationship types), by label.
|
|
338
|
+
entity_types = {_label(g, s) for s in g.subjects(RDF.type, _eng("EntityNodeType"))}
|
|
339
|
+
relationship_types = {_label(g, s) for s in g.subjects(RDF.type, _eng("EntityRelationshipType"))}
|
|
340
|
+
|
|
341
|
+
return ContractOntologyView(
|
|
342
|
+
closed_vocab=closed_vocab, multivalued=multivalued,
|
|
343
|
+
function_applicable_dims=function_applicable_dims,
|
|
344
|
+
permission_polarity=permission_polarity, restrictive_functions=restrictive_functions,
|
|
345
|
+
value_synonyms=value_synonyms,
|
|
346
|
+
value_rollup=value_rollup,
|
|
347
|
+
entity_types=entity_types, relationship_types=relationship_types)
|
|
348
|
+
|
|
349
|
+
|
|
350
|
+
@dataclass(frozen=True)
|
|
351
|
+
class KgVertexType:
|
|
352
|
+
"""ADR-0067 P5b: a domain KG vertex-type declaration the store creates -- name, its `(property, SQL type)`
|
|
353
|
+
pairs, and the property to build a UNIQUE index on (if any)."""
|
|
354
|
+
|
|
355
|
+
name: str
|
|
356
|
+
properties: tuple[tuple[str, str], ...]
|
|
357
|
+
unique_index: str | None
|
|
358
|
+
|
|
359
|
+
|
|
360
|
+
@lru_cache(maxsize=8)
|
|
361
|
+
def load_kg_schema(path: str | None = None) -> tuple[tuple[KgVertexType, ...], frozenset[str]]:
|
|
362
|
+
"""ADR-0067 P5b: the DOMAIN KG node/edge storage schema from the ttl -- `(vertex types, structural edge names)`.
|
|
363
|
+
The engine infra (Chunk/Span/Entity) stays generic in store code; these domain types are pack-declared. Cached.
|
|
364
|
+
`path=None` is the engine's reference CONTRACT pack; a new domain passes its OWN pack `.ttl` (AC-journey)."""
|
|
365
|
+
g = Graph()
|
|
366
|
+
g.parse(str(path or _TTL_PATH), format="turtle")
|
|
367
|
+
vertices = []
|
|
368
|
+
for v in g.subjects(RDF.type, _eng("KgVertexType")):
|
|
369
|
+
props = tuple(sorted((str(p).split(":", 1)[0], str(p).split(":", 1)[1])
|
|
370
|
+
for p in g.objects(v, _eng("kgProperty")) if ":" in str(p)))
|
|
371
|
+
ui = g.value(v, _eng("uniqueIndexOn"))
|
|
372
|
+
vertices.append(KgVertexType(name=str(g.value(v, _eng("vertexName"))), properties=props,
|
|
373
|
+
unique_index=(str(ui) if ui is not None else None)))
|
|
374
|
+
vertices.sort(key=lambda x: x.name)
|
|
375
|
+
edges = frozenset(str(g.value(e, _eng("edgeName")))
|
|
376
|
+
for e in g.subjects(RDF.type, _eng("KgStructuralEdge")))
|
|
377
|
+
return tuple(vertices), edges
|
|
378
|
+
|
|
379
|
+
|
|
380
|
+
@lru_cache(maxsize=4)
|
|
381
|
+
def load_typed_edges(path: str = str(_TTL_PATH)) -> tuple[dict[str, str], dict[str, str]]:
|
|
382
|
+
"""ADR-0067 P5a: the KG typed-edge map from contract_bridge.ttl. Returns `(dim_edge, edge_iri)`:
|
|
383
|
+
`{dimension value -> KG edge type}` (from `cbr:kgEdge`) and `{edge type -> predicate IRI}` (from
|
|
384
|
+
`cbr:predicateIri` on each `cbr:KgEdgeType`). The property-graph edge types are ontology-authoritative;
|
|
385
|
+
`store/arcadedb.py` builds `_TYPED_DIMENSION_EDGE` / `_edge_predicate_iri` from this. Cached per path."""
|
|
386
|
+
g = Graph()
|
|
387
|
+
g.parse(str(path), format="turtle")
|
|
388
|
+
edge_iri = {str(g.value(e, RDFS.label)): str(g.value(e, _cbr("predicateIri")))
|
|
389
|
+
for e in g.subjects(RDF.type, _cbr("KgEdgeType"))}
|
|
390
|
+
dim_edge = {str(g.value(dim, RDFS.label)): str(g.value(edge, RDFS.label))
|
|
391
|
+
for dim, edge in g.subject_objects(_cbr("kgEdge"))}
|
|
392
|
+
return dim_edge, edge_iri
|
|
393
|
+
|
|
394
|
+
|
|
395
|
+
def _dim_class():
|
|
396
|
+
from rdflib import URIRef
|
|
397
|
+
return URIRef(_DIMENSION_CLASS)
|
|
398
|
+
|
|
399
|
+
|
|
400
|
+
def _cbr(frag: str):
|
|
401
|
+
from rdflib import URIRef
|
|
402
|
+
return URIRef(_CBR + frag)
|
|
403
|
+
|
|
404
|
+
|
|
405
|
+
def _eng(frag: str):
|
|
406
|
+
from rdflib import URIRef
|
|
407
|
+
return URIRef(_ENG + frag)
|
|
408
|
+
|
|
409
|
+
|
|
410
|
+
def load_template_fields(path: Path | str = _TTL_PATH):
|
|
411
|
+
"""ADR-0066 P1b-1: the extraction template's fields as captured in the ttl, as `TemplateFieldSpec`s in the
|
|
412
|
+
same order the template declares them (`cbr:fieldOrder`). Round-trips `bootstrap_template_capture` -- the drift
|
|
413
|
+
test asserts this equals a fresh introspection of `clause_template.py`."""
|
|
414
|
+
from rag_wright.ontology.template_introspect import TemplateFieldSpec
|
|
415
|
+
|
|
416
|
+
g = Graph()
|
|
417
|
+
g.parse(str(path), format="turtle")
|
|
418
|
+
specs: list = []
|
|
419
|
+
for node in g.subjects(RDF.type, _cbr("TemplateField")):
|
|
420
|
+
order = g.value(node, _cbr("fieldOrder"))
|
|
421
|
+
ml = g.value(node, _cbr("maxLength"))
|
|
422
|
+
specs.append((int(order), TemplateFieldSpec(
|
|
423
|
+
model=str(g.value(node, _cbr("onModel"))),
|
|
424
|
+
name=str(g.value(node, RDFS.label)),
|
|
425
|
+
kind=str(g.value(node, _cbr("fieldKind"))),
|
|
426
|
+
default_token=str(g.value(node, _cbr("default"))),
|
|
427
|
+
definition=str(g.value(node, SKOS.definition) or ""),
|
|
428
|
+
enum_class=(str(v) if (v := g.value(node, _cbr("enumClass"))) is not None else None),
|
|
429
|
+
model_ref=(str(v) if (v := g.value(node, _cbr("modelRef"))) is not None else None),
|
|
430
|
+
edge_label=(str(v) if (v := g.value(node, _cbr("edgeLabel"))) is not None else None),
|
|
431
|
+
max_length=(int(ml) if ml is not None else None),
|
|
432
|
+
examples=(tuple(str(e) for e in Collection(g, exlist))
|
|
433
|
+
if (exlist := g.value(node, _cbr("examples"))) is not None else ()),
|
|
434
|
+
)))
|
|
435
|
+
return [spec for _, spec in sorted(specs, key=lambda t: t[0])]
|
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
# ftc_16cfr255.ttl -- the FTC 16 CFR Part 255 (Endorsement Guides) DOMAIN PACK (ADR-0066 P4b).
|
|
2
|
+
#
|
|
3
|
+
# The REFERENCE domain pack: the curated per-section overrides for the FTC endorsement guides, as
|
|
4
|
+
# cmp:SectionOverride instances (the shape is declared in compliance_bridge.ttl). Loaded by the query side to
|
|
5
|
+
# override DEON-1 rule scope + DEON-8 applicable claim types for FTC-cited requirements. A customer domain ships
|
|
6
|
+
# its OWN pack; nothing FTC-specific is hardcoded in engine code.
|
|
7
|
+
#
|
|
8
|
+
# KEY FINDING (CC-6): the FTC endorsement guides apply by CONTEXT (is the ad an endorsement?), NOT by claim_type,
|
|
9
|
+
# so every operative section applies to ALL claim types; only the definitions section (255.0) applies to none.
|
|
10
|
+
# 255.5 (material-connection disclosure) + 255.4 (organization endorsements) are CONTEXT rules (always included).
|
|
11
|
+
|
|
12
|
+
@prefix ftc: <https://ragwright.local/ontology/packs/ftc-16cfr255#> .
|
|
13
|
+
@prefix cmp: <https://ragwright.local/ontology/compliance-bridge#> .
|
|
14
|
+
@prefix owl: <http://www.w3.org/2002/07/owl#> .
|
|
15
|
+
@prefix rdfs: <http://www.w3.org/2000/01/rdf-schema#> .
|
|
16
|
+
|
|
17
|
+
<https://ragwright.local/ontology/packs/ftc-16cfr255> a owl:Ontology ;
|
|
18
|
+
rdfs:label "FTC 16 CFR 255 domain pack" ;
|
|
19
|
+
rdfs:comment "Curated per-section overrides for the FTC Endorsement Guides (the engine's reference domain pack)." .
|
|
20
|
+
|
|
21
|
+
ftc:s255_0 a cmp:SectionOverride ; cmp:section "255.0" ; cmp:appliesToAllClaimTypes false . # definitions -> none
|
|
22
|
+
ftc:s255_1 a cmp:SectionOverride ; cmp:section "255.1" ; cmp:appliesToAllClaimTypes true . # general considerations
|
|
23
|
+
ftc:s255_2 a cmp:SectionOverride ; cmp:section "255.2" ; cmp:appliesToAllClaimTypes true . # consumer endorsements
|
|
24
|
+
ftc:s255_3 a cmp:SectionOverride ; cmp:section "255.3" ; cmp:appliesToAllClaimTypes true . # expert endorsements
|
|
25
|
+
ftc:s255_4 a cmp:SectionOverride ; cmp:section "255.4" ; cmp:appliesToAllClaimTypes true ; # organization endorsements
|
|
26
|
+
cmp:overrideRuleScope cmp:context .
|
|
27
|
+
ftc:s255_5 a cmp:SectionOverride ; cmp:section "255.5" ; cmp:appliesToAllClaimTypes true ; # material-connection disclosure
|
|
28
|
+
cmp:overrideRuleScope cmp:context .
|
|
29
|
+
ftc:s255_6 a cmp:SectionOverride ; cmp:section "255.6" ; cmp:appliesToAllClaimTypes true . # endorsements to children
|
|
@@ -0,0 +1,87 @@
|
|
|
1
|
+
"""The entity registry (T8, FR-C.8 / FR-C.7) -- a DOMAIN-NEUTRAL closed-world registry of canonical entities.
|
|
2
|
+
|
|
3
|
+
ADR-0067: the registry is generic. A domain's canonical `entity_id`s and their surface normalization are the
|
|
4
|
+
domain's concern: the surface-form normalizer is INJECTABLE (`EntityRegistry(normalize=...)`, default = a generic
|
|
5
|
+
name key), and the domain's own builder constructs the registry (the reference builder lives in the corpus
|
|
6
|
+
layer, keyed by that corpus's canonical ids). This module imports nothing corpus-specific.
|
|
7
|
+
|
|
8
|
+
Lookup is **closed-world**: `resolve` returns `None` for an unknown surface form, never a fabricated id. The
|
|
9
|
+
concrete matching strategy (fuzzy / embedding / language-model-assisted, SPEC section 16.3) is T24's; this
|
|
10
|
+
registry is the closed set T24 resolves against, plus an exact normalized-surface-form index.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import re
|
|
16
|
+
from typing import Callable, Optional, Protocol, runtime_checkable
|
|
17
|
+
|
|
18
|
+
from pydantic import BaseModel
|
|
19
|
+
|
|
20
|
+
from rag_wright.contracts.identifiers import EntityId
|
|
21
|
+
|
|
22
|
+
_NON_ALNUM = re.compile(r"[^a-z0-9]+")
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def default_surface_key(name: str) -> str:
|
|
26
|
+
"""The default DOMAIN-NEUTRAL surface-form key: lowercase, non-alphanumeric runs folded to a single space."""
|
|
27
|
+
return _NON_ALNUM.sub(" ", (name or "").lower()).strip()
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
@runtime_checkable
|
|
31
|
+
class EntityResolver(Protocol):
|
|
32
|
+
"""The entity-resolution seam (DD-3, ADR-0067 P5c): map a surface form to a canonical `entity_id`, or
|
|
33
|
+
`None` (closed-world -- never a fabricated id). The RESOLUTION STRATEGY is the domain's concern and is
|
|
34
|
+
injected into `resolve_entities` (FR-C.7): the generic default is the exact-normalized surface-form
|
|
35
|
+
`EntityRegistry` (below); a domain pack injects its own registry built by that corpus's loader in the
|
|
36
|
+
corpus layer (keyed by that domain's canonical ids). A product may bind any strategy (fuzzy / embedding /
|
|
37
|
+
an external service) as long as it honors this signature and the closed-world contract. The generic
|
|
38
|
+
resolution invariants (cluster surface-form sweep, two-channel dedup, self-loop dropping) stay in the
|
|
39
|
+
capability, not the resolver. This module names no specific domain (the SEC-free scope guard enforces it)."""
|
|
40
|
+
|
|
41
|
+
def resolve(self, surface_form: str) -> Optional[EntityId]:
|
|
42
|
+
...
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
class RegistryRecord(BaseModel):
|
|
46
|
+
"""One registered entity: its canonical id, conformed name, ticker, and known aliases."""
|
|
47
|
+
|
|
48
|
+
entity_id: EntityId
|
|
49
|
+
canonical_name: str
|
|
50
|
+
ticker: Optional[str] = None
|
|
51
|
+
aliases: list[str] = []
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
class EntityRegistry:
|
|
55
|
+
"""A closed-world registry of canonical entities, indexed by normalized surface form."""
|
|
56
|
+
|
|
57
|
+
def __init__(self, *, normalize: Optional[Callable[[str], str]] = None) -> None:
|
|
58
|
+
self._by_id: dict[str, RegistryRecord] = {}
|
|
59
|
+
self._index: dict[str, EntityId] = {} # normalized surface form -> entity_id
|
|
60
|
+
self.skipped_ids: list[str] = [] # raw canonical-id values that failed normalization (builder-specific)
|
|
61
|
+
self._normalize = normalize or default_surface_key # ADR-0067: domain-neutral, injectable
|
|
62
|
+
|
|
63
|
+
def _index_surface(self, surface: str, entity_id: EntityId) -> None:
|
|
64
|
+
key = self._normalize(surface)
|
|
65
|
+
if key:
|
|
66
|
+
self._index.setdefault(key, entity_id)
|
|
67
|
+
|
|
68
|
+
def add(self, record: RegistryRecord) -> None:
|
|
69
|
+
self._by_id[record.entity_id.value] = record
|
|
70
|
+
self._index_surface(record.canonical_name, record.entity_id)
|
|
71
|
+
if record.ticker:
|
|
72
|
+
self._index_surface(record.ticker, record.entity_id)
|
|
73
|
+
for alias in record.aliases:
|
|
74
|
+
self._index_surface(alias, record.entity_id)
|
|
75
|
+
|
|
76
|
+
def get(self, entity_id: EntityId) -> Optional[RegistryRecord]:
|
|
77
|
+
return self._by_id.get(entity_id.value)
|
|
78
|
+
|
|
79
|
+
def resolve(self, surface_form: str) -> Optional[EntityId]:
|
|
80
|
+
"""The canonical `entity_id` for a known surface form, or `None` (closed-world)."""
|
|
81
|
+
return self._index.get(self._normalize(surface_form))
|
|
82
|
+
|
|
83
|
+
def __len__(self) -> int:
|
|
84
|
+
return len(self._by_id)
|
|
85
|
+
|
|
86
|
+
def __contains__(self, entity_id: EntityId) -> bool:
|
|
87
|
+
return entity_id.value in self._by_id
|