rag-wright 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (184) hide show
  1. rag_wright/__init__.py +13 -0
  2. rag_wright/api/__init__.py +33 -0
  3. rag_wright/api/config.py +59 -0
  4. rag_wright/api/discover.py +70 -0
  5. rag_wright/api/documents.py +39 -0
  6. rag_wright/api/ids.py +31 -0
  7. rag_wright/api/invoke.py +99 -0
  8. rag_wright/api/kg.py +61 -0
  9. rag_wright/api/mcp.py +94 -0
  10. rag_wright/api/usage.py +30 -0
  11. rag_wright/api/workspace.py +85 -0
  12. rag_wright/capabilities/__init__.py +8 -0
  13. rag_wright/capabilities/answer_generator.py +427 -0
  14. rag_wright/capabilities/ard.py +286 -0
  15. rag_wright/capabilities/assertion_extraction.py +79 -0
  16. rag_wright/capabilities/chunk_read.py +58 -0
  17. rag_wright/capabilities/chunk_write.py +163 -0
  18. rag_wright/capabilities/claim_extraction.py +153 -0
  19. rag_wright/capabilities/clause_exception_linking.py +117 -0
  20. rag_wright/capabilities/compliance_judgment.py +322 -0
  21. rag_wright/capabilities/compliance_store.py +87 -0
  22. rag_wright/capabilities/contract_kg_serve.py +156 -0
  23. rag_wright/capabilities/contract_kg_store.py +251 -0
  24. rag_wright/capabilities/dg_extraction.py +585 -0
  25. rag_wright/capabilities/disambiguation.py +163 -0
  26. rag_wright/capabilities/document_parse.py +87 -0
  27. rag_wright/capabilities/document_scope.py +49 -0
  28. rag_wright/capabilities/embedding.py +164 -0
  29. rag_wright/capabilities/embedding_profiles.py +43 -0
  30. rag_wright/capabilities/entity_resolution.py +154 -0
  31. rag_wright/capabilities/fusion.py +64 -0
  32. rag_wright/capabilities/graph_extraction.py +243 -0
  33. rag_wright/capabilities/graph_query.py +73 -0
  34. rag_wright/capabilities/graph_storage.py +111 -0
  35. rag_wright/capabilities/highlight_serve.py +142 -0
  36. rag_wright/capabilities/hybrid_search.py +65 -0
  37. rag_wright/capabilities/invoke.py +31 -0
  38. rag_wright/capabilities/jev_decision.py +38 -0
  39. rag_wright/capabilities/manifests.py +872 -0
  40. rag_wright/capabilities/okf_navigate.py +456 -0
  41. rag_wright/capabilities/parsing.py +286 -0
  42. rag_wright/capabilities/property_boosted_retrieval.py +125 -0
  43. rag_wright/capabilities/query_function_classifier.py +94 -0
  44. rag_wright/capabilities/query_understanding.py +109 -0
  45. rag_wright/capabilities/registry.py +262 -0
  46. rag_wright/capabilities/remote_encoders.py +94 -0
  47. rag_wright/capabilities/requirement_extraction.py +247 -0
  48. rag_wright/capabilities/reranking.py +123 -0
  49. rag_wright/capabilities/retrieval_core.py +126 -0
  50. rag_wright/capabilities/rlm_chunking.py +808 -0
  51. rag_wright/capabilities/rlm_synthesis.py +316 -0
  52. rag_wright/capabilities/scan_quality.py +136 -0
  53. rag_wright/capabilities/span_relevance_judgment.py +191 -0
  54. rag_wright/capabilities/vision_to_text.py +85 -0
  55. rag_wright/capabilities/vlm_ocr.py +85 -0
  56. rag_wright/contracts/__init__.py +6 -0
  57. rag_wright/contracts/chunk.py +79 -0
  58. rag_wright/contracts/compliance.py +303 -0
  59. rag_wright/contracts/contract_meta.py +27 -0
  60. rag_wright/contracts/extraction.py +130 -0
  61. rag_wright/contracts/function.py +167 -0
  62. rag_wright/contracts/function_routing.py +91 -0
  63. rag_wright/contracts/highlight.py +74 -0
  64. rag_wright/contracts/identifiers.py +153 -0
  65. rag_wright/contracts/jurisdiction.py +96 -0
  66. rag_wright/contracts/ontology.py +142 -0
  67. rag_wright/contracts/property.py +201 -0
  68. rag_wright/contracts/provenance.py +78 -0
  69. rag_wright/contracts/query_intent.py +53 -0
  70. rag_wright/contracts/span.py +76 -0
  71. rag_wright/contracts/value_match.py +84 -0
  72. rag_wright/corpus/__init__.py +0 -0
  73. rag_wright/corpus/canonicalize.py +116 -0
  74. rag_wright/corpus/cuad.py +153 -0
  75. rag_wright/corpus/cuad_ingestion.py +72 -0
  76. rag_wright/corpus/document_parser.py +299 -0
  77. rag_wright/corpus/edgar.py +231 -0
  78. rag_wright/corpus/gcs_ingestion.py +120 -0
  79. rag_wright/corpus/http.py +110 -0
  80. rag_wright/corpus/selection.py +152 -0
  81. rag_wright/mcp/__init__.py +11 -0
  82. rag_wright/mcp/compliance_server.py +299 -0
  83. rag_wright/mcp/intra_document_qa_server.py +170 -0
  84. rag_wright/mcp/relational_qa_server.py +171 -0
  85. rag_wright/mcp/session_store.py +64 -0
  86. rag_wright/mcp/typed_property_retrieval_server.py +191 -0
  87. rag_wright/models/__init__.py +8 -0
  88. rag_wright/models/profiles.py +331 -0
  89. rag_wright/models/seam.py +497 -0
  90. rag_wright/models/tag_structured.py +285 -0
  91. rag_wright/models/tracing.py +179 -0
  92. rag_wright/models/usage.py +102 -0
  93. rag_wright/okf/__init__.py +11 -0
  94. rag_wright/okf/compile.py +292 -0
  95. rag_wright/okf/document.py +47 -0
  96. rag_wright/okf/enrich.py +176 -0
  97. rag_wright/okf/links.py +190 -0
  98. rag_wright/okf/lint.py +105 -0
  99. rag_wright/ontology/__init__.py +6 -0
  100. rag_wright/ontology/_generated_template_meta.py +60 -0
  101. rag_wright/ontology/_generated_vocab.py +52 -0
  102. rag_wright/ontology/clause_template.py +964 -0
  103. rag_wright/ontology/codegen.py +84 -0
  104. rag_wright/ontology/compliance_bridge.ttl +186 -0
  105. rag_wright/ontology/contract_bridge.ttl +2685 -0
  106. rag_wright/ontology/contract_taxonomy.py +24 -0
  107. rag_wright/ontology/derive.py +58 -0
  108. rag_wright/ontology/loader.py +435 -0
  109. rag_wright/ontology/packs/ftc_16cfr255.ttl +29 -0
  110. rag_wright/ontology/registry.py +87 -0
  111. rag_wright/ontology/template_introspect.py +100 -0
  112. rag_wright/py.typed +0 -0
  113. rag_wright/reference/__init__.py +2 -0
  114. rag_wright/reference/compliance.py +41 -0
  115. rag_wright/reference/contract_seam.py +123 -0
  116. rag_wright/skills/__init__.py +7 -0
  117. rag_wright/skills/claim_extraction/SKILL.md +47 -0
  118. rag_wright/skills/claim_extraction/__init__.py +1 -0
  119. rag_wright/skills/claim_extraction/template.py +50 -0
  120. rag_wright/skills/compliance_judgment/SKILL.md +59 -0
  121. rag_wright/skills/corpus_ingest/SKILL.md +106 -0
  122. rag_wright/skills/extraction_semantic_judge/SKILL.md +51 -0
  123. rag_wright/skills/extraction_semantic_judge/__init__.py +1 -0
  124. rag_wright/skills/generation/SKILL.md +64 -0
  125. rag_wright/skills/generation/__init__.py +1 -0
  126. rag_wright/skills/generic_compliance_judgment/SKILL.md +58 -0
  127. rag_wright/skills/okf_navigate/SKILL.md +137 -0
  128. rag_wright/skills/requirement_extraction/SKILL.md +47 -0
  129. rag_wright/skills/requirement_extraction/__init__.py +1 -0
  130. rag_wright/skills/requirement_extraction/template.py +50 -0
  131. rag_wright/skills/rlm/SKILL.md +186 -0
  132. rag_wright/skills/rlm/__init__.py +31 -0
  133. rag_wright/skills/rlm/agent.py +292 -0
  134. rag_wright/skills/span_relevance_judgment/SKILL.md +67 -0
  135. rag_wright/skills/vision_to_text/SKILL.md +36 -0
  136. rag_wright/skills/vision_to_text/__init__.py +1 -0
  137. rag_wright/spans/__init__.py +1 -0
  138. rag_wright/spans/boundary.py +78 -0
  139. rag_wright/spans/clause_function_classifier.py +490 -0
  140. rag_wright/spans/clause_kg_extractor.py +337 -0
  141. rag_wright/spans/cuad_labels.py +81 -0
  142. rag_wright/spans/dim_classifier.py +158 -0
  143. rag_wright/spans/dim_fleet.json +411 -0
  144. rag_wright/spans/function_classifier.py +77 -0
  145. rag_wright/spans/function_families.py +62 -0
  146. rag_wright/spans/hybrid_classifier.py +103 -0
  147. rag_wright/spans/legalbert_classifier.py +83 -0
  148. rag_wright/spans/model_capabilities.py +107 -0
  149. rag_wright/spans/new_function_labels.py +111 -0
  150. rag_wright/spans/page_map.py +68 -0
  151. rag_wright/spans/property_extractor.py +365 -0
  152. rag_wright/spans/property_grounding.py +182 -0
  153. rag_wright/spans/reclassify.py +77 -0
  154. rag_wright/spans/scarce_function_labels.py +105 -0
  155. rag_wright/spans/segment.py +341 -0
  156. rag_wright/spans/semantic_judge.py +197 -0
  157. rag_wright/spans/symbolic_validation.py +131 -0
  158. rag_wright/spans/tag_clause_extractor.py +182 -0
  159. rag_wright/store/__init__.py +6 -0
  160. rag_wright/store/arcadedb.py +1135 -0
  161. rag_wright/store/chunk_text.py +66 -0
  162. rag_wright/store/seam.py +213 -0
  163. rag_wright/subgraphs/__init__.py +0 -0
  164. rag_wright/subgraphs/async_ingestion.py +204 -0
  165. rag_wright/subgraphs/compliance_check.py +1042 -0
  166. rag_wright/subgraphs/compliance_ingestion.py +306 -0
  167. rag_wright/subgraphs/contract_ingestion_pipeline.py +999 -0
  168. rag_wright/subgraphs/graph_extraction.py +102 -0
  169. rag_wright/subgraphs/intra_document_qa.py +328 -0
  170. rag_wright/subgraphs/observability.py +140 -0
  171. rag_wright/subgraphs/query_constraint_extraction.py +73 -0
  172. rag_wright/subgraphs/relational_qa.py +165 -0
  173. rag_wright/subgraphs/requirement_extraction.py +137 -0
  174. rag_wright/subgraphs/scaffold.py +65 -0
  175. rag_wright/subgraphs/semantic_chunking.py +183 -0
  176. rag_wright/subgraphs/typed_clause_extraction.py +172 -0
  177. rag_wright/subgraphs/typed_property_retrieval.py +278 -0
  178. rag_wright/util/__init__.py +1 -0
  179. rag_wright/util/concurrent.py +153 -0
  180. rag_wright/util/spacy_model.py +45 -0
  181. rag_wright-0.1.0.dist-info/METADATA +168 -0
  182. rag_wright-0.1.0.dist-info/RECORD +184 -0
  183. rag_wright-0.1.0.dist-info/WHEEL +4 -0
  184. rag_wright-0.1.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,130 @@
1
+ """The graph extraction contract and extractor seam (FR-C.6, FR-I.4).
2
+
3
+ Graph extraction is a hybrid stack (FR-C.6): docling-graph contract extraction for schema entities,
4
+ a lightweight NER-plus-dependency path for the bulk, an open-ended language-model escalation for
5
+ hard cases, and later Open Information Extraction (OpenIE). This module is the *contract* those
6
+ extractors conform to, not the extractors themselves (those are the graph-extraction capability,
7
+ T23).
8
+
9
+ The load-bearing part is the seam: `Extractor` is a real interface, and `run_extractors` iterates a
10
+ list of extractors and merges their results. Adding a new extractor (the deferred OpenIE path) is
11
+ just appending an `Extractor` to that list; neither `run_extractors` nor `ExtractionResult` is
12
+ reopened. Every extractor yields an `ExtractionResult` whose facts conform to the ontology (T4) and
13
+ are anchored to the originating `chunk_id` (FR-I.4).
14
+ """
15
+
16
+ from __future__ import annotations
17
+
18
+ from collections.abc import Iterable, Sequence
19
+ from typing import Protocol, runtime_checkable
20
+
21
+ from pydantic import BaseModel, field_validator, model_validator
22
+
23
+ from rag_wright.contracts.identifiers import ChunkId
24
+ from rag_wright.contracts.ontology import ClauseFact, RelationshipFact
25
+ from rag_wright.contracts.provenance import ConfidenceTag
26
+
27
+
28
+ class EntityMention(BaseModel):
29
+ """A pre-resolution entity mention: a surface form, its ontology type, and a confidence tag.
30
+
31
+ Graph extraction produces typed mentions (NER labels); entity resolution (FR-C.7 / T24) later
32
+ maps the surface form to a canonical `entity_id` and creates the canonical `EntityNode`.
33
+
34
+ A mention IS an ontology-conforming graph fact (FR-S.4): it is read from the text, so a spaCy NER
35
+ or contract-extracted mention carries a `confidence` tag like any other fact (ADR-0012). Its
36
+ `chunk_id` provenance is the containing `ExtractionResult.chunk_id` (mentions are anchored by the
37
+ result, not individually provenanced, since resolution collapses many mentions to one node). This
38
+ is deliberately how the spaCy path satisfies "each path produces facts carrying chunk_id +
39
+ confidence" — by emitting confidence-bearing mentions, NOT by inventing edges: co-occurrence of
40
+ two organizations in legal text is frequently non-contractual (a non-compete, a governing-law or
41
+ payment-clause reference), so a proximity edge is a false-edge generator, and nothing downstream
42
+ filters edges (T26 surfaces confidence, it does not gate on it — FR-C.5/FR-Q.3). CONTRACTS_WITH
43
+ comes from signing-party structure (the contract extractor), never proximity (ADR-0012).
44
+
45
+ `text` here is the *same notion* as a `RelationshipFact`'s `source_ref` / `target_ref`: both are
46
+ pre-resolution entity surface forms. Standalone mentions and relationship endpoints are two
47
+ channels for the same entities, so entity resolution (T24) must resolve them as one mention
48
+ stream; an entity appearing as both must resolve to a single node, not a duplicate.
49
+ """
50
+
51
+ text: str
52
+ entity_type: str # opaque domain entity type (DD-5); the caller/domain pack names it
53
+ confidence: ConfidenceTag
54
+
55
+ @field_validator("text")
56
+ @classmethod
57
+ def _text_non_empty(cls, v: str) -> str:
58
+ if not v.strip():
59
+ raise ValueError("entity mention text must be non-empty")
60
+ return v
61
+
62
+
63
+ class ExtractionResult(BaseModel):
64
+ """What one extractor produces for one chunk: ontology-conforming facts plus typed mentions.
65
+
66
+ Every fact is anchored to `chunk_id`: its provenance must point at this chunk, so the extraction
67
+ result carries the originating `chunk_id` end to end (FR-I.4). Facts already conform to the
68
+ ontology (their type fields are the T4 enums), so a non-ontology fact cannot be built at all.
69
+ """
70
+
71
+ chunk_id: ChunkId
72
+ entity_mentions: list[EntityMention] = []
73
+ clause_facts: list[ClauseFact] = []
74
+ relationship_facts: list[RelationshipFact] = []
75
+
76
+ @model_validator(mode="after")
77
+ def _facts_anchored_to_chunk(self) -> ExtractionResult:
78
+ for fact in (*self.clause_facts, *self.relationship_facts):
79
+ if fact.provenance.chunk_id != self.chunk_id:
80
+ raise ValueError(
81
+ "every fact in an ExtractionResult must be anchored to the result's chunk_id "
82
+ "(fact provenance chunk_id does not match)"
83
+ )
84
+ return self
85
+
86
+ @classmethod
87
+ def merge(cls, chunk_id: ChunkId, results: Sequence[ExtractionResult]) -> ExtractionResult:
88
+ """Merge several extractors' results for one chunk into a single result.
89
+
90
+ All results must be for `chunk_id`; a result for another chunk is a defect and is rejected.
91
+ Merging is a union (dedup, if any, is entity resolution's and graph storage's concern).
92
+ """
93
+ for result in results:
94
+ if result.chunk_id != chunk_id:
95
+ raise ValueError("cannot merge extraction results from different chunks")
96
+ return cls(
97
+ chunk_id=chunk_id,
98
+ entity_mentions=[m for r in results for m in r.entity_mentions],
99
+ clause_facts=[f for r in results for f in r.clause_facts],
100
+ relationship_facts=[f for r in results for f in r.relationship_facts],
101
+ )
102
+
103
+
104
+ @runtime_checkable
105
+ class Extractor(Protocol):
106
+ """The extractor seam. Each extractor in the hybrid stack (FR-C.6) implements this, and the
107
+ graph-extraction capability (T23) iterates over a list of them. The deferred OpenIE path is a
108
+ future `Extractor` added to that list, behind this same contract, with no change here.
109
+
110
+ Note: `@runtime_checkable` makes `isinstance(x, Extractor)` a *presence* check only (it verifies
111
+ `extract` and `name` exist, not their signatures or return type). Signature and output
112
+ conformance are enforced downstream by `ExtractionResult` validation, which is what the
113
+ result-validation tests exercise, not `isinstance`.
114
+ """
115
+
116
+ name: str
117
+
118
+ def extract(self, chunk_id: ChunkId, text: str) -> ExtractionResult: ...
119
+
120
+
121
+ def run_extractors(
122
+ extractors: Iterable[Extractor], chunk_id: ChunkId, text: str
123
+ ) -> ExtractionResult:
124
+ """Run every extractor over one chunk and merge into a single anchored `ExtractionResult`.
125
+
126
+ This is the seam the capability drives: registering a new extractor means adding it to
127
+ `extractors`, nothing here changes.
128
+ """
129
+ results = [extractor.extract(chunk_id, text) for extractor in extractors]
130
+ return ExtractionResult.merge(chunk_id, results)
@@ -0,0 +1,167 @@
1
+ """The retrieval FUNCTION taxonomy (T57, FR-C.6, ADR-0025).
2
+
3
+ The set of clause types the FUNCTION classifier (T56) routes on. It is a SUPERSET of the 41 CUAD
4
+ `ClauseCategory` (mirrored by value, so there is no drift) plus the three ACORD query families CUAD
5
+ has no class for: Indemnification (14/57 test queries), the indirect/consequential damages waiver,
6
+ and the warranty disclaimer (the bulk of the "Limitation of Liability" family beyond Cap/Uncapped).
7
+
8
+ This is kept DISTINCT from `ClauseCategory` on purpose. `ClauseCategory` is the graph EXTRACTION
9
+ ontology (ADR-0002): a closed vocabulary the knowledge graph conforms to, not reopened here. The
10
+ FUNCTION taxonomy is a RETRIEVAL concern (which clause type a span is routed under). The two serve
11
+ different layers even though 41 labels coincide by value, so extending function routing does not
12
+ reopen the extraction ontology.
13
+
14
+ `FUNCTION_LABELS` is the classifier's full label space and its retrain target (the T56 LegalBERT is
15
+ retrained over these classes plus its own NONE sentinel; NONE is not a function type and is not listed
16
+ here). ADR-0048 step 2 grew it 44 -> 52 with 8 taxonomy-gap functions the full-corpus classifier
17
+ surfaced (curated LLM-assisted with human oversight), plus a curated FOLD alias map (`_FUNCTION_ALIASES`)
18
+ that resolves recurring synonyms to their canonical label.
19
+ """
20
+
21
+ from __future__ import annotations
22
+
23
+ from enum import Enum
24
+
25
+ from pydantic import BaseModel, field_validator
26
+
27
+ from rag_wright.contracts.ontology import ClauseCategory
28
+
29
+
30
+ class FunctionClassification(BaseModel):
31
+ """CAP-REG-2: the ranked FUNCTION_LABELS a span or query is classified into (most relevant first).
32
+ The shared output contract of the LegalBERT `clause_function_classification` (model) and the
33
+ taxonomy-constrained `query_function_classification` (agent_skill)."""
34
+
35
+ labels: list[str]
36
+
37
+
38
+ class ExtendedFunction(str, Enum):
39
+ """The ACORD query families the 41 CUAD `ClauseCategory` has no class for (T57). Values are the
40
+ canonical function labels the classifier emits and the query-decomposer targets."""
41
+
42
+ INDEMNIFICATION = "Indemnification"
43
+ INDIRECT_DAMAGES_WAIVER = "Indirect/Consequential Damages Waiver"
44
+ WARRANTY_DISCLAIMER = "Warranty Disclaimer"
45
+
46
+
47
+ class TaxonomyGapFunction(str, Enum):
48
+ """INGEST-LLM-CLASSIFIER step 2 (ADR-0048): 8 clause functions the full-corpus LLM classifier surfaced as
49
+ recurring OTHER (out-of-taxonomy) clause types across the unified CUAD+ACORD corpus, then curated LLM-assisted
50
+ with human oversight (`data/eval/taxonomy_gaps/curation_proposal.json`, my recommended 8-ADD delta approved).
51
+ Genuinely distinct from the 44 CUAD/ACORD labels: CUAD has no generic Confidentiality, Royalties, Payment
52
+ Terms, Force Majeure, Dispute Resolution, Record Retention, Security Interest, or Condition Precedent class."""
53
+
54
+ CONFIDENTIALITY = "Confidentiality"
55
+ ROYALTIES = "Royalties"
56
+ PAYMENT_TERMS = "Payment Terms"
57
+ DISPUTE_RESOLUTION = "Dispute Resolution"
58
+ RECORD_RETENTION = "Record Retention"
59
+ SECURITY_INTEREST = "Security Interest"
60
+ CONDITION_PRECEDENT = "Condition Precedent"
61
+ FORCE_MAJEURE = "Force Majeure"
62
+
63
+
64
+ # The classifier's full label space: the 41 CUAD categories (by value, no drift), then the 3 ACORD
65
+ # extensions, then the 8 ADR-0048 taxonomy-gap additions. Order is stable (CUAD, extension, gap) so a
66
+ # retrain's label<->id map is reproducible. NONE is the off-taxonomy sentinel, not a function -> not listed.
67
+ FUNCTION_LABELS: tuple[str, ...] = tuple(
68
+ [c.value for c in ClauseCategory]
69
+ + [f.value for f in ExtendedFunction]
70
+ + [f.value for f in TaxonomyGapFunction]
71
+ )
72
+ FUNCTION_LABEL_SET: frozenset[str] = frozenset(FUNCTION_LABELS)
73
+
74
+ # The no-clause-function sentinel (the classifier's off-taxonomy NONE). NOT a function type, so NOT in
75
+ # FUNCTION_LABELS. A `ClausePropertyRecord` carries it ONLY for a QUERY-constraint record (a query has no clause
76
+ # function -- only its extracted properties matter); a real ingested clause never uses it (the ingest extracts
77
+ # clauses only for canonical functions).
78
+ NO_FUNCTION: str = "NONE"
79
+
80
+ # The classifier was trained on CUAD's label strings, which differ in CASE from the canonical taxonomy for
81
+ # a few labels (e.g. CUAD "Ip Ownership Assignment" vs the canonical "IP Ownership Assignment"). Normalize
82
+ # the classifier output to the canonical label at the boundary (memory: normalize at the boundary), keyed
83
+ # case-insensitively.
84
+ _FUNCTION_BY_CASEFOLD: dict[str, str] = {label.casefold(): label for label in FUNCTION_LABELS}
85
+
86
+ # ADR-0048 step 2: the curated FOLD map -- recurring synonyms/spelling/variant clause types the full-corpus
87
+ # classifier surfaced, each mapped to its canonical `FUNCTION_LABELS` entry (LLM-proposed, human-reconciled:
88
+ # the 3 LLM mis-folds were dropped, the royalty family retargeted to the new Royalties label, and RoFR +
89
+ # Milestone Payment folded rather than dropped). Authored in code (no external map to drift), keyed by
90
+ # canonical target for readability; inverted + casefolded into `_FUNCTION_ALIAS_BY_CASEFOLD` below.
91
+ _FUNCTION_ALIASES: dict[str, tuple[str, ...]] = {
92
+ "Cap On Liability": ("Limitation of Liability", "Limitations of Liability", "Liability Limitation"),
93
+ "Anti-Assignment": ("Assignment", "Assignment of Capacity", "Non-Assignment"),
94
+ "Termination For Convenience": (
95
+ "Termination", "Termination For Cause", "Effect of Termination", "Termination Clause",
96
+ "Cancellation", "Termination Effects", "Rights and Obligations Upon Termination"),
97
+ "Expiration Date": ("Term",),
98
+ "Insurance": ("Insurance Requirement", "Insurance Type", "Insurance Coverage Details"),
99
+ "No-Solicit Of Employees": ("Non-Solicit Of Employees",),
100
+ "No-Solicit Of Customers": ("Non-Solicit Of Customers",),
101
+ "Warranty Duration": ("Warranty Grant", "Product Warranty", "Warranty"),
102
+ "Warranty Disclaimer": (
103
+ "Disclaimer of Warranty", "Disclaimer of Representations and Warranties", "Disclaimer",
104
+ "Liability Disclaimer"),
105
+ "Liquidated Damages": ("Penalty", "Make-Whole Payment"),
106
+ "Indemnification": ("Intellectual Property Indemnification", "Indemnity and Limitation of Liability"),
107
+ "License Grant": (
108
+ "Content License Restrictions", "Use Restrictions", "License Grant Restrictions", "Restriction Of Use"),
109
+ "Notice Period To Terminate Renewal": ("Notice Period To Terminate",),
110
+ "Non-Compete": ("Non-Compete Exception",),
111
+ "Governing Law": ("Jurisdiction",),
112
+ "Audit Rights": ("Inspection", "Inspection Rights"),
113
+ "IP Ownership Assignment": (
114
+ "Ownership", "Domain Name Assignment", "Intellectual Property Rights", "Patents",
115
+ "Intellectual Property Assignment"),
116
+ "Revenue/Profit Sharing": ("Revenue Sharing", "Payment/Revenue Sharing"),
117
+ "Competitive Restriction Exception": (
118
+ "Definition of Class C Breaches", "Other Restriction Exception", "Other Restriction"),
119
+ "Rofr/Rofo/Rofn": ("Right of First Refusal",),
120
+ # the royalty family consolidates into the new Royalties label; Milestone Payment -> the new Payment Terms
121
+ "Royalties": ("Royalty Grant", "Royalty", "Royalty Obligation", "Royalty Payment"),
122
+ "Payment Terms": ("Milestone Payment",),
123
+ }
124
+ _FUNCTION_ALIAS_BY_CASEFOLD: dict[str, str] = {
125
+ alias.casefold(): canon for canon, aliases in _FUNCTION_ALIASES.items() for alias in aliases
126
+ }
127
+
128
+
129
+ def canonical_function(label: str) -> str | None:
130
+ """Map a function label to its canonical `FUNCTION_LABELS` entry, or None if it matches none. Case-
131
+ insensitive, and resolves the ADR-0048 curated FOLD aliases (e.g. 'Limitation of Liability' -> 'Cap On
132
+ Liability', 'Royalty Grant' -> 'Royalties'). Exact/cased match wins over an alias."""
133
+ key = label.strip().casefold()
134
+ return _FUNCTION_BY_CASEFOLD.get(key) or _FUNCTION_ALIAS_BY_CASEFOLD.get(key)
135
+
136
+
137
+ class FunctionConfidence(str, Enum):
138
+ """INGEST-LLM-CLASSIFIER (ADR-0048): the LLM clause classifier's coarse confidence in a function assignment.
139
+ Ordinal, not a float -- LLMs are not calibrated on numeric self-confidence; a floor (>= medium) filters weak
140
+ labels, so a clause with one clear function stays single while a genuinely mixed clause keeps 2-3."""
141
+
142
+ HIGH = "high"
143
+ MEDIUM = "medium"
144
+ LOW = "low"
145
+
146
+
147
+ class FunctionScore(BaseModel):
148
+ """INGEST-LLM-CLASSIFIER (ADR-0048): one function a clause is classified into, with coarse confidence. In a
149
+ ranked list the first is the PRIMARY (the label kept on `Clause.function` for the query legs). `function` is
150
+ normalized to its canonical `FUNCTION_LABELS` entry at the boundary; a non-canonical label (incl. the NONE
151
+ sentinel) is rejected (strict contract; normalize upstream)."""
152
+
153
+ function: str
154
+ confidence: FunctionConfidence
155
+
156
+ @field_validator("function")
157
+ @classmethod
158
+ def _canonicalize(cls, v: str) -> str:
159
+ canon = canonical_function(v)
160
+ if canon is None:
161
+ raise ValueError(f"function must be a canonical FUNCTION_LABELS label, got {v!r}")
162
+ return canon
163
+
164
+
165
+ def primary_function(scores: list[FunctionScore]) -> str | None:
166
+ """The PRIMARY (highest-ranked) function of a ranked `FunctionScore` list (primary first), or None if empty."""
167
+ return scores[0].function if scores else None
@@ -0,0 +1,91 @@
1
+ """KG-5e (FR-Q): the query-side FUNCTION ROUTER, driven by the granite-extracted DIMENSIONS.
2
+
3
+ KG-5c found that query-side function routing is the dominant retrieval lever (the oracle->real gap is
4
+ -0.22 recall@20, dwarfing every reranker lever) and that the clause-trained LegalBERT classifier is
5
+ out-of-distribution on short query text. This routes instead from the query's typed DIMENSIONS -- which
6
+ KG-5b showed the granite extraction gets right even where `clause_type` is unreliable -- through a
7
+ corpus-derived `dimension -> function` co-occurrence prior. No extra LLM call, no clause-trained model:
8
+ the typed extraction we already compute builds the pool.
9
+
10
+ The prior is built from a HELD-OUT corpus (CUAD) so no ACORD eval data enters the router; the two corpora
11
+ share the property-dimension schema and the CUAD-type function taxonomy, so the prior transfers.
12
+ """
13
+
14
+ from __future__ import annotations
15
+
16
+ import math
17
+ from collections import defaultdict
18
+ from typing import Iterable
19
+
20
+
21
+ def build_cooccurrence(clause_rows: Iterable[tuple[str, str, str]]) -> dict[str, dict[str, int]]:
22
+ """`dimension -> {function -> clause-count}` from `(clause_id, function, dimension)` typed-edge rows.
23
+
24
+ Counted once per (clause, dimension, function) so a clause with several values for one dimension does not
25
+ over-weight its function. An empty/`NONE` function is dropped (not a routable target). Pure -- no store.
26
+ """
27
+ seen: set[tuple[str, str, str]] = set()
28
+ cooc: dict[str, dict[str, int]] = defaultdict(lambda: defaultdict(int))
29
+ for clause_id, function, dimension in clause_rows:
30
+ if not function or function == "NONE" or not dimension:
31
+ continue
32
+ key = (clause_id, dimension, function)
33
+ if key in seen:
34
+ continue
35
+ seen.add(key)
36
+ cooc[dimension][function] += 1
37
+ return {d: dict(fs) for d, fs in cooc.items()}
38
+
39
+
40
+ def _function_marginal(cooc: dict[str, dict[str, int]]) -> dict[str, float]:
41
+ """P(function) proxy: each function's share of all edges in the prior -- the base rate that lift/PMI
42
+ divide out so a distinctive dimension beats a merely-common function."""
43
+ tot: dict[str, float] = defaultdict(float)
44
+ grand = 0.0
45
+ for col in cooc.values():
46
+ for fn, c in col.items():
47
+ tot[fn] += c
48
+ grand += c
49
+ return {fn: c / grand for fn, c in tot.items()} if grand else {}
50
+
51
+
52
+ def route_functions(
53
+ query_dimensions: Iterable[str],
54
+ cooc: dict[str, dict[str, int]],
55
+ *,
56
+ k: int,
57
+ score: str = "conditional",
58
+ min_support: int = 1,
59
+ ) -> list[str]:
60
+ """The top-`k` functions for a query, scored by summing a per-dimension signal over the query's dimensions:
61
+
62
+ - ``conditional`` (default): ``P(function | dimension)`` -- biased toward high-frequency functions.
63
+ - ``lift``: ``P(function|dimension) / P(function)`` -- corrects for the function base rate (a dimension
64
+ routes to the function it is DISTINCTIVE of, not merely the most common one that has it).
65
+ - ``pmi``: ``log(P(function|dimension) / P(function))`` -- the log-odds form of lift.
66
+
67
+ `min_support` drops sparse `(dimension, function)` evidence (< that many clauses) so lift/PMI are not blown
68
+ up by a single-clause coincidence. A dimension with no corpus signal contributes nothing; no dimensions
69
+ (or none seen) -> ``[]``. Ties break on the function name so the routing is deterministic.
70
+ """
71
+ marginal = _function_marginal(cooc) if score in ("lift", "pmi") else {}
72
+ total_score: dict[str, float] = defaultdict(float)
73
+ for d in query_dimensions:
74
+ col = cooc.get(d)
75
+ if not col:
76
+ continue
77
+ total = sum(col.values())
78
+ if not total:
79
+ continue
80
+ for fn, c in col.items():
81
+ if c < min_support:
82
+ continue
83
+ p_f_given_d = c / total
84
+ if score == "conditional":
85
+ total_score[fn] += p_f_given_d
86
+ else:
87
+ p_f = marginal.get(fn, 0.0)
88
+ if p_f <= 0:
89
+ continue
90
+ total_score[fn] += p_f_given_d / p_f if score == "lift" else math.log(p_f_given_d / p_f)
91
+ return [fn for fn, _ in sorted(total_score.items(), key=lambda x: (-x[1], x[0]))[:k]]
@@ -0,0 +1,74 @@
1
+ """The serve-side highlight response contract (CU-A1 / CU-C2, ADR-0029).
2
+
3
+ What the pipeline returns for a query about a known contract: the SET of spans that pertain (possibly empty ->
4
+ "not present"), each with its exact document location so an app can highlight it, plus provenance for citation.
5
+ `HighlightResult` also carries the presence/absence and out-of-taxonomy/low-confidence signals the app/agent
6
+ routes on. Offsets mirror `SpanRecord` (document-absolute char offsets; optional page/bbox).
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ from pydantic import BaseModel, model_validator
12
+
13
+
14
+ class HighlightSpan(BaseModel):
15
+ """One span to highlight, with its citation (location + parent-clause reference)."""
16
+
17
+ model_config = {"frozen": True}
18
+
19
+ span_id: str
20
+ contract_id: str
21
+ function: str # the clause type this span was matched under
22
+ clause_ref: str # human-readable parent-clause reference (heading/number or parent_chunk_id)
23
+ text: str
24
+ doc_start: int | None = None # document-absolute char offsets (the highlight range)
25
+ doc_end: int | None = None
26
+ page: int | None = None # optional PDF-overlay location (FIRST page; == pages[0] when known)
27
+ pages: list[int] = [] # issue 0032: ALL source pages this span overlaps (page-level click-through)
28
+ bbox: tuple[float, float, float, float] | None = None
29
+ extracted_value: str | None = None # for value-type categories: the pinpointed value within the span
30
+ confidence: float = 1.0
31
+
32
+ @model_validator(mode="after")
33
+ def _check_offsets(self) -> "HighlightSpan":
34
+ if self.doc_start is not None and self.doc_end is not None and self.doc_end < self.doc_start:
35
+ raise ValueError(f"doc_end ({self.doc_end}) must be >= doc_start ({self.doc_start})")
36
+ return self
37
+
38
+
39
+ class SpanLocation(BaseModel):
40
+ """Where one span sits in the ORIGINAL document, for a citation PREVIEW (EP-REF-1b): a location + the text
41
+ to confirm it landed, plus the clause ids extracted from it (so an answer's `citation_id` -- a span id OR a
42
+ clause id, two different spaces -- resolves either way). Lighter than `HighlightSpan` (no function / clause
43
+ ref / extracted value / confidence): a preview answers "show me this span", not "which spans pertain". `pages`
44
+ is a LIST (a span can cross a page break; the first page is where the preview opens) and `bbox` is best-effort
45
+ -- a page is nearly always known and a rectangle usually is, so a missing rectangle never costs the page."""
46
+
47
+ span_id: str
48
+ clause_ids: list[str] = []
49
+ pages: list[int] = []
50
+ bbox: tuple[float, float, float, float] | None = None
51
+ doc_start: int | None = None
52
+ doc_end: int | None = None
53
+ text: str = ""
54
+
55
+
56
+ class HighlightResult(BaseModel):
57
+ """The full response to one query about one contract."""
58
+
59
+ model_config = {"frozen": True}
60
+
61
+ query: str
62
+ contract_id: str
63
+ clause_types: list[str] = [] # the types searched (from QueryIntent)
64
+ intent: str = "highlight"
65
+ spans: list[HighlightSpan] = [] # the pertaining set; empty => not present
66
+ present: bool = False # spans is non-empty
67
+ in_taxonomy: bool = True
68
+ low_confidence: bool = False # out-of-taxonomy semantic fallback used
69
+
70
+ @model_validator(mode="after")
71
+ def _check_present(self) -> "HighlightResult":
72
+ if self.present != bool(self.spans):
73
+ raise ValueError("`present` must equal whether `spans` is non-empty")
74
+ return self
@@ -0,0 +1,153 @@
1
+ """Shared identifier contracts: `chunk_id` (FR-S.2) and `entity_id` (FR-S.3).
2
+
3
+ These schemes are load-bearing and fixed here before anything is built. A re-chunk that changes
4
+ a `chunk_id` breaks the link between a chunk and its extracted graph nodes, and a non-canonical
5
+ `entity_id` fragments the graph across surface-form variants. Changing either scheme is an
6
+ ask-first change (SPEC.md section 14; CLAUDE.md boundaries).
7
+
8
+ Both identifiers are frozen Pydantic models, so they are immutable and hashable and can serve as
9
+ dictionary keys and graph-node identity directly.
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ import hashlib
15
+ import re
16
+
17
+ from pydantic import BaseModel, ConfigDict, field_validator
18
+
19
+ _SHA256_HEX = re.compile(r"^[0-9a-f]{64}$")
20
+ _SOURCE_DOC_ID = re.compile(r"^[A-Za-z0-9._-]+$")
21
+ _SOURCE_DOC_UNSAFE = re.compile(r"[^A-Za-z0-9._-]+")
22
+
23
+
24
+ def canonical_source_doc_id(raw: str) -> str:
25
+ """The ONE canonical filename/title -> `source_doc_id` slug (FR-S.2; HYG-1).
26
+
27
+ Every ingestion path MUST derive a `source_doc_id` through this function so the same document gets the
28
+ same id everywhere. Any run of characters outside the delimiter-safe set ``[A-Za-z0-9._-]`` (notably
29
+ spaces, ``&``, commas) collapses to a single ``_``; leading/trailing ``_`` are stripped. Existing safe
30
+ delimiters (``-``, ``.``, ``_`` -- e.g. inside ``EX-10.1`` / ``10-Q``) are preserved. Idempotent on an
31
+ already-canonical id.
32
+
33
+ The ``_`` replacement (never ``-``) is the fix for the HYG-1 divergence: two ingestion paths slugged the
34
+ same title with different characters (``FLEET_MAINTENANCE`` vs ``FLEET-MAINTENANCE``), breaking the
35
+ cross-graph join. An empty result raises rather than silently colliding every empty title into one id.
36
+ """
37
+ slug = _SOURCE_DOC_UNSAFE.sub("_", raw).strip("_") if isinstance(raw, str) else ""
38
+ if not slug:
39
+ raise ValueError(
40
+ f"source_doc_id slug is empty for {raw!r}; supply a non-empty, sluggable document id"
41
+ )
42
+ return slug
43
+
44
+
45
+ class ChunkId(BaseModel):
46
+ """The stable identifier for a chunk (FR-S.2).
47
+
48
+ Scheme: source-document identifier, chunk index, and a content hash of the chunk text. The
49
+ content hash is what makes the identifier change when (and only when) the chunk content
50
+ changes, so an unchanged document re-chunks to the same ids (the content-hash gate in FR-I.1
51
+ and FR-I.5 relies on this).
52
+ """
53
+
54
+ model_config = ConfigDict(frozen=True)
55
+
56
+ source_doc_id: str
57
+ chunk_index: int
58
+ content_hash: str # lowercase hex SHA-256 digest of the chunk content
59
+
60
+ @field_validator("source_doc_id")
61
+ @classmethod
62
+ def _delimiter_safe_source(cls, v: str) -> str:
63
+ v = v.strip()
64
+ if not _SOURCE_DOC_ID.match(v):
65
+ raise ValueError(
66
+ "source_doc_id must be non-empty and use only [A-Za-z0-9._-], so the ':'-delimited "
67
+ "value string stays unambiguous for provenance and citation lookup; assign a "
68
+ "delimiter-safe id upstream (slugify the filename if needed)"
69
+ )
70
+ return v
71
+
72
+ @field_validator("chunk_index")
73
+ @classmethod
74
+ def _nonnegative_index(cls, v: int) -> int:
75
+ if v < 0:
76
+ raise ValueError("chunk_index must be >= 0")
77
+ return v
78
+
79
+ @field_validator("content_hash")
80
+ @classmethod
81
+ def _valid_sha256(cls, v: str) -> str:
82
+ v = v.strip().lower()
83
+ if not _SHA256_HEX.match(v):
84
+ raise ValueError("content_hash must be a 64-character lowercase hex SHA-256 digest")
85
+ return v
86
+
87
+ @classmethod
88
+ def of(cls, source_doc_id: str, chunk_index: int, content: str | bytes) -> ChunkId:
89
+ """Build a `ChunkId`, computing the content hash deterministically (SHA-256).
90
+
91
+ This is where determinism lives (RAC-1): identical `(source_doc_id, chunk_index, content)`
92
+ always yield an identical `ChunkId`.
93
+ """
94
+ data = content.encode("utf-8") if isinstance(content, str) else content
95
+ return cls(
96
+ source_doc_id=source_doc_id,
97
+ chunk_index=chunk_index,
98
+ content_hash=hashlib.sha256(data).hexdigest(),
99
+ )
100
+
101
+ @property
102
+ def value(self) -> str:
103
+ """The canonical string form: ``<source_doc_id>:<chunk_index>:<content_hash>``.
104
+
105
+ Because ``source_doc_id`` is constrained to a delimiter-safe character set, this string is
106
+ safe to string-match and to parse back with ``rsplit(":", 2)`` for provenance and citation
107
+ lookup. Identity itself remains field-based (frozen-model equality and hashing), not
108
+ string-based.
109
+ """
110
+ return f"{self.source_doc_id}:{self.chunk_index}:{self.content_hash}"
111
+
112
+ def __str__(self) -> str:
113
+ return self.value
114
+
115
+
116
+ class EntityId(BaseModel):
117
+ """The canonical entity identifier (FR-S.3): an opaque canonical-registry id string.
118
+
119
+ The engine is domain-agnostic (DD-4, ADR-0067/0117), so the FORMAT of a canonical id is owned by
120
+ the resolver / domain pack, NOT by this contract. The SEC pack resolves to a 10-digit zero-padded
121
+ EDGAR Central Index Key (CIK), e.g. ``"0000320193"`` (shaped in ``corpus/edgar.normalize_cik``); a
122
+ generic pack uses an exact-normalized surface-form key; another domain uses its own scheme. This
123
+ contract's only invariant is therefore the domain-neutral one: a non-empty string. That is still a
124
+ real invariant -- every downstream holder of an `EntityId` can trust it is a present, non-blank id --
125
+ while the format check lives at the one boundary that knows the domain (the resolver/loader), where
126
+ the world's mess actually arrives, per the "normalize at the boundary" rule.
127
+ """
128
+
129
+ model_config = ConfigDict(frozen=True)
130
+
131
+ value: str # an opaque canonical id; its FORMAT is the resolver/pack's concern, not this contract's
132
+
133
+ @field_validator("value", mode="before")
134
+ @classmethod
135
+ def _nonempty(cls, v: object) -> str:
136
+ if not isinstance(v, str) or not v.strip():
137
+ raise ValueError(
138
+ "EntityId.value must be a non-empty string. The canonical-id FORMAT is owned by the "
139
+ "resolver / domain pack (e.g. corpus/edgar.normalize_cik for SEC CIKs), not this contract."
140
+ )
141
+ return v
142
+
143
+ @classmethod
144
+ def of(cls, value: str) -> EntityId:
145
+ """Build an `EntityId` from an already-canonical id string.
146
+
147
+ This does not normalize or format-check beyond non-emptiness; shaping the raw domain form into
148
+ the canonical id is the resolver / domain pack's job (e.g. `corpus/edgar.normalize_cik`).
149
+ """
150
+ return cls(value=value)
151
+
152
+ def __str__(self) -> str:
153
+ return self.value