rag-wright 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (184) hide show
  1. rag_wright/__init__.py +13 -0
  2. rag_wright/api/__init__.py +33 -0
  3. rag_wright/api/config.py +59 -0
  4. rag_wright/api/discover.py +70 -0
  5. rag_wright/api/documents.py +39 -0
  6. rag_wright/api/ids.py +31 -0
  7. rag_wright/api/invoke.py +99 -0
  8. rag_wright/api/kg.py +61 -0
  9. rag_wright/api/mcp.py +94 -0
  10. rag_wright/api/usage.py +30 -0
  11. rag_wright/api/workspace.py +85 -0
  12. rag_wright/capabilities/__init__.py +8 -0
  13. rag_wright/capabilities/answer_generator.py +427 -0
  14. rag_wright/capabilities/ard.py +286 -0
  15. rag_wright/capabilities/assertion_extraction.py +79 -0
  16. rag_wright/capabilities/chunk_read.py +58 -0
  17. rag_wright/capabilities/chunk_write.py +163 -0
  18. rag_wright/capabilities/claim_extraction.py +153 -0
  19. rag_wright/capabilities/clause_exception_linking.py +117 -0
  20. rag_wright/capabilities/compliance_judgment.py +322 -0
  21. rag_wright/capabilities/compliance_store.py +87 -0
  22. rag_wright/capabilities/contract_kg_serve.py +156 -0
  23. rag_wright/capabilities/contract_kg_store.py +251 -0
  24. rag_wright/capabilities/dg_extraction.py +585 -0
  25. rag_wright/capabilities/disambiguation.py +163 -0
  26. rag_wright/capabilities/document_parse.py +87 -0
  27. rag_wright/capabilities/document_scope.py +49 -0
  28. rag_wright/capabilities/embedding.py +164 -0
  29. rag_wright/capabilities/embedding_profiles.py +43 -0
  30. rag_wright/capabilities/entity_resolution.py +154 -0
  31. rag_wright/capabilities/fusion.py +64 -0
  32. rag_wright/capabilities/graph_extraction.py +243 -0
  33. rag_wright/capabilities/graph_query.py +73 -0
  34. rag_wright/capabilities/graph_storage.py +111 -0
  35. rag_wright/capabilities/highlight_serve.py +142 -0
  36. rag_wright/capabilities/hybrid_search.py +65 -0
  37. rag_wright/capabilities/invoke.py +31 -0
  38. rag_wright/capabilities/jev_decision.py +38 -0
  39. rag_wright/capabilities/manifests.py +872 -0
  40. rag_wright/capabilities/okf_navigate.py +456 -0
  41. rag_wright/capabilities/parsing.py +286 -0
  42. rag_wright/capabilities/property_boosted_retrieval.py +125 -0
  43. rag_wright/capabilities/query_function_classifier.py +94 -0
  44. rag_wright/capabilities/query_understanding.py +109 -0
  45. rag_wright/capabilities/registry.py +262 -0
  46. rag_wright/capabilities/remote_encoders.py +94 -0
  47. rag_wright/capabilities/requirement_extraction.py +247 -0
  48. rag_wright/capabilities/reranking.py +123 -0
  49. rag_wright/capabilities/retrieval_core.py +126 -0
  50. rag_wright/capabilities/rlm_chunking.py +808 -0
  51. rag_wright/capabilities/rlm_synthesis.py +316 -0
  52. rag_wright/capabilities/scan_quality.py +136 -0
  53. rag_wright/capabilities/span_relevance_judgment.py +191 -0
  54. rag_wright/capabilities/vision_to_text.py +85 -0
  55. rag_wright/capabilities/vlm_ocr.py +85 -0
  56. rag_wright/contracts/__init__.py +6 -0
  57. rag_wright/contracts/chunk.py +79 -0
  58. rag_wright/contracts/compliance.py +303 -0
  59. rag_wright/contracts/contract_meta.py +27 -0
  60. rag_wright/contracts/extraction.py +130 -0
  61. rag_wright/contracts/function.py +167 -0
  62. rag_wright/contracts/function_routing.py +91 -0
  63. rag_wright/contracts/highlight.py +74 -0
  64. rag_wright/contracts/identifiers.py +153 -0
  65. rag_wright/contracts/jurisdiction.py +96 -0
  66. rag_wright/contracts/ontology.py +142 -0
  67. rag_wright/contracts/property.py +201 -0
  68. rag_wright/contracts/provenance.py +78 -0
  69. rag_wright/contracts/query_intent.py +53 -0
  70. rag_wright/contracts/span.py +76 -0
  71. rag_wright/contracts/value_match.py +84 -0
  72. rag_wright/corpus/__init__.py +0 -0
  73. rag_wright/corpus/canonicalize.py +116 -0
  74. rag_wright/corpus/cuad.py +153 -0
  75. rag_wright/corpus/cuad_ingestion.py +72 -0
  76. rag_wright/corpus/document_parser.py +299 -0
  77. rag_wright/corpus/edgar.py +231 -0
  78. rag_wright/corpus/gcs_ingestion.py +120 -0
  79. rag_wright/corpus/http.py +110 -0
  80. rag_wright/corpus/selection.py +152 -0
  81. rag_wright/mcp/__init__.py +11 -0
  82. rag_wright/mcp/compliance_server.py +299 -0
  83. rag_wright/mcp/intra_document_qa_server.py +170 -0
  84. rag_wright/mcp/relational_qa_server.py +171 -0
  85. rag_wright/mcp/session_store.py +64 -0
  86. rag_wright/mcp/typed_property_retrieval_server.py +191 -0
  87. rag_wright/models/__init__.py +8 -0
  88. rag_wright/models/profiles.py +331 -0
  89. rag_wright/models/seam.py +497 -0
  90. rag_wright/models/tag_structured.py +285 -0
  91. rag_wright/models/tracing.py +179 -0
  92. rag_wright/models/usage.py +102 -0
  93. rag_wright/okf/__init__.py +11 -0
  94. rag_wright/okf/compile.py +292 -0
  95. rag_wright/okf/document.py +47 -0
  96. rag_wright/okf/enrich.py +176 -0
  97. rag_wright/okf/links.py +190 -0
  98. rag_wright/okf/lint.py +105 -0
  99. rag_wright/ontology/__init__.py +6 -0
  100. rag_wright/ontology/_generated_template_meta.py +60 -0
  101. rag_wright/ontology/_generated_vocab.py +52 -0
  102. rag_wright/ontology/clause_template.py +964 -0
  103. rag_wright/ontology/codegen.py +84 -0
  104. rag_wright/ontology/compliance_bridge.ttl +186 -0
  105. rag_wright/ontology/contract_bridge.ttl +2685 -0
  106. rag_wright/ontology/contract_taxonomy.py +24 -0
  107. rag_wright/ontology/derive.py +58 -0
  108. rag_wright/ontology/loader.py +435 -0
  109. rag_wright/ontology/packs/ftc_16cfr255.ttl +29 -0
  110. rag_wright/ontology/registry.py +87 -0
  111. rag_wright/ontology/template_introspect.py +100 -0
  112. rag_wright/py.typed +0 -0
  113. rag_wright/reference/__init__.py +2 -0
  114. rag_wright/reference/compliance.py +41 -0
  115. rag_wright/reference/contract_seam.py +123 -0
  116. rag_wright/skills/__init__.py +7 -0
  117. rag_wright/skills/claim_extraction/SKILL.md +47 -0
  118. rag_wright/skills/claim_extraction/__init__.py +1 -0
  119. rag_wright/skills/claim_extraction/template.py +50 -0
  120. rag_wright/skills/compliance_judgment/SKILL.md +59 -0
  121. rag_wright/skills/corpus_ingest/SKILL.md +106 -0
  122. rag_wright/skills/extraction_semantic_judge/SKILL.md +51 -0
  123. rag_wright/skills/extraction_semantic_judge/__init__.py +1 -0
  124. rag_wright/skills/generation/SKILL.md +64 -0
  125. rag_wright/skills/generation/__init__.py +1 -0
  126. rag_wright/skills/generic_compliance_judgment/SKILL.md +58 -0
  127. rag_wright/skills/okf_navigate/SKILL.md +137 -0
  128. rag_wright/skills/requirement_extraction/SKILL.md +47 -0
  129. rag_wright/skills/requirement_extraction/__init__.py +1 -0
  130. rag_wright/skills/requirement_extraction/template.py +50 -0
  131. rag_wright/skills/rlm/SKILL.md +186 -0
  132. rag_wright/skills/rlm/__init__.py +31 -0
  133. rag_wright/skills/rlm/agent.py +292 -0
  134. rag_wright/skills/span_relevance_judgment/SKILL.md +67 -0
  135. rag_wright/skills/vision_to_text/SKILL.md +36 -0
  136. rag_wright/skills/vision_to_text/__init__.py +1 -0
  137. rag_wright/spans/__init__.py +1 -0
  138. rag_wright/spans/boundary.py +78 -0
  139. rag_wright/spans/clause_function_classifier.py +490 -0
  140. rag_wright/spans/clause_kg_extractor.py +337 -0
  141. rag_wright/spans/cuad_labels.py +81 -0
  142. rag_wright/spans/dim_classifier.py +158 -0
  143. rag_wright/spans/dim_fleet.json +411 -0
  144. rag_wright/spans/function_classifier.py +77 -0
  145. rag_wright/spans/function_families.py +62 -0
  146. rag_wright/spans/hybrid_classifier.py +103 -0
  147. rag_wright/spans/legalbert_classifier.py +83 -0
  148. rag_wright/spans/model_capabilities.py +107 -0
  149. rag_wright/spans/new_function_labels.py +111 -0
  150. rag_wright/spans/page_map.py +68 -0
  151. rag_wright/spans/property_extractor.py +365 -0
  152. rag_wright/spans/property_grounding.py +182 -0
  153. rag_wright/spans/reclassify.py +77 -0
  154. rag_wright/spans/scarce_function_labels.py +105 -0
  155. rag_wright/spans/segment.py +341 -0
  156. rag_wright/spans/semantic_judge.py +197 -0
  157. rag_wright/spans/symbolic_validation.py +131 -0
  158. rag_wright/spans/tag_clause_extractor.py +182 -0
  159. rag_wright/store/__init__.py +6 -0
  160. rag_wright/store/arcadedb.py +1135 -0
  161. rag_wright/store/chunk_text.py +66 -0
  162. rag_wright/store/seam.py +213 -0
  163. rag_wright/subgraphs/__init__.py +0 -0
  164. rag_wright/subgraphs/async_ingestion.py +204 -0
  165. rag_wright/subgraphs/compliance_check.py +1042 -0
  166. rag_wright/subgraphs/compliance_ingestion.py +306 -0
  167. rag_wright/subgraphs/contract_ingestion_pipeline.py +999 -0
  168. rag_wright/subgraphs/graph_extraction.py +102 -0
  169. rag_wright/subgraphs/intra_document_qa.py +328 -0
  170. rag_wright/subgraphs/observability.py +140 -0
  171. rag_wright/subgraphs/query_constraint_extraction.py +73 -0
  172. rag_wright/subgraphs/relational_qa.py +165 -0
  173. rag_wright/subgraphs/requirement_extraction.py +137 -0
  174. rag_wright/subgraphs/scaffold.py +65 -0
  175. rag_wright/subgraphs/semantic_chunking.py +183 -0
  176. rag_wright/subgraphs/typed_clause_extraction.py +172 -0
  177. rag_wright/subgraphs/typed_property_retrieval.py +278 -0
  178. rag_wright/util/__init__.py +1 -0
  179. rag_wright/util/concurrent.py +153 -0
  180. rag_wright/util/spacy_model.py +45 -0
  181. rag_wright-0.1.0.dist-info/METADATA +168 -0
  182. rag_wright-0.1.0.dist-info/RECORD +184 -0
  183. rag_wright-0.1.0.dist-info/WHEEL +4 -0
  184. rag_wright-0.1.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,96 @@
1
+ """KG-5a: deterministic jurisdiction canonicalization for the contract KG.
2
+
3
+ The typed KG stores `jurisdiction` as the extracted surface string (`England`, `England and Wales`,
4
+ `English law`, `State of New York`, `State of New York, USA`, ...). Retrieval matching (Leg B) fails on
5
+ those variants because the query side and the clause side don't share a normalized value. This maps a
6
+ surface form to a **canonical jurisdiction slug** (or None for non-jurisdictions), deterministically -- no
7
+ LLM, no network. Applied additively: the value node keeps its surface `value` and gains a `canonical_value`;
8
+ the query constraint is canonicalized the same way, so both meet on the canonical.
9
+
10
+ Method: strip governance boilerplate prefixes/suffixes (`State of`, `Commonwealth of`, `, USA`, ` law`,
11
+ ` courts`) and normalize, then look up a gazetteer (50 US states + the countries seen in the corpus, each
12
+ with aliases). Non-jurisdictions (`Applicable Law`, `Not specified`, `worldwide`, redactions, compound
13
+ `Illinois or New York`) resolve to None (left as their surface value, unmatched).
14
+ """
15
+
16
+ from __future__ import annotations
17
+
18
+ import re
19
+
20
+ # canonical slug -> alias surface forms (already normalized: lowercase, no prefixes/suffixes)
21
+ _US_STATES = [
22
+ "alabama", "alaska", "arizona", "arkansas", "california", "colorado", "connecticut", "delaware",
23
+ "florida", "georgia", "hawaii", "idaho", "illinois", "indiana", "iowa", "kansas", "kentucky",
24
+ "louisiana", "maine", "maryland", "massachusetts", "michigan", "minnesota", "mississippi", "missouri",
25
+ "montana", "nebraska", "nevada", "new hampshire", "new jersey", "new mexico", "new york",
26
+ "north carolina", "north dakota", "ohio", "oklahoma", "oregon", "pennsylvania", "rhode island",
27
+ "south carolina", "south dakota", "tennessee", "texas", "utah", "vermont", "virginia", "washington",
28
+ "west virginia", "wisconsin", "wyoming",
29
+ ]
30
+
31
+ # country / non-US aliases -> canonical slug (normalized)
32
+ _COUNTRY_ALIASES: dict[str, str] = {
33
+ "england": "england", "english": "england", "england and wales": "england",
34
+ "united kingdom": "england", "uk": "england", "great britain": "england", "britain": "england",
35
+ "scotland": "scotland", "wales": "wales", "northern ireland": "northern_ireland",
36
+ "china": "china", "prc": "china", "people's republic of china": "china",
37
+ "united states": "united_states", "united states of america": "united_states", "usa": "united_states",
38
+ "u.s.a.": "united_states", "us": "united_states",
39
+ "canada": "canada", "british columbia": "british_columbia", "ontario": "ontario",
40
+ "belgium": "belgium", "germany": "germany", "france": "france", "japan": "japan", "spain": "spain",
41
+ "italy": "italy", "italian": "italy", "south africa": "south_africa", "israel": "israel",
42
+ "taiwan": "taiwan", "netherlands": "netherlands", "switzerland": "switzerland", "australia": "australia",
43
+ "singapore": "singapore", "hong kong": "hong_kong", "ireland": "ireland",
44
+ }
45
+
46
+ # build the normalized-alias -> canonical index
47
+ _INDEX: dict[str, str] = {s: s.replace(" ", "_") for s in _US_STATES}
48
+ _INDEX.update(_COUNTRY_ALIASES)
49
+
50
+ # surfaces that are NOT a jurisdiction (governance boilerplate / nulls / references / vague) -> None
51
+ _JUNK = re.compile(
52
+ r"applicable\s+law|governing\s+law|not\s+specified|not\s+explicitly|unspecified|none\s+specified"
53
+ r"|^none$|^other$|best's|exhibit|section\b|bankruptcy\s+code|any\s+jurisdiction|jurisdiction\s+governing"
54
+ r"|franchised\s+restaurant|^territory$|^union$|^worldwide$|\*|\[|last\s+sentence|internal\s+laws",
55
+ re.IGNORECASE,
56
+ )
57
+ _PREFIXES = ("the state of ", "state of ", "the commonwealth of ", "commonwealth of ",
58
+ "the province of ", "province of ", "the ")
59
+ _SUFFIXES = (", u.s.a.", ", usa", ", united states of america", ", united states", " (u.s.a.)",
60
+ " (usa)", " and its territories")
61
+
62
+
63
+ def _normalize(surface: str) -> str:
64
+ s = " ".join(surface.lower().strip().split())
65
+ for p in _PREFIXES:
66
+ if s.startswith(p):
67
+ s = s[len(p):]
68
+ break
69
+ for suf in _SUFFIXES:
70
+ s = s.replace(suf, "")
71
+ s = re.sub(r"\s+(law|laws|courts|court|state)$", "", s).strip()
72
+ return s
73
+
74
+
75
+ # longest alias first, for the containment fallback (word-bounded)
76
+ _ALIAS_ITEMS = sorted(_INDEX.items(), key=lambda kv: -len(kv[0]))
77
+
78
+
79
+ def _containment(norm: str) -> str | None:
80
+ """Fallback for surfaces the exact lookup misses: scan for known jurisdiction aliases as whole words.
81
+ Resolve only if it names exactly one place -- treating a US state named alongside 'the United States'
82
+ (federal) as that state ('New York and ... the United States of America' -> new_york). Genuinely
83
+ ambiguous compounds ('Illinois or New York') stay None."""
84
+ found = {canon for alias, canon in _ALIAS_ITEMS if re.search(rf"\b{re.escape(alias)}\b", norm)}
85
+ if len(found) > 1:
86
+ found.discard("united_states") # a state + US federal -> the state
87
+ return next(iter(found)) if len(found) == 1 else None
88
+
89
+
90
+ def canonicalize_jurisdiction(surface: str) -> str | None:
91
+ """Map a jurisdiction surface form to a canonical slug (e.g. 'england', 'new_york'), or None if it is
92
+ not a resolvable single jurisdiction (boilerplate, null, reference, ambiguous compound). Deterministic."""
93
+ if not surface or _JUNK.search(surface):
94
+ return None
95
+ norm = _normalize(surface)
96
+ return _INDEX.get(norm) or _containment(norm)
@@ -0,0 +1,142 @@
1
+ """The ontology and extraction-target models (FR-C.8, §16.2, ADR-0002).
2
+
3
+ The ontology is the closed vocabulary the knowledge graph conforms to. T4 owns the ontology
4
+ *structure* (the type enums and the fact models that reference them); T8 owns the ontology
5
+ *membership* (deriving the concrete types from the Data Catalog, FR-C.8). Both are real: deferring
6
+ membership entirely would leave the downstream contract and extraction work with nothing to bind.
7
+
8
+ - `ClauseCategory`: the 41 CUAD clause categories. Membership here is authoritative (ADR-0002). The
9
+ values are the canonical CUAD label names; T8 reconciles them against the exact label strings in
10
+ the CUAD data once the corpus is acquired (T7).
11
+ - Entity/relationship taxonomy (DD-5, ADR-0066/0117): the party/entity node types and the entity-to-entity
12
+ relationship (edge) types are NO LONGER a hardcoded engine enum. `EntityNode.entity_type` and
13
+ `RelationshipFact.relationship_type` are OPAQUE domain strings the caller names; the closed value sets are
14
+ DOMAIN knowledge declared by the pack (the reference contract pack's are in `ontology/contract_taxonomy.py`).
15
+ A new domain supplies its own without reopening these contracts.
16
+
17
+ The extraction-target models (`EntityNode`, `ClauseFact`, `RelationshipFact`) are what graph
18
+ extraction (T5) produces and graph storage (T24) writes. Their type fields are the ontology enums,
19
+ so a fact whose type is not in the ontology is rejected at construction (RAC-4). The fact models
20
+ extend `GraphFact` (T2), so they carry provenance and a confidence tag; `EntityNode` is a canonical
21
+ node (identifier, name, type, no facts), per the thin entity skeleton in SPEC.md section 8.
22
+ """
23
+
24
+ from __future__ import annotations
25
+
26
+ from enum import Enum
27
+
28
+ from pydantic import BaseModel, field_validator, model_validator
29
+
30
+ from rag_wright.contracts.identifiers import EntityId
31
+ from rag_wright.contracts.provenance import GraphFact
32
+
33
+
34
+ class ClauseCategory(str, Enum):
35
+ """The 41 CUAD clause categories (ADR-0002). Authoritative; T8 reconciles exact label strings."""
36
+
37
+ DOCUMENT_NAME = "Document Name"
38
+ PARTIES = "Parties"
39
+ AGREEMENT_DATE = "Agreement Date"
40
+ EFFECTIVE_DATE = "Effective Date"
41
+ EXPIRATION_DATE = "Expiration Date"
42
+ RENEWAL_TERM = "Renewal Term"
43
+ NOTICE_PERIOD_TO_TERMINATE_RENEWAL = "Notice Period To Terminate Renewal"
44
+ GOVERNING_LAW = "Governing Law"
45
+ MOST_FAVORED_NATION = "Most Favored Nation"
46
+ NON_COMPETE = "Non-Compete"
47
+ EXCLUSIVITY = "Exclusivity"
48
+ NO_SOLICIT_OF_CUSTOMERS = "No-Solicit Of Customers"
49
+ COMPETITIVE_RESTRICTION_EXCEPTION = "Competitive Restriction Exception"
50
+ NO_SOLICIT_OF_EMPLOYEES = "No-Solicit Of Employees"
51
+ NON_DISPARAGEMENT = "Non-Disparagement"
52
+ TERMINATION_FOR_CONVENIENCE = "Termination For Convenience"
53
+ ROFR_ROFO_ROFN = "Rofr/Rofo/Rofn"
54
+ CHANGE_OF_CONTROL = "Change Of Control"
55
+ ANTI_ASSIGNMENT = "Anti-Assignment"
56
+ REVENUE_PROFIT_SHARING = "Revenue/Profit Sharing"
57
+ PRICE_RESTRICTIONS = "Price Restrictions"
58
+ MINIMUM_COMMITMENT = "Minimum Commitment"
59
+ VOLUME_RESTRICTION = "Volume Restriction"
60
+ IP_OWNERSHIP_ASSIGNMENT = "IP Ownership Assignment"
61
+ JOINT_IP_OWNERSHIP = "Joint IP Ownership"
62
+ LICENSE_GRANT = "License Grant"
63
+ NON_TRANSFERABLE_LICENSE = "Non-Transferable License"
64
+ AFFILIATE_LICENSE_LICENSOR = "Affiliate License-Licensor"
65
+ AFFILIATE_LICENSE_LICENSEE = "Affiliate License-Licensee"
66
+ UNLIMITED_ALL_YOU_CAN_EAT_LICENSE = "Unlimited/All-You-Can-Eat-License"
67
+ IRREVOCABLE_OR_PERPETUAL_LICENSE = "Irrevocable Or Perpetual License"
68
+ SOURCE_CODE_ESCROW = "Source Code Escrow"
69
+ POST_TERMINATION_SERVICES = "Post-Termination Services"
70
+ AUDIT_RIGHTS = "Audit Rights"
71
+ UNCAPPED_LIABILITY = "Uncapped Liability"
72
+ CAP_ON_LIABILITY = "Cap On Liability"
73
+ LIQUIDATED_DAMAGES = "Liquidated Damages"
74
+ WARRANTY_DURATION = "Warranty Duration"
75
+ INSURANCE = "Insurance"
76
+ COVENANT_NOT_TO_SUE = "Covenant Not To Sue"
77
+ THIRD_PARTY_BENEFICIARY = "Third Party Beneficiary"
78
+
79
+
80
+ class EntityNode(BaseModel):
81
+ """A canonical entity node in the graph skeleton (SPEC.md section 8): identifier, name, type.
82
+
83
+ No facts and no confidence: entity nodes are canonical (resolved against the registry, FR-C.7), not
84
+ extracted facts. DD-5 (ADR-0066/0117): `entity_type` is an OPAQUE string the domain names -- the engine
85
+ does not constrain the taxonomy. The reference contract pack's value set lives in
86
+ `ontology/contract_taxonomy.py` (e.g. "Organization"/"Person"); a new domain names its own.
87
+ """
88
+
89
+ entity_id: EntityId
90
+ entity_type: str
91
+ name: str
92
+
93
+
94
+ class ClauseFact(GraphFact):
95
+ """A clause occurrence extracted from a chunk (extends `GraphFact`: provenance + confidence).
96
+
97
+ `category` must be one of the 41 CUAD clause categories, so a non-ontology category is rejected.
98
+ """
99
+
100
+ category: ClauseCategory
101
+
102
+
103
+ class RelationshipFact(GraphFact):
104
+ """A directed entity-to-entity relationship extracted from a chunk (extends `GraphFact`:
105
+ provenance + confidence).
106
+
107
+ The endpoints are pre-resolution entity mentions (surface forms), directed `source_ref ->
108
+ target_ref`: source and target are distinct roles, not a symmetric pair, so T8 can add directed
109
+ corporate-hierarchy relationship types without reopening this model. Entity resolution
110
+ (FR-C.7 / T24) later maps each ref to a canonical `entity_id`.
111
+
112
+ `relationship_type` is an OPAQUE domain string (DD-5, ADR-0066/0117): the engine does not constrain the
113
+ edge taxonomy; the caller (a domain graph) names it, and the reference contract pack's value set lives in
114
+ `ontology/contract_taxonomy.py` (e.g. "Contracts With"/"Affiliate Of"). The agreement a co-party fact
115
+ derives from is its provenance's source document (`provenance.source_doc_id`); because every `GraphFact`
116
+ requires provenance, that reference is always present, which makes shared-party multi-hop questions
117
+ answerable from the graph.
118
+
119
+ Self-loop is rejected here only at the ref level (the same mention as both source and target).
120
+ The post-resolution check (two *distinct* mentions that resolve to the same `entity_id`) belongs
121
+ with entity resolution (T24), because two mentions can legitimately resolve to one entity.
122
+ """
123
+
124
+ source_ref: str # pre-resolution entity mention (surface form)
125
+ relationship_type: str # opaque domain edge type (DD-5); the caller/domain pack names it
126
+ target_ref: str # pre-resolution entity mention (surface form)
127
+
128
+ @field_validator("source_ref", "target_ref")
129
+ @classmethod
130
+ def _ref_non_empty(cls, v: str) -> str:
131
+ if not v.strip():
132
+ raise ValueError("source_ref and target_ref must be non-empty entity mentions")
133
+ return v
134
+
135
+ @model_validator(mode="after")
136
+ def _no_ref_self_loop(self) -> RelationshipFact:
137
+ if self.source_ref.strip() == self.target_ref.strip():
138
+ raise ValueError(
139
+ "source_ref and target_ref must be distinct mentions (ref-level self-loop); the "
140
+ "post-resolution same-entity_id check belongs with entity resolution (T24)"
141
+ )
142
+ return self
@@ -0,0 +1,201 @@
1
+ """The clause PROPERTY schema and contract (T57, FR-C.6, ADR-0025).
2
+
3
+ Demand-derived from the 57 ACORD test queries (schema review gate, approved). Every ACORD query is a
4
+ FUNCTION (clause type, `contracts/function.py`) plus zero-or-more PROPERTY constraints; this module is
5
+ the contract for the property layer the extractor (T57b) populates and the property graph (T57c)
6
+ persists. The clause itself stays source-of-truth in the clause OKF bundle; the property graph points
7
+ back to it (`clause_id`), and the schema stores no clause text.
8
+
9
+ Two tiers, mirroring the approved design:
10
+ - cross-cutting dimensions that recur across the liability/indemnity family (mutuality, favorability,
11
+ carve_out, covered_subject, covered_parties, party_asymmetry);
12
+ - function-specific dimensions (cap basis/quantum, damage type, warranty scope, claim scope,
13
+ procedural right, governing-law multiplicity, IP ownership, non-solicit target, renewal mechanism,
14
+ notice period, jurisdiction).
15
+
16
+ Each PROPERTY is a graph fact: `PropertyAssertion` extends `GraphFact` (FR-S.4 provenance +
17
+ EXTRACTED/INFERRED/AMBIGUOUS confidence) and cites the operative span it was read from (`span_id`,
18
+ the ADR-0025 join key) -- no claim without a citation (FR-Q.6). Closed-vocabulary dimensions validate
19
+ their value against `CLOSED_VOCAB`; a value outside the vocabulary is admissible ONLY as an AMBIGUOUS
20
+ assertion (the `other` escape, approved), so the extractor and the T58 query-decomposer share exactly
21
+ one vocabulary. Multi-valued dimensions (a carve-out set) are several assertions of the same
22
+ dimension; scalar dimensions are at most one -- the graph writer turns each assertion into one
23
+ typed edge to a (deduped) value node.
24
+ """
25
+
26
+ from __future__ import annotations
27
+
28
+ from enum import Enum
29
+
30
+ from pydantic import BaseModel, field_validator, model_validator
31
+
32
+ from rag_wright.contracts.function import FUNCTION_LABEL_SET, NO_FUNCTION, FunctionScore
33
+ from rag_wright.contracts.ontology import ClauseCategory
34
+ from rag_wright.contracts.provenance import ConfidenceTag, GraphFact
35
+ from rag_wright.ontology._generated_vocab import VOCAB as _GENERATED_VOCAB # ADR-0066: generated FROM the ttl
36
+
37
+
38
+ class PropertyDimension(str, Enum):
39
+ """The property axes the ACORD queries filter on. Tier 1 = cross-cutting; tier 2 = function-specific."""
40
+
41
+ # tier 1 -- cross-cutting (recur across the liability / indemnity family)
42
+ MUTUALITY = "mutuality"
43
+ FAVORABILITY = "favorability"
44
+ CARVE_OUT = "carve_out"
45
+ COVERED_SUBJECT = "covered_subject"
46
+ COVERED_PARTIES = "covered_parties"
47
+ PARTY_ASYMMETRY = "party_asymmetry"
48
+ # tier 2 -- function-specific
49
+ CAP_BASIS = "cap_basis"
50
+ CAP_QUANTUM = "cap_quantum" # open-valued (e.g. "12_months", "1x_fees")
51
+ DAMAGE_TYPE = "damage_type"
52
+ WARRANTY_SCOPE = "warranty_scope"
53
+ CLAIM_SCOPE = "claim_scope"
54
+ PROCEDURAL = "procedural"
55
+ JURISDICTION = "jurisdiction" # open-valued (e.g. "england", "new_york")
56
+ LAW_MULTIPLICITY = "law_multiplicity"
57
+ IP_OWNERSHIP = "ip_ownership"
58
+ NONSOLICIT_TARGET = "nonsolicit_target"
59
+ TEMPORAL_BOUND = "temporal_bound" # open-valued (e.g. "12_months", "unbounded")
60
+ RENEWAL_MECHANISM = "renewal_mechanism"
61
+ NOTICE_PERIOD = "notice_period" # open-valued
62
+ # tier 3 -- CUAD-family extensions (KG-4: full CUAD clause coverage beyond the ACORD-derived set)
63
+ EXCLUSIVITY_TYPE = "exclusivity_type"
64
+ RIGHT_OF_FIRST_TYPE = "right_of_first_type"
65
+ RESTRICTION_SCOPE = "restriction_scope" # non-compete scope
66
+ COC_CONSENT = "coc_consent" # change-of-control consent regime
67
+ ASSIGNMENT_CONSENT = "assignment_consent" # anti-assignment consent regime
68
+ ESCROW_RELEASE_TRIGGER = "escrow_release_trigger" # source-code escrow
69
+ MFN_SCOPE = "mfn_scope"
70
+ TERMINATION_RIGHT = "termination_right" # termination-for-convenience
71
+ AUDIT_FREQUENCY = "audit_frequency" # open-valued (e.g. "annual", "quarterly")
72
+ COMMITMENT_QUANTUM = "commitment_quantum" # open-valued (minimum commitment / volume restriction)
73
+ LD_TRIGGER = "ld_trigger" # open-valued (liquidated-damages trigger)
74
+ # ADR-0049 (2): new closed-vocab dimensions for the taxonomy-gap clause types (the type-specific facet each
75
+ # one carries that had no existing dimension). Vocab domain-designed + corpus-checked (ADR-0049 step 2).
76
+ DISPUTE_METHOD = "dispute_method" # how disputes are resolved (Dispute Resolution)
77
+ COLLATERAL_TYPE = "collateral_type" # collateral a security interest attaches to (Security Interest; list)
78
+ FORCE_MAJEURE_EVENT = "force_majeure_event" # excused events (Force Majeure; list)
79
+ ROYALTY_BASIS = "royalty_basis" # how a royalty is calculated (Royalties)
80
+ CONFIDENTIALITY_EXCEPTION = "confidentiality_exception" # permitted disclosures (Confidentiality; list)
81
+ CONDITION_TYPE = "condition_type" # kind of condition (Condition Precedent)
82
+
83
+
84
+ # Closed controlled vocabularies (approved OQ3). A dimension NOT in this map is open-valued
85
+ # (jurisdiction, cap_quantum, temporal_bound, notice_period) -- any non-empty value with an
86
+ # EXTRACTED/INFERRED confidence is admissible. `cap_basis` keeps a closed enum (the shape of the cap)
87
+ # while `cap_quantum` carries the light open scalar (no structured money object -- SPEC section 8).
88
+ # ADR-0066: the closed vocabularies are GENERATED FROM contract_bridge.ttl (the source of truth) into
89
+ # _generated_vocab.VOCAB (string-keyed); here they are re-keyed by PropertyDimension. To change a vocabulary,
90
+ # edit the ttl and re-run scripts/generate_contract_python.py -- never edit the value sets in Python.
91
+ CLOSED_VOCAB: dict[PropertyDimension, frozenset[str]] = {
92
+ PropertyDimension(dim): values for dim, values in _GENERATED_VOCAB.items()
93
+ }
94
+
95
+ _FOLIO_BASE = "https://folio.openlegalstandard.org/"
96
+
97
+ # Clause-TYPE -> FOLIO IRI (naming alignment only, no OWL import; verified against the live FOLIO API,
98
+ # T57). 15/16 aligned types have a home; Joint IP Ownership and Revenue/Profit Sharing have no clean
99
+ # FOLIO class (native, no IRI). Keyed by the function label string (CUAD value or extension value).
100
+ FOLIO_CLAUSE_IRI: dict[str, str] = {
101
+ ClauseCategory.CAP_ON_LIABILITY.value: _FOLIO_BASE + "RD0R9lAU0GYr2Rm3CDcMWQn",
102
+ ClauseCategory.GOVERNING_LAW.value: _FOLIO_BASE + "RCinm0jvGGkzcHth7AnasRI",
103
+ ClauseCategory.LIQUIDATED_DAMAGES.value: _FOLIO_BASE + "R8gVw3PYPaZJ9kce3wE60ag",
104
+ ClauseCategory.CHANGE_OF_CONTROL.value: _FOLIO_BASE + "Rx73OtOSnOdjzb248cqESZ",
105
+ ClauseCategory.AUDIT_RIGHTS.value: _FOLIO_BASE + "Rbjf6IGvMubNB3VG6OHa2J",
106
+ ClauseCategory.NO_SOLICIT_OF_EMPLOYEES.value: _FOLIO_BASE + "RBPNQSqdDfSS0uPPJ8pfxVL",
107
+ ClauseCategory.NO_SOLICIT_OF_CUSTOMERS.value: _FOLIO_BASE + "RBPNQSqdDfSS0uPPJ8pfxVL",
108
+ ClauseCategory.ROFR_ROFO_ROFN.value: _FOLIO_BASE + "R8vrLOm6RKTfx8fw40CYWsh",
109
+ ClauseCategory.THIRD_PARTY_BENEFICIARY.value: _FOLIO_BASE + "R97DdQGgeUgH9OJAvTnJreN",
110
+ ClauseCategory.IP_OWNERSHIP_ASSIGNMENT.value: _FOLIO_BASE + "RCvIzbBC4HsPoR3TCjrDPSr",
111
+ ClauseCategory.MINIMUM_COMMITMENT.value: _FOLIO_BASE + "RClWiJIjOouUllydauOfq00",
112
+ ClauseCategory.RENEWAL_TERM.value: _FOLIO_BASE + "R6ZPNSiwrrkYAEVTRoOKiv",
113
+ "Indemnification": _FOLIO_BASE + "R9oz08cWJcI23x0nYU1h0it",
114
+ "Indirect/Consequential Damages Waiver": _FOLIO_BASE + "RBpLvGtycyCm93U686txQg2",
115
+ "Warranty Disclaimer": _FOLIO_BASE + "RC8mge0bMEuSUAMJUlgN0rZ",
116
+ }
117
+
118
+ # Carve-out / covered SUBJECT value -> FOLIO IRI. Only these four subjects have a standalone FOLIO
119
+ # concept IRI (T57); the rest are native (no IRI). Used to tag shared `Exception`/`Subject` value nodes.
120
+ FOLIO_SUBJECT_IRI: dict[str, str] = {
121
+ "fraud": _FOLIO_BASE + "RqGxSnAp9vX42GRKHqwvBe",
122
+ "gross_negligence": _FOLIO_BASE + "RB2XGaLZqJPXLJOm052Pwrf",
123
+ "willful_misconduct": _FOLIO_BASE + "RCuhDmyUHjn92exJ8dx1zO1",
124
+ "confidentiality": _FOLIO_BASE + "ROqkYuzXx4hg7XafuJWBfJ",
125
+ }
126
+
127
+
128
+ class PropertyAssertion(GraphFact):
129
+ """One property of a clause: a (dimension, value) read from an operative span, carrying provenance
130
+ + confidence (FR-S.4) and the span citation (FR-Q.6, ADR-0025). A closed-vocabulary value outside
131
+ its vocabulary is admissible ONLY as AMBIGUOUS (the `other` escape) -- this keeps the extractor and
132
+ the query-decomposer on one shared vocabulary while still recording genuinely novel values."""
133
+
134
+ dimension: PropertyDimension
135
+ value: str
136
+ span_id: str = "" # the operative span cited (ADR-0025 join key); "" = clause-level only
137
+
138
+ @field_validator("value")
139
+ @classmethod
140
+ def _value_non_empty(cls, v: str) -> str:
141
+ if not v.strip():
142
+ raise ValueError("property value must be non-empty")
143
+ return v
144
+
145
+ @model_validator(mode="after")
146
+ def _value_in_vocab_or_ambiguous(self) -> PropertyAssertion:
147
+ vocab = CLOSED_VOCAB.get(self.dimension)
148
+ if vocab is not None and self.value not in vocab and self.confidence != ConfidenceTag.AMBIGUOUS:
149
+ raise ValueError(
150
+ f"value {self.value!r} is not in the closed vocabulary for {self.dimension.value} "
151
+ f"({sorted(vocab)}); an out-of-vocabulary value is admissible only as an AMBIGUOUS "
152
+ "assertion (the 'other' escape)"
153
+ )
154
+ return self
155
+
156
+
157
+ class ClausePropertyRecord(BaseModel):
158
+ """The property layer for one clause: its FUNCTION plus its property assertions.
159
+
160
+ `clause_id` is the parent chunk id's string form (the clause is source-of-truth in the clause OKF
161
+ bundle; the property graph points back to it). `function` must be a member of the retrieval
162
+ function taxonomy (FUNCTION_LABELS). Every assertion is anchored to this clause: its provenance's
163
+ chunk id must be this `clause_id`, so a property cannot cite a different clause (FR-Q.6). `folio_iri`
164
+ names the clause TYPE (naming alignment only) and is filled from `FOLIO_CLAUSE_IRI` when known.
165
+ """
166
+
167
+ clause_id: str
168
+ function: str
169
+ folio_iri: str = ""
170
+ # The operative span this clause was extracted from (1:1; ADR-0025). Known at extraction (op.span_id) and
171
+ # persisted here so a PROPERTY-LESS clause still has a reliable, one-to-one span link for citation/rehydration
172
+ # -- not lost, and never guessed by function label (which is one-to-many). "" only for legacy pre-backfill rows.
173
+ span_id: str = ""
174
+ assertions: list[PropertyAssertion] = []
175
+ # INGEST-LLM-CLASSIFIER (ADR-0048): the multi-label classification, ranked primary-first. `function` above is
176
+ # the PRIMARY (functions[0].function) -- the label query readers use; this additive list carries the
177
+ # secondaries + confidence for the deferred multi-label consumers. Empty on legacy / LegalBERT-single records.
178
+ functions: list[FunctionScore] = []
179
+
180
+ @field_validator("function")
181
+ @classmethod
182
+ def _function_in_taxonomy(cls, v: str) -> str:
183
+ # a real clause's function is a taxonomy member; the `NO_FUNCTION` sentinel is allowed ONLY for a
184
+ # query-constraint record (a query has no clause function -- only its extracted properties are used).
185
+ if v != NO_FUNCTION and v not in FUNCTION_LABEL_SET:
186
+ raise ValueError(
187
+ f"function {v!r} is not in the retrieval function taxonomy (FUNCTION_LABELS) "
188
+ f"or the {NO_FUNCTION!r} no-function sentinel"
189
+ )
190
+ return v
191
+
192
+ @model_validator(mode="after")
193
+ def _assertions_anchored_to_clause(self) -> ClausePropertyRecord:
194
+ for a in self.assertions:
195
+ if str(a.provenance.chunk_id) != self.clause_id:
196
+ raise ValueError(
197
+ "every assertion must be anchored to the record's clause_id "
198
+ f"(assertion provenance chunk_id {str(a.provenance.chunk_id)!r} != "
199
+ f"clause_id {self.clause_id!r})"
200
+ )
201
+ return self
@@ -0,0 +1,78 @@
1
+ """Provenance and confidence contracts (FR-S.4).
2
+
3
+ Every stored unit carries provenance: for text, the source document and the chunk it came from;
4
+ for graph-derived facts, additionally a confidence tag. Provenance is what makes "no claim without
5
+ a citation" (FR-Q.6) enforceable, and the confidence tag is what marks a graph fact as evidence to
6
+ be verified, not truth (SPEC.md section 14).
7
+
8
+ `Provenance` and `ConfidenceTag` are the reusable primitives; `GraphFact` is the base that graph
9
+ extraction (T5) and graph storage (T24) build their nodes, edges, and facts on.
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ from enum import Enum
15
+
16
+ from pydantic import BaseModel, ConfigDict, model_validator
17
+
18
+ from rag_wright.contracts.identifiers import ChunkId
19
+
20
+
21
+ class ConfidenceTag(str, Enum):
22
+ """The confidence a graph-derived fact carries (FR-S.4). A closed set, no other value.
23
+
24
+ - ``EXTRACTED``: read directly from a source chunk.
25
+ - ``INFERRED``: derived by reasoning over one or more chunks, not stated verbatim.
26
+ - ``AMBIGUOUS``: supported but with competing readings or unresolved mentions.
27
+ """
28
+
29
+ EXTRACTED = "EXTRACTED"
30
+ INFERRED = "INFERRED"
31
+ AMBIGUOUS = "AMBIGUOUS"
32
+
33
+
34
+ class Provenance(BaseModel):
35
+ """The source document and chunk a stored text unit came from (FR-S.4).
36
+
37
+ The `chunk_id` (FR-S.2) already carries its source-document identifier; `source_doc_id` is kept
38
+ as an explicit, denormalized field so a citation is self-describing, so records can be filtered
39
+ and indexed by source document at the store level (metadata filters, FR-Q.1), and so downstream
40
+ code never has to parse `chunk_id` to recover the document. The redundancy is safe only because
41
+ the two fields cannot disagree: the `_source_matches_chunk` validator runs on every construction
42
+ and deserialization path (raw constructor, `model_validate`, `model_validate_json`), and
43
+ `Provenance.of(chunk_id)` derives `source_doc_id` from the chunk so callers cannot create an
44
+ inconsistent one. Store deserialization must therefore use a validating path
45
+ (`model_validate` / `model_validate_json`), not `model_construct`, which bypasses all validation.
46
+ """
47
+
48
+ model_config = ConfigDict(frozen=True)
49
+
50
+ source_doc_id: str
51
+ chunk_id: ChunkId
52
+
53
+ @model_validator(mode="after")
54
+ def _source_matches_chunk(self) -> Provenance:
55
+ if self.source_doc_id != self.chunk_id.source_doc_id:
56
+ raise ValueError(
57
+ "source_doc_id must match chunk_id.source_doc_id "
58
+ f"({self.source_doc_id!r} != {self.chunk_id.source_doc_id!r}); "
59
+ "use Provenance.of(chunk_id) to derive it"
60
+ )
61
+ return self
62
+
63
+ @classmethod
64
+ def of(cls, chunk_id: ChunkId) -> Provenance:
65
+ """Build a `Provenance` from a chunk id, deriving `source_doc_id` from it (no drift)."""
66
+ return cls(source_doc_id=chunk_id.source_doc_id, chunk_id=chunk_id)
67
+
68
+
69
+ class GraphFact(BaseModel):
70
+ """The base for a graph-derived fact: it carries provenance and a confidence tag (FR-S.4).
71
+
72
+ Graph nodes and edges (FR-I.4) carry the originating `chunk_id` (through `provenance`) and a
73
+ `confidence` tag. Graph extraction (T5) and graph storage (T24) extend this base with their own
74
+ ontology-conforming fields.
75
+ """
76
+
77
+ provenance: Provenance
78
+ confidence: ConfidenceTag
@@ -0,0 +1,53 @@
1
+ """The NL->type query-understanding output contract (CU-A1 / CU-C1, ADR-0029).
2
+
3
+ The front door of the CUAD pipeline: a natural-language question about a known contract is parsed (one LLM
4
+ call) into this structured intent. `clause_types` are validated to the retrieval FUNCTION taxonomy
5
+ (`FUNCTION_LABELS`), normalized case-insensitively at the boundary. `intent` routes the serve stage:
6
+ - highlight -> return the clause's spans as-is
7
+ - extract -> also field-extract `value_to_extract` from the matched clause (value-type categories)
8
+ - discriminate -> `value_condition` selects among same-type clauses (case (b): detected now, stage stubbed)
9
+ `in_taxonomy=False` marks an out-of-taxonomy query -> serve falls back to semantic span search + a
10
+ low-confidence flag. Multi-type is allowed (`clause_types` is a list).
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ from typing import Literal
16
+
17
+ from pydantic import BaseModel, field_validator
18
+
19
+ from rag_wright.contracts.function import canonical_function
20
+
21
+ Intent = Literal["highlight", "extract", "discriminate"]
22
+
23
+
24
+ class QueryIntent(BaseModel):
25
+ """Structured NL->type intent produced by query understanding (CU-C1)."""
26
+
27
+ model_config = {"frozen": True}
28
+
29
+ clause_types: list[str] = [] # subset of FUNCTION_LABELS (canonicalized); empty iff out-of-taxonomy
30
+ intent: Intent = "highlight"
31
+ value_to_extract: str | None = None # for intent=extract: which value to pull from the clause
32
+ value_condition: str | None = None # for intent=discriminate: the value condition selecting the clause
33
+ in_taxonomy: bool = True # False -> semantic fallback + low-confidence flag at serve
34
+ confidence: float = 1.0 # 0..1 query-understanding confidence
35
+
36
+ @field_validator("clause_types")
37
+ @classmethod
38
+ def _canonicalize_types(cls, v: list[str]) -> list[str]:
39
+ out: list[str] = []
40
+ for label in v:
41
+ canon = canonical_function(label)
42
+ if canon is None:
43
+ raise ValueError(f"clause_type {label!r} is not in the FUNCTION taxonomy")
44
+ if canon not in out:
45
+ out.append(canon)
46
+ return out
47
+
48
+ @field_validator("confidence")
49
+ @classmethod
50
+ def _check_confidence(cls, v: float) -> float:
51
+ if not 0.0 <= v <= 1.0:
52
+ raise ValueError(f"confidence must be in [0, 1], got {v}")
53
+ return v
@@ -0,0 +1,76 @@
1
+ """Operative-span store record (FR-R, ADR-0025).
2
+
3
+ One record per operative span in the ArcadeDB `Span` hybrid index. Unlike a `ChunkRecord` (dense over the
4
+ summary, text kept in a sidecar), the span IS the small retrieval unit, so the record carries the span text:
5
+ the dense vector is over the span, the sparse vector is over the span, the `function` is the classifier tag
6
+ (T56), and the parent pointer (`parent_chunk_id` + the clause's OKF path) locates the full clause for the
7
+ rerank stage. `span_id` embeds the parent (identifier rule).
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ import math
13
+
14
+ from pydantic import BaseModel, field_validator, model_validator
15
+
16
+ from rag_wright.contracts.chunk import BGE_M3_DENSE_DIM
17
+
18
+
19
+ class SpanRecord(BaseModel):
20
+ """One operative-span record for the `Span` hybrid index (dense + sparse over the span text).
21
+
22
+ CUAD-highlighting fields (CU-A1, ADR-0029) are OPTIONAL/defaulted so the ACORD span leg (which does not
23
+ set them) is unaffected: `contract_id` is the source-document id used by the within-contract typed filter
24
+ (derivable from `parent_chunk_id` but stored explicitly for an indexed WHERE); `doc_start`/`doc_end` are
25
+ document-absolute character offsets of the span (the citation the app highlights on); `page`/`bbox` are the
26
+ optional PDF-overlay provenance (Docling-supplied where available). `parent_chunk_id` is the parent-clause
27
+ pointer (a clause == a chunk), so no separate clause_id field is added.
28
+ """
29
+
30
+ model_config = {"frozen": True}
31
+
32
+ span_id: str # "{parent_chunk_id}#{span_index}"
33
+ parent_chunk_id: str # the parent CLAUSE id (a clause is a chunk); the span<->clause link
34
+ parent_okf_path: str = "" # where the parent clause lives in the clause OKF bundle
35
+ span_index: int
36
+ text: str
37
+ function: str = "" # the PRIMARY function-classifier tag (T56); "" until classified. == functions[0] when set.
38
+ functions: list[str] = [] # T55/ADR-0114: the top-k soft tags primary-first (SetFit ensemble); `function` is
39
+ # functions[0]. Realizes the multi-tag soft-tagger so a span is discoverable under several clause types.
40
+ dense_vector: list[float] # dense over the span; length == BGE_M3_DENSE_DIM
41
+ sparse_vector: dict[int, float] # sparse over the span: token-id -> non-negative weight
42
+ contract_id: str = "" # CU-A1: source contract/document id (within-contract typed filter)
43
+ doc_start: int | None = None # CU-A1: document-absolute char offset (citation); None on the ACORD leg
44
+ doc_end: int | None = None # CU-A1: exclusive
45
+ page: int | None = None # CU-A1: 1-based FIRST page for PDF-overlay highlight (== pages[0] when known)
46
+ pages: list[int] = [] # issue 0032/CU-B5: ALL 1-based source pages this span overlaps (a clause can cross a
47
+ # page boundary); empty when the parse carried no page provenance (e.g. the text-only ingest leg)
48
+ bbox: tuple[float, float, float, float] | None = None # CU-A1: (left, top, right, bottom) on `page`, best-effort
49
+
50
+ @model_validator(mode="after")
51
+ def _check_offsets(self) -> "SpanRecord":
52
+ if self.doc_start is not None and self.doc_start < 0:
53
+ raise ValueError("doc_start must be non-negative")
54
+ if self.doc_start is not None and self.doc_end is not None and self.doc_end < self.doc_start:
55
+ raise ValueError(f"doc_end ({self.doc_end}) must be >= doc_start ({self.doc_start})")
56
+ if self.page is not None and self.page < 1:
57
+ raise ValueError("page is 1-based; must be >= 1")
58
+ if any(p < 1 for p in self.pages):
59
+ raise ValueError("pages are 1-based; each must be >= 1")
60
+ return self
61
+
62
+ @field_validator("dense_vector")
63
+ @classmethod
64
+ def _check_dense(cls, v: list[float]) -> list[float]:
65
+ if len(v) != BGE_M3_DENSE_DIM:
66
+ raise ValueError(f"dense_vector must have length {BGE_M3_DENSE_DIM}, got {len(v)}")
67
+ if not all(math.isfinite(x) for x in v):
68
+ raise ValueError("dense_vector must contain only finite values")
69
+ return v
70
+
71
+ @field_validator("sparse_vector")
72
+ @classmethod
73
+ def _check_sparse(cls, v: dict[int, float]) -> dict[int, float]:
74
+ if any(weight < 0 for weight in v.values()):
75
+ raise ValueError("sparse_vector weights must be non-negative")
76
+ return v