rag-wright 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- rag_wright/__init__.py +13 -0
- rag_wright/api/__init__.py +33 -0
- rag_wright/api/config.py +59 -0
- rag_wright/api/discover.py +70 -0
- rag_wright/api/documents.py +39 -0
- rag_wright/api/ids.py +31 -0
- rag_wright/api/invoke.py +99 -0
- rag_wright/api/kg.py +61 -0
- rag_wright/api/mcp.py +94 -0
- rag_wright/api/usage.py +30 -0
- rag_wright/api/workspace.py +85 -0
- rag_wright/capabilities/__init__.py +8 -0
- rag_wright/capabilities/answer_generator.py +427 -0
- rag_wright/capabilities/ard.py +286 -0
- rag_wright/capabilities/assertion_extraction.py +79 -0
- rag_wright/capabilities/chunk_read.py +58 -0
- rag_wright/capabilities/chunk_write.py +163 -0
- rag_wright/capabilities/claim_extraction.py +153 -0
- rag_wright/capabilities/clause_exception_linking.py +117 -0
- rag_wright/capabilities/compliance_judgment.py +322 -0
- rag_wright/capabilities/compliance_store.py +87 -0
- rag_wright/capabilities/contract_kg_serve.py +156 -0
- rag_wright/capabilities/contract_kg_store.py +251 -0
- rag_wright/capabilities/dg_extraction.py +585 -0
- rag_wright/capabilities/disambiguation.py +163 -0
- rag_wright/capabilities/document_parse.py +87 -0
- rag_wright/capabilities/document_scope.py +49 -0
- rag_wright/capabilities/embedding.py +164 -0
- rag_wright/capabilities/embedding_profiles.py +43 -0
- rag_wright/capabilities/entity_resolution.py +154 -0
- rag_wright/capabilities/fusion.py +64 -0
- rag_wright/capabilities/graph_extraction.py +243 -0
- rag_wright/capabilities/graph_query.py +73 -0
- rag_wright/capabilities/graph_storage.py +111 -0
- rag_wright/capabilities/highlight_serve.py +142 -0
- rag_wright/capabilities/hybrid_search.py +65 -0
- rag_wright/capabilities/invoke.py +31 -0
- rag_wright/capabilities/jev_decision.py +38 -0
- rag_wright/capabilities/manifests.py +872 -0
- rag_wright/capabilities/okf_navigate.py +456 -0
- rag_wright/capabilities/parsing.py +286 -0
- rag_wright/capabilities/property_boosted_retrieval.py +125 -0
- rag_wright/capabilities/query_function_classifier.py +94 -0
- rag_wright/capabilities/query_understanding.py +109 -0
- rag_wright/capabilities/registry.py +262 -0
- rag_wright/capabilities/remote_encoders.py +94 -0
- rag_wright/capabilities/requirement_extraction.py +247 -0
- rag_wright/capabilities/reranking.py +123 -0
- rag_wright/capabilities/retrieval_core.py +126 -0
- rag_wright/capabilities/rlm_chunking.py +808 -0
- rag_wright/capabilities/rlm_synthesis.py +316 -0
- rag_wright/capabilities/scan_quality.py +136 -0
- rag_wright/capabilities/span_relevance_judgment.py +191 -0
- rag_wright/capabilities/vision_to_text.py +85 -0
- rag_wright/capabilities/vlm_ocr.py +85 -0
- rag_wright/contracts/__init__.py +6 -0
- rag_wright/contracts/chunk.py +79 -0
- rag_wright/contracts/compliance.py +303 -0
- rag_wright/contracts/contract_meta.py +27 -0
- rag_wright/contracts/extraction.py +130 -0
- rag_wright/contracts/function.py +167 -0
- rag_wright/contracts/function_routing.py +91 -0
- rag_wright/contracts/highlight.py +74 -0
- rag_wright/contracts/identifiers.py +153 -0
- rag_wright/contracts/jurisdiction.py +96 -0
- rag_wright/contracts/ontology.py +142 -0
- rag_wright/contracts/property.py +201 -0
- rag_wright/contracts/provenance.py +78 -0
- rag_wright/contracts/query_intent.py +53 -0
- rag_wright/contracts/span.py +76 -0
- rag_wright/contracts/value_match.py +84 -0
- rag_wright/corpus/__init__.py +0 -0
- rag_wright/corpus/canonicalize.py +116 -0
- rag_wright/corpus/cuad.py +153 -0
- rag_wright/corpus/cuad_ingestion.py +72 -0
- rag_wright/corpus/document_parser.py +299 -0
- rag_wright/corpus/edgar.py +231 -0
- rag_wright/corpus/gcs_ingestion.py +120 -0
- rag_wright/corpus/http.py +110 -0
- rag_wright/corpus/selection.py +152 -0
- rag_wright/mcp/__init__.py +11 -0
- rag_wright/mcp/compliance_server.py +299 -0
- rag_wright/mcp/intra_document_qa_server.py +170 -0
- rag_wright/mcp/relational_qa_server.py +171 -0
- rag_wright/mcp/session_store.py +64 -0
- rag_wright/mcp/typed_property_retrieval_server.py +191 -0
- rag_wright/models/__init__.py +8 -0
- rag_wright/models/profiles.py +331 -0
- rag_wright/models/seam.py +497 -0
- rag_wright/models/tag_structured.py +285 -0
- rag_wright/models/tracing.py +179 -0
- rag_wright/models/usage.py +102 -0
- rag_wright/okf/__init__.py +11 -0
- rag_wright/okf/compile.py +292 -0
- rag_wright/okf/document.py +47 -0
- rag_wright/okf/enrich.py +176 -0
- rag_wright/okf/links.py +190 -0
- rag_wright/okf/lint.py +105 -0
- rag_wright/ontology/__init__.py +6 -0
- rag_wright/ontology/_generated_template_meta.py +60 -0
- rag_wright/ontology/_generated_vocab.py +52 -0
- rag_wright/ontology/clause_template.py +964 -0
- rag_wright/ontology/codegen.py +84 -0
- rag_wright/ontology/compliance_bridge.ttl +186 -0
- rag_wright/ontology/contract_bridge.ttl +2685 -0
- rag_wright/ontology/contract_taxonomy.py +24 -0
- rag_wright/ontology/derive.py +58 -0
- rag_wright/ontology/loader.py +435 -0
- rag_wright/ontology/packs/ftc_16cfr255.ttl +29 -0
- rag_wright/ontology/registry.py +87 -0
- rag_wright/ontology/template_introspect.py +100 -0
- rag_wright/py.typed +0 -0
- rag_wright/reference/__init__.py +2 -0
- rag_wright/reference/compliance.py +41 -0
- rag_wright/reference/contract_seam.py +123 -0
- rag_wright/skills/__init__.py +7 -0
- rag_wright/skills/claim_extraction/SKILL.md +47 -0
- rag_wright/skills/claim_extraction/__init__.py +1 -0
- rag_wright/skills/claim_extraction/template.py +50 -0
- rag_wright/skills/compliance_judgment/SKILL.md +59 -0
- rag_wright/skills/corpus_ingest/SKILL.md +106 -0
- rag_wright/skills/extraction_semantic_judge/SKILL.md +51 -0
- rag_wright/skills/extraction_semantic_judge/__init__.py +1 -0
- rag_wright/skills/generation/SKILL.md +64 -0
- rag_wright/skills/generation/__init__.py +1 -0
- rag_wright/skills/generic_compliance_judgment/SKILL.md +58 -0
- rag_wright/skills/okf_navigate/SKILL.md +137 -0
- rag_wright/skills/requirement_extraction/SKILL.md +47 -0
- rag_wright/skills/requirement_extraction/__init__.py +1 -0
- rag_wright/skills/requirement_extraction/template.py +50 -0
- rag_wright/skills/rlm/SKILL.md +186 -0
- rag_wright/skills/rlm/__init__.py +31 -0
- rag_wright/skills/rlm/agent.py +292 -0
- rag_wright/skills/span_relevance_judgment/SKILL.md +67 -0
- rag_wright/skills/vision_to_text/SKILL.md +36 -0
- rag_wright/skills/vision_to_text/__init__.py +1 -0
- rag_wright/spans/__init__.py +1 -0
- rag_wright/spans/boundary.py +78 -0
- rag_wright/spans/clause_function_classifier.py +490 -0
- rag_wright/spans/clause_kg_extractor.py +337 -0
- rag_wright/spans/cuad_labels.py +81 -0
- rag_wright/spans/dim_classifier.py +158 -0
- rag_wright/spans/dim_fleet.json +411 -0
- rag_wright/spans/function_classifier.py +77 -0
- rag_wright/spans/function_families.py +62 -0
- rag_wright/spans/hybrid_classifier.py +103 -0
- rag_wright/spans/legalbert_classifier.py +83 -0
- rag_wright/spans/model_capabilities.py +107 -0
- rag_wright/spans/new_function_labels.py +111 -0
- rag_wright/spans/page_map.py +68 -0
- rag_wright/spans/property_extractor.py +365 -0
- rag_wright/spans/property_grounding.py +182 -0
- rag_wright/spans/reclassify.py +77 -0
- rag_wright/spans/scarce_function_labels.py +105 -0
- rag_wright/spans/segment.py +341 -0
- rag_wright/spans/semantic_judge.py +197 -0
- rag_wright/spans/symbolic_validation.py +131 -0
- rag_wright/spans/tag_clause_extractor.py +182 -0
- rag_wright/store/__init__.py +6 -0
- rag_wright/store/arcadedb.py +1135 -0
- rag_wright/store/chunk_text.py +66 -0
- rag_wright/store/seam.py +213 -0
- rag_wright/subgraphs/__init__.py +0 -0
- rag_wright/subgraphs/async_ingestion.py +204 -0
- rag_wright/subgraphs/compliance_check.py +1042 -0
- rag_wright/subgraphs/compliance_ingestion.py +306 -0
- rag_wright/subgraphs/contract_ingestion_pipeline.py +999 -0
- rag_wright/subgraphs/graph_extraction.py +102 -0
- rag_wright/subgraphs/intra_document_qa.py +328 -0
- rag_wright/subgraphs/observability.py +140 -0
- rag_wright/subgraphs/query_constraint_extraction.py +73 -0
- rag_wright/subgraphs/relational_qa.py +165 -0
- rag_wright/subgraphs/requirement_extraction.py +137 -0
- rag_wright/subgraphs/scaffold.py +65 -0
- rag_wright/subgraphs/semantic_chunking.py +183 -0
- rag_wright/subgraphs/typed_clause_extraction.py +172 -0
- rag_wright/subgraphs/typed_property_retrieval.py +278 -0
- rag_wright/util/__init__.py +1 -0
- rag_wright/util/concurrent.py +153 -0
- rag_wright/util/spacy_model.py +45 -0
- rag_wright-0.1.0.dist-info/METADATA +168 -0
- rag_wright-0.1.0.dist-info/RECORD +184 -0
- rag_wright-0.1.0.dist-info/WHEEL +4 -0
- rag_wright-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,96 @@
|
|
|
1
|
+
"""KG-5a: deterministic jurisdiction canonicalization for the contract KG.
|
|
2
|
+
|
|
3
|
+
The typed KG stores `jurisdiction` as the extracted surface string (`England`, `England and Wales`,
|
|
4
|
+
`English law`, `State of New York`, `State of New York, USA`, ...). Retrieval matching (Leg B) fails on
|
|
5
|
+
those variants because the query side and the clause side don't share a normalized value. This maps a
|
|
6
|
+
surface form to a **canonical jurisdiction slug** (or None for non-jurisdictions), deterministically -- no
|
|
7
|
+
LLM, no network. Applied additively: the value node keeps its surface `value` and gains a `canonical_value`;
|
|
8
|
+
the query constraint is canonicalized the same way, so both meet on the canonical.
|
|
9
|
+
|
|
10
|
+
Method: strip governance boilerplate prefixes/suffixes (`State of`, `Commonwealth of`, `, USA`, ` law`,
|
|
11
|
+
` courts`) and normalize, then look up a gazetteer (50 US states + the countries seen in the corpus, each
|
|
12
|
+
with aliases). Non-jurisdictions (`Applicable Law`, `Not specified`, `worldwide`, redactions, compound
|
|
13
|
+
`Illinois or New York`) resolve to None (left as their surface value, unmatched).
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import re
|
|
19
|
+
|
|
20
|
+
# canonical slug -> alias surface forms (already normalized: lowercase, no prefixes/suffixes)
|
|
21
|
+
_US_STATES = [
|
|
22
|
+
"alabama", "alaska", "arizona", "arkansas", "california", "colorado", "connecticut", "delaware",
|
|
23
|
+
"florida", "georgia", "hawaii", "idaho", "illinois", "indiana", "iowa", "kansas", "kentucky",
|
|
24
|
+
"louisiana", "maine", "maryland", "massachusetts", "michigan", "minnesota", "mississippi", "missouri",
|
|
25
|
+
"montana", "nebraska", "nevada", "new hampshire", "new jersey", "new mexico", "new york",
|
|
26
|
+
"north carolina", "north dakota", "ohio", "oklahoma", "oregon", "pennsylvania", "rhode island",
|
|
27
|
+
"south carolina", "south dakota", "tennessee", "texas", "utah", "vermont", "virginia", "washington",
|
|
28
|
+
"west virginia", "wisconsin", "wyoming",
|
|
29
|
+
]
|
|
30
|
+
|
|
31
|
+
# country / non-US aliases -> canonical slug (normalized)
|
|
32
|
+
_COUNTRY_ALIASES: dict[str, str] = {
|
|
33
|
+
"england": "england", "english": "england", "england and wales": "england",
|
|
34
|
+
"united kingdom": "england", "uk": "england", "great britain": "england", "britain": "england",
|
|
35
|
+
"scotland": "scotland", "wales": "wales", "northern ireland": "northern_ireland",
|
|
36
|
+
"china": "china", "prc": "china", "people's republic of china": "china",
|
|
37
|
+
"united states": "united_states", "united states of america": "united_states", "usa": "united_states",
|
|
38
|
+
"u.s.a.": "united_states", "us": "united_states",
|
|
39
|
+
"canada": "canada", "british columbia": "british_columbia", "ontario": "ontario",
|
|
40
|
+
"belgium": "belgium", "germany": "germany", "france": "france", "japan": "japan", "spain": "spain",
|
|
41
|
+
"italy": "italy", "italian": "italy", "south africa": "south_africa", "israel": "israel",
|
|
42
|
+
"taiwan": "taiwan", "netherlands": "netherlands", "switzerland": "switzerland", "australia": "australia",
|
|
43
|
+
"singapore": "singapore", "hong kong": "hong_kong", "ireland": "ireland",
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
# build the normalized-alias -> canonical index
|
|
47
|
+
_INDEX: dict[str, str] = {s: s.replace(" ", "_") for s in _US_STATES}
|
|
48
|
+
_INDEX.update(_COUNTRY_ALIASES)
|
|
49
|
+
|
|
50
|
+
# surfaces that are NOT a jurisdiction (governance boilerplate / nulls / references / vague) -> None
|
|
51
|
+
_JUNK = re.compile(
|
|
52
|
+
r"applicable\s+law|governing\s+law|not\s+specified|not\s+explicitly|unspecified|none\s+specified"
|
|
53
|
+
r"|^none$|^other$|best's|exhibit|section\b|bankruptcy\s+code|any\s+jurisdiction|jurisdiction\s+governing"
|
|
54
|
+
r"|franchised\s+restaurant|^territory$|^union$|^worldwide$|\*|\[|last\s+sentence|internal\s+laws",
|
|
55
|
+
re.IGNORECASE,
|
|
56
|
+
)
|
|
57
|
+
_PREFIXES = ("the state of ", "state of ", "the commonwealth of ", "commonwealth of ",
|
|
58
|
+
"the province of ", "province of ", "the ")
|
|
59
|
+
_SUFFIXES = (", u.s.a.", ", usa", ", united states of america", ", united states", " (u.s.a.)",
|
|
60
|
+
" (usa)", " and its territories")
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def _normalize(surface: str) -> str:
|
|
64
|
+
s = " ".join(surface.lower().strip().split())
|
|
65
|
+
for p in _PREFIXES:
|
|
66
|
+
if s.startswith(p):
|
|
67
|
+
s = s[len(p):]
|
|
68
|
+
break
|
|
69
|
+
for suf in _SUFFIXES:
|
|
70
|
+
s = s.replace(suf, "")
|
|
71
|
+
s = re.sub(r"\s+(law|laws|courts|court|state)$", "", s).strip()
|
|
72
|
+
return s
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
# longest alias first, for the containment fallback (word-bounded)
|
|
76
|
+
_ALIAS_ITEMS = sorted(_INDEX.items(), key=lambda kv: -len(kv[0]))
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def _containment(norm: str) -> str | None:
|
|
80
|
+
"""Fallback for surfaces the exact lookup misses: scan for known jurisdiction aliases as whole words.
|
|
81
|
+
Resolve only if it names exactly one place -- treating a US state named alongside 'the United States'
|
|
82
|
+
(federal) as that state ('New York and ... the United States of America' -> new_york). Genuinely
|
|
83
|
+
ambiguous compounds ('Illinois or New York') stay None."""
|
|
84
|
+
found = {canon for alias, canon in _ALIAS_ITEMS if re.search(rf"\b{re.escape(alias)}\b", norm)}
|
|
85
|
+
if len(found) > 1:
|
|
86
|
+
found.discard("united_states") # a state + US federal -> the state
|
|
87
|
+
return next(iter(found)) if len(found) == 1 else None
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def canonicalize_jurisdiction(surface: str) -> str | None:
|
|
91
|
+
"""Map a jurisdiction surface form to a canonical slug (e.g. 'england', 'new_york'), or None if it is
|
|
92
|
+
not a resolvable single jurisdiction (boilerplate, null, reference, ambiguous compound). Deterministic."""
|
|
93
|
+
if not surface or _JUNK.search(surface):
|
|
94
|
+
return None
|
|
95
|
+
norm = _normalize(surface)
|
|
96
|
+
return _INDEX.get(norm) or _containment(norm)
|
|
@@ -0,0 +1,142 @@
|
|
|
1
|
+
"""The ontology and extraction-target models (FR-C.8, §16.2, ADR-0002).
|
|
2
|
+
|
|
3
|
+
The ontology is the closed vocabulary the knowledge graph conforms to. T4 owns the ontology
|
|
4
|
+
*structure* (the type enums and the fact models that reference them); T8 owns the ontology
|
|
5
|
+
*membership* (deriving the concrete types from the Data Catalog, FR-C.8). Both are real: deferring
|
|
6
|
+
membership entirely would leave the downstream contract and extraction work with nothing to bind.
|
|
7
|
+
|
|
8
|
+
- `ClauseCategory`: the 41 CUAD clause categories. Membership here is authoritative (ADR-0002). The
|
|
9
|
+
values are the canonical CUAD label names; T8 reconciles them against the exact label strings in
|
|
10
|
+
the CUAD data once the corpus is acquired (T7).
|
|
11
|
+
- Entity/relationship taxonomy (DD-5, ADR-0066/0117): the party/entity node types and the entity-to-entity
|
|
12
|
+
relationship (edge) types are NO LONGER a hardcoded engine enum. `EntityNode.entity_type` and
|
|
13
|
+
`RelationshipFact.relationship_type` are OPAQUE domain strings the caller names; the closed value sets are
|
|
14
|
+
DOMAIN knowledge declared by the pack (the reference contract pack's are in `ontology/contract_taxonomy.py`).
|
|
15
|
+
A new domain supplies its own without reopening these contracts.
|
|
16
|
+
|
|
17
|
+
The extraction-target models (`EntityNode`, `ClauseFact`, `RelationshipFact`) are what graph
|
|
18
|
+
extraction (T5) produces and graph storage (T24) writes. Their type fields are the ontology enums,
|
|
19
|
+
so a fact whose type is not in the ontology is rejected at construction (RAC-4). The fact models
|
|
20
|
+
extend `GraphFact` (T2), so they carry provenance and a confidence tag; `EntityNode` is a canonical
|
|
21
|
+
node (identifier, name, type, no facts), per the thin entity skeleton in SPEC.md section 8.
|
|
22
|
+
"""
|
|
23
|
+
|
|
24
|
+
from __future__ import annotations
|
|
25
|
+
|
|
26
|
+
from enum import Enum
|
|
27
|
+
|
|
28
|
+
from pydantic import BaseModel, field_validator, model_validator
|
|
29
|
+
|
|
30
|
+
from rag_wright.contracts.identifiers import EntityId
|
|
31
|
+
from rag_wright.contracts.provenance import GraphFact
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
class ClauseCategory(str, Enum):
|
|
35
|
+
"""The 41 CUAD clause categories (ADR-0002). Authoritative; T8 reconciles exact label strings."""
|
|
36
|
+
|
|
37
|
+
DOCUMENT_NAME = "Document Name"
|
|
38
|
+
PARTIES = "Parties"
|
|
39
|
+
AGREEMENT_DATE = "Agreement Date"
|
|
40
|
+
EFFECTIVE_DATE = "Effective Date"
|
|
41
|
+
EXPIRATION_DATE = "Expiration Date"
|
|
42
|
+
RENEWAL_TERM = "Renewal Term"
|
|
43
|
+
NOTICE_PERIOD_TO_TERMINATE_RENEWAL = "Notice Period To Terminate Renewal"
|
|
44
|
+
GOVERNING_LAW = "Governing Law"
|
|
45
|
+
MOST_FAVORED_NATION = "Most Favored Nation"
|
|
46
|
+
NON_COMPETE = "Non-Compete"
|
|
47
|
+
EXCLUSIVITY = "Exclusivity"
|
|
48
|
+
NO_SOLICIT_OF_CUSTOMERS = "No-Solicit Of Customers"
|
|
49
|
+
COMPETITIVE_RESTRICTION_EXCEPTION = "Competitive Restriction Exception"
|
|
50
|
+
NO_SOLICIT_OF_EMPLOYEES = "No-Solicit Of Employees"
|
|
51
|
+
NON_DISPARAGEMENT = "Non-Disparagement"
|
|
52
|
+
TERMINATION_FOR_CONVENIENCE = "Termination For Convenience"
|
|
53
|
+
ROFR_ROFO_ROFN = "Rofr/Rofo/Rofn"
|
|
54
|
+
CHANGE_OF_CONTROL = "Change Of Control"
|
|
55
|
+
ANTI_ASSIGNMENT = "Anti-Assignment"
|
|
56
|
+
REVENUE_PROFIT_SHARING = "Revenue/Profit Sharing"
|
|
57
|
+
PRICE_RESTRICTIONS = "Price Restrictions"
|
|
58
|
+
MINIMUM_COMMITMENT = "Minimum Commitment"
|
|
59
|
+
VOLUME_RESTRICTION = "Volume Restriction"
|
|
60
|
+
IP_OWNERSHIP_ASSIGNMENT = "IP Ownership Assignment"
|
|
61
|
+
JOINT_IP_OWNERSHIP = "Joint IP Ownership"
|
|
62
|
+
LICENSE_GRANT = "License Grant"
|
|
63
|
+
NON_TRANSFERABLE_LICENSE = "Non-Transferable License"
|
|
64
|
+
AFFILIATE_LICENSE_LICENSOR = "Affiliate License-Licensor"
|
|
65
|
+
AFFILIATE_LICENSE_LICENSEE = "Affiliate License-Licensee"
|
|
66
|
+
UNLIMITED_ALL_YOU_CAN_EAT_LICENSE = "Unlimited/All-You-Can-Eat-License"
|
|
67
|
+
IRREVOCABLE_OR_PERPETUAL_LICENSE = "Irrevocable Or Perpetual License"
|
|
68
|
+
SOURCE_CODE_ESCROW = "Source Code Escrow"
|
|
69
|
+
POST_TERMINATION_SERVICES = "Post-Termination Services"
|
|
70
|
+
AUDIT_RIGHTS = "Audit Rights"
|
|
71
|
+
UNCAPPED_LIABILITY = "Uncapped Liability"
|
|
72
|
+
CAP_ON_LIABILITY = "Cap On Liability"
|
|
73
|
+
LIQUIDATED_DAMAGES = "Liquidated Damages"
|
|
74
|
+
WARRANTY_DURATION = "Warranty Duration"
|
|
75
|
+
INSURANCE = "Insurance"
|
|
76
|
+
COVENANT_NOT_TO_SUE = "Covenant Not To Sue"
|
|
77
|
+
THIRD_PARTY_BENEFICIARY = "Third Party Beneficiary"
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
class EntityNode(BaseModel):
|
|
81
|
+
"""A canonical entity node in the graph skeleton (SPEC.md section 8): identifier, name, type.
|
|
82
|
+
|
|
83
|
+
No facts and no confidence: entity nodes are canonical (resolved against the registry, FR-C.7), not
|
|
84
|
+
extracted facts. DD-5 (ADR-0066/0117): `entity_type` is an OPAQUE string the domain names -- the engine
|
|
85
|
+
does not constrain the taxonomy. The reference contract pack's value set lives in
|
|
86
|
+
`ontology/contract_taxonomy.py` (e.g. "Organization"/"Person"); a new domain names its own.
|
|
87
|
+
"""
|
|
88
|
+
|
|
89
|
+
entity_id: EntityId
|
|
90
|
+
entity_type: str
|
|
91
|
+
name: str
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
class ClauseFact(GraphFact):
|
|
95
|
+
"""A clause occurrence extracted from a chunk (extends `GraphFact`: provenance + confidence).
|
|
96
|
+
|
|
97
|
+
`category` must be one of the 41 CUAD clause categories, so a non-ontology category is rejected.
|
|
98
|
+
"""
|
|
99
|
+
|
|
100
|
+
category: ClauseCategory
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
class RelationshipFact(GraphFact):
|
|
104
|
+
"""A directed entity-to-entity relationship extracted from a chunk (extends `GraphFact`:
|
|
105
|
+
provenance + confidence).
|
|
106
|
+
|
|
107
|
+
The endpoints are pre-resolution entity mentions (surface forms), directed `source_ref ->
|
|
108
|
+
target_ref`: source and target are distinct roles, not a symmetric pair, so T8 can add directed
|
|
109
|
+
corporate-hierarchy relationship types without reopening this model. Entity resolution
|
|
110
|
+
(FR-C.7 / T24) later maps each ref to a canonical `entity_id`.
|
|
111
|
+
|
|
112
|
+
`relationship_type` is an OPAQUE domain string (DD-5, ADR-0066/0117): the engine does not constrain the
|
|
113
|
+
edge taxonomy; the caller (a domain graph) names it, and the reference contract pack's value set lives in
|
|
114
|
+
`ontology/contract_taxonomy.py` (e.g. "Contracts With"/"Affiliate Of"). The agreement a co-party fact
|
|
115
|
+
derives from is its provenance's source document (`provenance.source_doc_id`); because every `GraphFact`
|
|
116
|
+
requires provenance, that reference is always present, which makes shared-party multi-hop questions
|
|
117
|
+
answerable from the graph.
|
|
118
|
+
|
|
119
|
+
Self-loop is rejected here only at the ref level (the same mention as both source and target).
|
|
120
|
+
The post-resolution check (two *distinct* mentions that resolve to the same `entity_id`) belongs
|
|
121
|
+
with entity resolution (T24), because two mentions can legitimately resolve to one entity.
|
|
122
|
+
"""
|
|
123
|
+
|
|
124
|
+
source_ref: str # pre-resolution entity mention (surface form)
|
|
125
|
+
relationship_type: str # opaque domain edge type (DD-5); the caller/domain pack names it
|
|
126
|
+
target_ref: str # pre-resolution entity mention (surface form)
|
|
127
|
+
|
|
128
|
+
@field_validator("source_ref", "target_ref")
|
|
129
|
+
@classmethod
|
|
130
|
+
def _ref_non_empty(cls, v: str) -> str:
|
|
131
|
+
if not v.strip():
|
|
132
|
+
raise ValueError("source_ref and target_ref must be non-empty entity mentions")
|
|
133
|
+
return v
|
|
134
|
+
|
|
135
|
+
@model_validator(mode="after")
|
|
136
|
+
def _no_ref_self_loop(self) -> RelationshipFact:
|
|
137
|
+
if self.source_ref.strip() == self.target_ref.strip():
|
|
138
|
+
raise ValueError(
|
|
139
|
+
"source_ref and target_ref must be distinct mentions (ref-level self-loop); the "
|
|
140
|
+
"post-resolution same-entity_id check belongs with entity resolution (T24)"
|
|
141
|
+
)
|
|
142
|
+
return self
|
|
@@ -0,0 +1,201 @@
|
|
|
1
|
+
"""The clause PROPERTY schema and contract (T57, FR-C.6, ADR-0025).
|
|
2
|
+
|
|
3
|
+
Demand-derived from the 57 ACORD test queries (schema review gate, approved). Every ACORD query is a
|
|
4
|
+
FUNCTION (clause type, `contracts/function.py`) plus zero-or-more PROPERTY constraints; this module is
|
|
5
|
+
the contract for the property layer the extractor (T57b) populates and the property graph (T57c)
|
|
6
|
+
persists. The clause itself stays source-of-truth in the clause OKF bundle; the property graph points
|
|
7
|
+
back to it (`clause_id`), and the schema stores no clause text.
|
|
8
|
+
|
|
9
|
+
Two tiers, mirroring the approved design:
|
|
10
|
+
- cross-cutting dimensions that recur across the liability/indemnity family (mutuality, favorability,
|
|
11
|
+
carve_out, covered_subject, covered_parties, party_asymmetry);
|
|
12
|
+
- function-specific dimensions (cap basis/quantum, damage type, warranty scope, claim scope,
|
|
13
|
+
procedural right, governing-law multiplicity, IP ownership, non-solicit target, renewal mechanism,
|
|
14
|
+
notice period, jurisdiction).
|
|
15
|
+
|
|
16
|
+
Each PROPERTY is a graph fact: `PropertyAssertion` extends `GraphFact` (FR-S.4 provenance +
|
|
17
|
+
EXTRACTED/INFERRED/AMBIGUOUS confidence) and cites the operative span it was read from (`span_id`,
|
|
18
|
+
the ADR-0025 join key) -- no claim without a citation (FR-Q.6). Closed-vocabulary dimensions validate
|
|
19
|
+
their value against `CLOSED_VOCAB`; a value outside the vocabulary is admissible ONLY as an AMBIGUOUS
|
|
20
|
+
assertion (the `other` escape, approved), so the extractor and the T58 query-decomposer share exactly
|
|
21
|
+
one vocabulary. Multi-valued dimensions (a carve-out set) are several assertions of the same
|
|
22
|
+
dimension; scalar dimensions are at most one -- the graph writer turns each assertion into one
|
|
23
|
+
typed edge to a (deduped) value node.
|
|
24
|
+
"""
|
|
25
|
+
|
|
26
|
+
from __future__ import annotations
|
|
27
|
+
|
|
28
|
+
from enum import Enum
|
|
29
|
+
|
|
30
|
+
from pydantic import BaseModel, field_validator, model_validator
|
|
31
|
+
|
|
32
|
+
from rag_wright.contracts.function import FUNCTION_LABEL_SET, NO_FUNCTION, FunctionScore
|
|
33
|
+
from rag_wright.contracts.ontology import ClauseCategory
|
|
34
|
+
from rag_wright.contracts.provenance import ConfidenceTag, GraphFact
|
|
35
|
+
from rag_wright.ontology._generated_vocab import VOCAB as _GENERATED_VOCAB # ADR-0066: generated FROM the ttl
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
class PropertyDimension(str, Enum):
|
|
39
|
+
"""The property axes the ACORD queries filter on. Tier 1 = cross-cutting; tier 2 = function-specific."""
|
|
40
|
+
|
|
41
|
+
# tier 1 -- cross-cutting (recur across the liability / indemnity family)
|
|
42
|
+
MUTUALITY = "mutuality"
|
|
43
|
+
FAVORABILITY = "favorability"
|
|
44
|
+
CARVE_OUT = "carve_out"
|
|
45
|
+
COVERED_SUBJECT = "covered_subject"
|
|
46
|
+
COVERED_PARTIES = "covered_parties"
|
|
47
|
+
PARTY_ASYMMETRY = "party_asymmetry"
|
|
48
|
+
# tier 2 -- function-specific
|
|
49
|
+
CAP_BASIS = "cap_basis"
|
|
50
|
+
CAP_QUANTUM = "cap_quantum" # open-valued (e.g. "12_months", "1x_fees")
|
|
51
|
+
DAMAGE_TYPE = "damage_type"
|
|
52
|
+
WARRANTY_SCOPE = "warranty_scope"
|
|
53
|
+
CLAIM_SCOPE = "claim_scope"
|
|
54
|
+
PROCEDURAL = "procedural"
|
|
55
|
+
JURISDICTION = "jurisdiction" # open-valued (e.g. "england", "new_york")
|
|
56
|
+
LAW_MULTIPLICITY = "law_multiplicity"
|
|
57
|
+
IP_OWNERSHIP = "ip_ownership"
|
|
58
|
+
NONSOLICIT_TARGET = "nonsolicit_target"
|
|
59
|
+
TEMPORAL_BOUND = "temporal_bound" # open-valued (e.g. "12_months", "unbounded")
|
|
60
|
+
RENEWAL_MECHANISM = "renewal_mechanism"
|
|
61
|
+
NOTICE_PERIOD = "notice_period" # open-valued
|
|
62
|
+
# tier 3 -- CUAD-family extensions (KG-4: full CUAD clause coverage beyond the ACORD-derived set)
|
|
63
|
+
EXCLUSIVITY_TYPE = "exclusivity_type"
|
|
64
|
+
RIGHT_OF_FIRST_TYPE = "right_of_first_type"
|
|
65
|
+
RESTRICTION_SCOPE = "restriction_scope" # non-compete scope
|
|
66
|
+
COC_CONSENT = "coc_consent" # change-of-control consent regime
|
|
67
|
+
ASSIGNMENT_CONSENT = "assignment_consent" # anti-assignment consent regime
|
|
68
|
+
ESCROW_RELEASE_TRIGGER = "escrow_release_trigger" # source-code escrow
|
|
69
|
+
MFN_SCOPE = "mfn_scope"
|
|
70
|
+
TERMINATION_RIGHT = "termination_right" # termination-for-convenience
|
|
71
|
+
AUDIT_FREQUENCY = "audit_frequency" # open-valued (e.g. "annual", "quarterly")
|
|
72
|
+
COMMITMENT_QUANTUM = "commitment_quantum" # open-valued (minimum commitment / volume restriction)
|
|
73
|
+
LD_TRIGGER = "ld_trigger" # open-valued (liquidated-damages trigger)
|
|
74
|
+
# ADR-0049 (2): new closed-vocab dimensions for the taxonomy-gap clause types (the type-specific facet each
|
|
75
|
+
# one carries that had no existing dimension). Vocab domain-designed + corpus-checked (ADR-0049 step 2).
|
|
76
|
+
DISPUTE_METHOD = "dispute_method" # how disputes are resolved (Dispute Resolution)
|
|
77
|
+
COLLATERAL_TYPE = "collateral_type" # collateral a security interest attaches to (Security Interest; list)
|
|
78
|
+
FORCE_MAJEURE_EVENT = "force_majeure_event" # excused events (Force Majeure; list)
|
|
79
|
+
ROYALTY_BASIS = "royalty_basis" # how a royalty is calculated (Royalties)
|
|
80
|
+
CONFIDENTIALITY_EXCEPTION = "confidentiality_exception" # permitted disclosures (Confidentiality; list)
|
|
81
|
+
CONDITION_TYPE = "condition_type" # kind of condition (Condition Precedent)
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
# Closed controlled vocabularies (approved OQ3). A dimension NOT in this map is open-valued
|
|
85
|
+
# (jurisdiction, cap_quantum, temporal_bound, notice_period) -- any non-empty value with an
|
|
86
|
+
# EXTRACTED/INFERRED confidence is admissible. `cap_basis` keeps a closed enum (the shape of the cap)
|
|
87
|
+
# while `cap_quantum` carries the light open scalar (no structured money object -- SPEC section 8).
|
|
88
|
+
# ADR-0066: the closed vocabularies are GENERATED FROM contract_bridge.ttl (the source of truth) into
|
|
89
|
+
# _generated_vocab.VOCAB (string-keyed); here they are re-keyed by PropertyDimension. To change a vocabulary,
|
|
90
|
+
# edit the ttl and re-run scripts/generate_contract_python.py -- never edit the value sets in Python.
|
|
91
|
+
CLOSED_VOCAB: dict[PropertyDimension, frozenset[str]] = {
|
|
92
|
+
PropertyDimension(dim): values for dim, values in _GENERATED_VOCAB.items()
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
_FOLIO_BASE = "https://folio.openlegalstandard.org/"
|
|
96
|
+
|
|
97
|
+
# Clause-TYPE -> FOLIO IRI (naming alignment only, no OWL import; verified against the live FOLIO API,
|
|
98
|
+
# T57). 15/16 aligned types have a home; Joint IP Ownership and Revenue/Profit Sharing have no clean
|
|
99
|
+
# FOLIO class (native, no IRI). Keyed by the function label string (CUAD value or extension value).
|
|
100
|
+
FOLIO_CLAUSE_IRI: dict[str, str] = {
|
|
101
|
+
ClauseCategory.CAP_ON_LIABILITY.value: _FOLIO_BASE + "RD0R9lAU0GYr2Rm3CDcMWQn",
|
|
102
|
+
ClauseCategory.GOVERNING_LAW.value: _FOLIO_BASE + "RCinm0jvGGkzcHth7AnasRI",
|
|
103
|
+
ClauseCategory.LIQUIDATED_DAMAGES.value: _FOLIO_BASE + "R8gVw3PYPaZJ9kce3wE60ag",
|
|
104
|
+
ClauseCategory.CHANGE_OF_CONTROL.value: _FOLIO_BASE + "Rx73OtOSnOdjzb248cqESZ",
|
|
105
|
+
ClauseCategory.AUDIT_RIGHTS.value: _FOLIO_BASE + "Rbjf6IGvMubNB3VG6OHa2J",
|
|
106
|
+
ClauseCategory.NO_SOLICIT_OF_EMPLOYEES.value: _FOLIO_BASE + "RBPNQSqdDfSS0uPPJ8pfxVL",
|
|
107
|
+
ClauseCategory.NO_SOLICIT_OF_CUSTOMERS.value: _FOLIO_BASE + "RBPNQSqdDfSS0uPPJ8pfxVL",
|
|
108
|
+
ClauseCategory.ROFR_ROFO_ROFN.value: _FOLIO_BASE + "R8vrLOm6RKTfx8fw40CYWsh",
|
|
109
|
+
ClauseCategory.THIRD_PARTY_BENEFICIARY.value: _FOLIO_BASE + "R97DdQGgeUgH9OJAvTnJreN",
|
|
110
|
+
ClauseCategory.IP_OWNERSHIP_ASSIGNMENT.value: _FOLIO_BASE + "RCvIzbBC4HsPoR3TCjrDPSr",
|
|
111
|
+
ClauseCategory.MINIMUM_COMMITMENT.value: _FOLIO_BASE + "RClWiJIjOouUllydauOfq00",
|
|
112
|
+
ClauseCategory.RENEWAL_TERM.value: _FOLIO_BASE + "R6ZPNSiwrrkYAEVTRoOKiv",
|
|
113
|
+
"Indemnification": _FOLIO_BASE + "R9oz08cWJcI23x0nYU1h0it",
|
|
114
|
+
"Indirect/Consequential Damages Waiver": _FOLIO_BASE + "RBpLvGtycyCm93U686txQg2",
|
|
115
|
+
"Warranty Disclaimer": _FOLIO_BASE + "RC8mge0bMEuSUAMJUlgN0rZ",
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
# Carve-out / covered SUBJECT value -> FOLIO IRI. Only these four subjects have a standalone FOLIO
|
|
119
|
+
# concept IRI (T57); the rest are native (no IRI). Used to tag shared `Exception`/`Subject` value nodes.
|
|
120
|
+
FOLIO_SUBJECT_IRI: dict[str, str] = {
|
|
121
|
+
"fraud": _FOLIO_BASE + "RqGxSnAp9vX42GRKHqwvBe",
|
|
122
|
+
"gross_negligence": _FOLIO_BASE + "RB2XGaLZqJPXLJOm052Pwrf",
|
|
123
|
+
"willful_misconduct": _FOLIO_BASE + "RCuhDmyUHjn92exJ8dx1zO1",
|
|
124
|
+
"confidentiality": _FOLIO_BASE + "ROqkYuzXx4hg7XafuJWBfJ",
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
class PropertyAssertion(GraphFact):
|
|
129
|
+
"""One property of a clause: a (dimension, value) read from an operative span, carrying provenance
|
|
130
|
+
+ confidence (FR-S.4) and the span citation (FR-Q.6, ADR-0025). A closed-vocabulary value outside
|
|
131
|
+
its vocabulary is admissible ONLY as AMBIGUOUS (the `other` escape) -- this keeps the extractor and
|
|
132
|
+
the query-decomposer on one shared vocabulary while still recording genuinely novel values."""
|
|
133
|
+
|
|
134
|
+
dimension: PropertyDimension
|
|
135
|
+
value: str
|
|
136
|
+
span_id: str = "" # the operative span cited (ADR-0025 join key); "" = clause-level only
|
|
137
|
+
|
|
138
|
+
@field_validator("value")
|
|
139
|
+
@classmethod
|
|
140
|
+
def _value_non_empty(cls, v: str) -> str:
|
|
141
|
+
if not v.strip():
|
|
142
|
+
raise ValueError("property value must be non-empty")
|
|
143
|
+
return v
|
|
144
|
+
|
|
145
|
+
@model_validator(mode="after")
|
|
146
|
+
def _value_in_vocab_or_ambiguous(self) -> PropertyAssertion:
|
|
147
|
+
vocab = CLOSED_VOCAB.get(self.dimension)
|
|
148
|
+
if vocab is not None and self.value not in vocab and self.confidence != ConfidenceTag.AMBIGUOUS:
|
|
149
|
+
raise ValueError(
|
|
150
|
+
f"value {self.value!r} is not in the closed vocabulary for {self.dimension.value} "
|
|
151
|
+
f"({sorted(vocab)}); an out-of-vocabulary value is admissible only as an AMBIGUOUS "
|
|
152
|
+
"assertion (the 'other' escape)"
|
|
153
|
+
)
|
|
154
|
+
return self
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
class ClausePropertyRecord(BaseModel):
|
|
158
|
+
"""The property layer for one clause: its FUNCTION plus its property assertions.
|
|
159
|
+
|
|
160
|
+
`clause_id` is the parent chunk id's string form (the clause is source-of-truth in the clause OKF
|
|
161
|
+
bundle; the property graph points back to it). `function` must be a member of the retrieval
|
|
162
|
+
function taxonomy (FUNCTION_LABELS). Every assertion is anchored to this clause: its provenance's
|
|
163
|
+
chunk id must be this `clause_id`, so a property cannot cite a different clause (FR-Q.6). `folio_iri`
|
|
164
|
+
names the clause TYPE (naming alignment only) and is filled from `FOLIO_CLAUSE_IRI` when known.
|
|
165
|
+
"""
|
|
166
|
+
|
|
167
|
+
clause_id: str
|
|
168
|
+
function: str
|
|
169
|
+
folio_iri: str = ""
|
|
170
|
+
# The operative span this clause was extracted from (1:1; ADR-0025). Known at extraction (op.span_id) and
|
|
171
|
+
# persisted here so a PROPERTY-LESS clause still has a reliable, one-to-one span link for citation/rehydration
|
|
172
|
+
# -- not lost, and never guessed by function label (which is one-to-many). "" only for legacy pre-backfill rows.
|
|
173
|
+
span_id: str = ""
|
|
174
|
+
assertions: list[PropertyAssertion] = []
|
|
175
|
+
# INGEST-LLM-CLASSIFIER (ADR-0048): the multi-label classification, ranked primary-first. `function` above is
|
|
176
|
+
# the PRIMARY (functions[0].function) -- the label query readers use; this additive list carries the
|
|
177
|
+
# secondaries + confidence for the deferred multi-label consumers. Empty on legacy / LegalBERT-single records.
|
|
178
|
+
functions: list[FunctionScore] = []
|
|
179
|
+
|
|
180
|
+
@field_validator("function")
|
|
181
|
+
@classmethod
|
|
182
|
+
def _function_in_taxonomy(cls, v: str) -> str:
|
|
183
|
+
# a real clause's function is a taxonomy member; the `NO_FUNCTION` sentinel is allowed ONLY for a
|
|
184
|
+
# query-constraint record (a query has no clause function -- only its extracted properties are used).
|
|
185
|
+
if v != NO_FUNCTION and v not in FUNCTION_LABEL_SET:
|
|
186
|
+
raise ValueError(
|
|
187
|
+
f"function {v!r} is not in the retrieval function taxonomy (FUNCTION_LABELS) "
|
|
188
|
+
f"or the {NO_FUNCTION!r} no-function sentinel"
|
|
189
|
+
)
|
|
190
|
+
return v
|
|
191
|
+
|
|
192
|
+
@model_validator(mode="after")
|
|
193
|
+
def _assertions_anchored_to_clause(self) -> ClausePropertyRecord:
|
|
194
|
+
for a in self.assertions:
|
|
195
|
+
if str(a.provenance.chunk_id) != self.clause_id:
|
|
196
|
+
raise ValueError(
|
|
197
|
+
"every assertion must be anchored to the record's clause_id "
|
|
198
|
+
f"(assertion provenance chunk_id {str(a.provenance.chunk_id)!r} != "
|
|
199
|
+
f"clause_id {self.clause_id!r})"
|
|
200
|
+
)
|
|
201
|
+
return self
|
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
"""Provenance and confidence contracts (FR-S.4).
|
|
2
|
+
|
|
3
|
+
Every stored unit carries provenance: for text, the source document and the chunk it came from;
|
|
4
|
+
for graph-derived facts, additionally a confidence tag. Provenance is what makes "no claim without
|
|
5
|
+
a citation" (FR-Q.6) enforceable, and the confidence tag is what marks a graph fact as evidence to
|
|
6
|
+
be verified, not truth (SPEC.md section 14).
|
|
7
|
+
|
|
8
|
+
`Provenance` and `ConfidenceTag` are the reusable primitives; `GraphFact` is the base that graph
|
|
9
|
+
extraction (T5) and graph storage (T24) build their nodes, edges, and facts on.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
from enum import Enum
|
|
15
|
+
|
|
16
|
+
from pydantic import BaseModel, ConfigDict, model_validator
|
|
17
|
+
|
|
18
|
+
from rag_wright.contracts.identifiers import ChunkId
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
class ConfidenceTag(str, Enum):
|
|
22
|
+
"""The confidence a graph-derived fact carries (FR-S.4). A closed set, no other value.
|
|
23
|
+
|
|
24
|
+
- ``EXTRACTED``: read directly from a source chunk.
|
|
25
|
+
- ``INFERRED``: derived by reasoning over one or more chunks, not stated verbatim.
|
|
26
|
+
- ``AMBIGUOUS``: supported but with competing readings or unresolved mentions.
|
|
27
|
+
"""
|
|
28
|
+
|
|
29
|
+
EXTRACTED = "EXTRACTED"
|
|
30
|
+
INFERRED = "INFERRED"
|
|
31
|
+
AMBIGUOUS = "AMBIGUOUS"
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
class Provenance(BaseModel):
|
|
35
|
+
"""The source document and chunk a stored text unit came from (FR-S.4).
|
|
36
|
+
|
|
37
|
+
The `chunk_id` (FR-S.2) already carries its source-document identifier; `source_doc_id` is kept
|
|
38
|
+
as an explicit, denormalized field so a citation is self-describing, so records can be filtered
|
|
39
|
+
and indexed by source document at the store level (metadata filters, FR-Q.1), and so downstream
|
|
40
|
+
code never has to parse `chunk_id` to recover the document. The redundancy is safe only because
|
|
41
|
+
the two fields cannot disagree: the `_source_matches_chunk` validator runs on every construction
|
|
42
|
+
and deserialization path (raw constructor, `model_validate`, `model_validate_json`), and
|
|
43
|
+
`Provenance.of(chunk_id)` derives `source_doc_id` from the chunk so callers cannot create an
|
|
44
|
+
inconsistent one. Store deserialization must therefore use a validating path
|
|
45
|
+
(`model_validate` / `model_validate_json`), not `model_construct`, which bypasses all validation.
|
|
46
|
+
"""
|
|
47
|
+
|
|
48
|
+
model_config = ConfigDict(frozen=True)
|
|
49
|
+
|
|
50
|
+
source_doc_id: str
|
|
51
|
+
chunk_id: ChunkId
|
|
52
|
+
|
|
53
|
+
@model_validator(mode="after")
|
|
54
|
+
def _source_matches_chunk(self) -> Provenance:
|
|
55
|
+
if self.source_doc_id != self.chunk_id.source_doc_id:
|
|
56
|
+
raise ValueError(
|
|
57
|
+
"source_doc_id must match chunk_id.source_doc_id "
|
|
58
|
+
f"({self.source_doc_id!r} != {self.chunk_id.source_doc_id!r}); "
|
|
59
|
+
"use Provenance.of(chunk_id) to derive it"
|
|
60
|
+
)
|
|
61
|
+
return self
|
|
62
|
+
|
|
63
|
+
@classmethod
|
|
64
|
+
def of(cls, chunk_id: ChunkId) -> Provenance:
|
|
65
|
+
"""Build a `Provenance` from a chunk id, deriving `source_doc_id` from it (no drift)."""
|
|
66
|
+
return cls(source_doc_id=chunk_id.source_doc_id, chunk_id=chunk_id)
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
class GraphFact(BaseModel):
|
|
70
|
+
"""The base for a graph-derived fact: it carries provenance and a confidence tag (FR-S.4).
|
|
71
|
+
|
|
72
|
+
Graph nodes and edges (FR-I.4) carry the originating `chunk_id` (through `provenance`) and a
|
|
73
|
+
`confidence` tag. Graph extraction (T5) and graph storage (T24) extend this base with their own
|
|
74
|
+
ontology-conforming fields.
|
|
75
|
+
"""
|
|
76
|
+
|
|
77
|
+
provenance: Provenance
|
|
78
|
+
confidence: ConfidenceTag
|
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
"""The NL->type query-understanding output contract (CU-A1 / CU-C1, ADR-0029).
|
|
2
|
+
|
|
3
|
+
The front door of the CUAD pipeline: a natural-language question about a known contract is parsed (one LLM
|
|
4
|
+
call) into this structured intent. `clause_types` are validated to the retrieval FUNCTION taxonomy
|
|
5
|
+
(`FUNCTION_LABELS`), normalized case-insensitively at the boundary. `intent` routes the serve stage:
|
|
6
|
+
- highlight -> return the clause's spans as-is
|
|
7
|
+
- extract -> also field-extract `value_to_extract` from the matched clause (value-type categories)
|
|
8
|
+
- discriminate -> `value_condition` selects among same-type clauses (case (b): detected now, stage stubbed)
|
|
9
|
+
`in_taxonomy=False` marks an out-of-taxonomy query -> serve falls back to semantic span search + a
|
|
10
|
+
low-confidence flag. Multi-type is allowed (`clause_types` is a list).
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
from typing import Literal
|
|
16
|
+
|
|
17
|
+
from pydantic import BaseModel, field_validator
|
|
18
|
+
|
|
19
|
+
from rag_wright.contracts.function import canonical_function
|
|
20
|
+
|
|
21
|
+
Intent = Literal["highlight", "extract", "discriminate"]
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class QueryIntent(BaseModel):
|
|
25
|
+
"""Structured NL->type intent produced by query understanding (CU-C1)."""
|
|
26
|
+
|
|
27
|
+
model_config = {"frozen": True}
|
|
28
|
+
|
|
29
|
+
clause_types: list[str] = [] # subset of FUNCTION_LABELS (canonicalized); empty iff out-of-taxonomy
|
|
30
|
+
intent: Intent = "highlight"
|
|
31
|
+
value_to_extract: str | None = None # for intent=extract: which value to pull from the clause
|
|
32
|
+
value_condition: str | None = None # for intent=discriminate: the value condition selecting the clause
|
|
33
|
+
in_taxonomy: bool = True # False -> semantic fallback + low-confidence flag at serve
|
|
34
|
+
confidence: float = 1.0 # 0..1 query-understanding confidence
|
|
35
|
+
|
|
36
|
+
@field_validator("clause_types")
|
|
37
|
+
@classmethod
|
|
38
|
+
def _canonicalize_types(cls, v: list[str]) -> list[str]:
|
|
39
|
+
out: list[str] = []
|
|
40
|
+
for label in v:
|
|
41
|
+
canon = canonical_function(label)
|
|
42
|
+
if canon is None:
|
|
43
|
+
raise ValueError(f"clause_type {label!r} is not in the FUNCTION taxonomy")
|
|
44
|
+
if canon not in out:
|
|
45
|
+
out.append(canon)
|
|
46
|
+
return out
|
|
47
|
+
|
|
48
|
+
@field_validator("confidence")
|
|
49
|
+
@classmethod
|
|
50
|
+
def _check_confidence(cls, v: float) -> float:
|
|
51
|
+
if not 0.0 <= v <= 1.0:
|
|
52
|
+
raise ValueError(f"confidence must be in [0, 1], got {v}")
|
|
53
|
+
return v
|
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
"""Operative-span store record (FR-R, ADR-0025).
|
|
2
|
+
|
|
3
|
+
One record per operative span in the ArcadeDB `Span` hybrid index. Unlike a `ChunkRecord` (dense over the
|
|
4
|
+
summary, text kept in a sidecar), the span IS the small retrieval unit, so the record carries the span text:
|
|
5
|
+
the dense vector is over the span, the sparse vector is over the span, the `function` is the classifier tag
|
|
6
|
+
(T56), and the parent pointer (`parent_chunk_id` + the clause's OKF path) locates the full clause for the
|
|
7
|
+
rerank stage. `span_id` embeds the parent (identifier rule).
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import math
|
|
13
|
+
|
|
14
|
+
from pydantic import BaseModel, field_validator, model_validator
|
|
15
|
+
|
|
16
|
+
from rag_wright.contracts.chunk import BGE_M3_DENSE_DIM
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class SpanRecord(BaseModel):
|
|
20
|
+
"""One operative-span record for the `Span` hybrid index (dense + sparse over the span text).
|
|
21
|
+
|
|
22
|
+
CUAD-highlighting fields (CU-A1, ADR-0029) are OPTIONAL/defaulted so the ACORD span leg (which does not
|
|
23
|
+
set them) is unaffected: `contract_id` is the source-document id used by the within-contract typed filter
|
|
24
|
+
(derivable from `parent_chunk_id` but stored explicitly for an indexed WHERE); `doc_start`/`doc_end` are
|
|
25
|
+
document-absolute character offsets of the span (the citation the app highlights on); `page`/`bbox` are the
|
|
26
|
+
optional PDF-overlay provenance (Docling-supplied where available). `parent_chunk_id` is the parent-clause
|
|
27
|
+
pointer (a clause == a chunk), so no separate clause_id field is added.
|
|
28
|
+
"""
|
|
29
|
+
|
|
30
|
+
model_config = {"frozen": True}
|
|
31
|
+
|
|
32
|
+
span_id: str # "{parent_chunk_id}#{span_index}"
|
|
33
|
+
parent_chunk_id: str # the parent CLAUSE id (a clause is a chunk); the span<->clause link
|
|
34
|
+
parent_okf_path: str = "" # where the parent clause lives in the clause OKF bundle
|
|
35
|
+
span_index: int
|
|
36
|
+
text: str
|
|
37
|
+
function: str = "" # the PRIMARY function-classifier tag (T56); "" until classified. == functions[0] when set.
|
|
38
|
+
functions: list[str] = [] # T55/ADR-0114: the top-k soft tags primary-first (SetFit ensemble); `function` is
|
|
39
|
+
# functions[0]. Realizes the multi-tag soft-tagger so a span is discoverable under several clause types.
|
|
40
|
+
dense_vector: list[float] # dense over the span; length == BGE_M3_DENSE_DIM
|
|
41
|
+
sparse_vector: dict[int, float] # sparse over the span: token-id -> non-negative weight
|
|
42
|
+
contract_id: str = "" # CU-A1: source contract/document id (within-contract typed filter)
|
|
43
|
+
doc_start: int | None = None # CU-A1: document-absolute char offset (citation); None on the ACORD leg
|
|
44
|
+
doc_end: int | None = None # CU-A1: exclusive
|
|
45
|
+
page: int | None = None # CU-A1: 1-based FIRST page for PDF-overlay highlight (== pages[0] when known)
|
|
46
|
+
pages: list[int] = [] # issue 0032/CU-B5: ALL 1-based source pages this span overlaps (a clause can cross a
|
|
47
|
+
# page boundary); empty when the parse carried no page provenance (e.g. the text-only ingest leg)
|
|
48
|
+
bbox: tuple[float, float, float, float] | None = None # CU-A1: (left, top, right, bottom) on `page`, best-effort
|
|
49
|
+
|
|
50
|
+
@model_validator(mode="after")
|
|
51
|
+
def _check_offsets(self) -> "SpanRecord":
|
|
52
|
+
if self.doc_start is not None and self.doc_start < 0:
|
|
53
|
+
raise ValueError("doc_start must be non-negative")
|
|
54
|
+
if self.doc_start is not None and self.doc_end is not None and self.doc_end < self.doc_start:
|
|
55
|
+
raise ValueError(f"doc_end ({self.doc_end}) must be >= doc_start ({self.doc_start})")
|
|
56
|
+
if self.page is not None and self.page < 1:
|
|
57
|
+
raise ValueError("page is 1-based; must be >= 1")
|
|
58
|
+
if any(p < 1 for p in self.pages):
|
|
59
|
+
raise ValueError("pages are 1-based; each must be >= 1")
|
|
60
|
+
return self
|
|
61
|
+
|
|
62
|
+
@field_validator("dense_vector")
|
|
63
|
+
@classmethod
|
|
64
|
+
def _check_dense(cls, v: list[float]) -> list[float]:
|
|
65
|
+
if len(v) != BGE_M3_DENSE_DIM:
|
|
66
|
+
raise ValueError(f"dense_vector must have length {BGE_M3_DENSE_DIM}, got {len(v)}")
|
|
67
|
+
if not all(math.isfinite(x) for x in v):
|
|
68
|
+
raise ValueError("dense_vector must contain only finite values")
|
|
69
|
+
return v
|
|
70
|
+
|
|
71
|
+
@field_validator("sparse_vector")
|
|
72
|
+
@classmethod
|
|
73
|
+
def _check_sparse(cls, v: dict[int, float]) -> dict[int, float]:
|
|
74
|
+
if any(weight < 0 for weight in v.values()):
|
|
75
|
+
raise ValueError("sparse_vector weights must be non-negative")
|
|
76
|
+
return v
|