rag-wright 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- rag_wright/__init__.py +13 -0
- rag_wright/api/__init__.py +33 -0
- rag_wright/api/config.py +59 -0
- rag_wright/api/discover.py +70 -0
- rag_wright/api/documents.py +39 -0
- rag_wright/api/ids.py +31 -0
- rag_wright/api/invoke.py +99 -0
- rag_wright/api/kg.py +61 -0
- rag_wright/api/mcp.py +94 -0
- rag_wright/api/usage.py +30 -0
- rag_wright/api/workspace.py +85 -0
- rag_wright/capabilities/__init__.py +8 -0
- rag_wright/capabilities/answer_generator.py +427 -0
- rag_wright/capabilities/ard.py +286 -0
- rag_wright/capabilities/assertion_extraction.py +79 -0
- rag_wright/capabilities/chunk_read.py +58 -0
- rag_wright/capabilities/chunk_write.py +163 -0
- rag_wright/capabilities/claim_extraction.py +153 -0
- rag_wright/capabilities/clause_exception_linking.py +117 -0
- rag_wright/capabilities/compliance_judgment.py +322 -0
- rag_wright/capabilities/compliance_store.py +87 -0
- rag_wright/capabilities/contract_kg_serve.py +156 -0
- rag_wright/capabilities/contract_kg_store.py +251 -0
- rag_wright/capabilities/dg_extraction.py +585 -0
- rag_wright/capabilities/disambiguation.py +163 -0
- rag_wright/capabilities/document_parse.py +87 -0
- rag_wright/capabilities/document_scope.py +49 -0
- rag_wright/capabilities/embedding.py +164 -0
- rag_wright/capabilities/embedding_profiles.py +43 -0
- rag_wright/capabilities/entity_resolution.py +154 -0
- rag_wright/capabilities/fusion.py +64 -0
- rag_wright/capabilities/graph_extraction.py +243 -0
- rag_wright/capabilities/graph_query.py +73 -0
- rag_wright/capabilities/graph_storage.py +111 -0
- rag_wright/capabilities/highlight_serve.py +142 -0
- rag_wright/capabilities/hybrid_search.py +65 -0
- rag_wright/capabilities/invoke.py +31 -0
- rag_wright/capabilities/jev_decision.py +38 -0
- rag_wright/capabilities/manifests.py +872 -0
- rag_wright/capabilities/okf_navigate.py +456 -0
- rag_wright/capabilities/parsing.py +286 -0
- rag_wright/capabilities/property_boosted_retrieval.py +125 -0
- rag_wright/capabilities/query_function_classifier.py +94 -0
- rag_wright/capabilities/query_understanding.py +109 -0
- rag_wright/capabilities/registry.py +262 -0
- rag_wright/capabilities/remote_encoders.py +94 -0
- rag_wright/capabilities/requirement_extraction.py +247 -0
- rag_wright/capabilities/reranking.py +123 -0
- rag_wright/capabilities/retrieval_core.py +126 -0
- rag_wright/capabilities/rlm_chunking.py +808 -0
- rag_wright/capabilities/rlm_synthesis.py +316 -0
- rag_wright/capabilities/scan_quality.py +136 -0
- rag_wright/capabilities/span_relevance_judgment.py +191 -0
- rag_wright/capabilities/vision_to_text.py +85 -0
- rag_wright/capabilities/vlm_ocr.py +85 -0
- rag_wright/contracts/__init__.py +6 -0
- rag_wright/contracts/chunk.py +79 -0
- rag_wright/contracts/compliance.py +303 -0
- rag_wright/contracts/contract_meta.py +27 -0
- rag_wright/contracts/extraction.py +130 -0
- rag_wright/contracts/function.py +167 -0
- rag_wright/contracts/function_routing.py +91 -0
- rag_wright/contracts/highlight.py +74 -0
- rag_wright/contracts/identifiers.py +153 -0
- rag_wright/contracts/jurisdiction.py +96 -0
- rag_wright/contracts/ontology.py +142 -0
- rag_wright/contracts/property.py +201 -0
- rag_wright/contracts/provenance.py +78 -0
- rag_wright/contracts/query_intent.py +53 -0
- rag_wright/contracts/span.py +76 -0
- rag_wright/contracts/value_match.py +84 -0
- rag_wright/corpus/__init__.py +0 -0
- rag_wright/corpus/canonicalize.py +116 -0
- rag_wright/corpus/cuad.py +153 -0
- rag_wright/corpus/cuad_ingestion.py +72 -0
- rag_wright/corpus/document_parser.py +299 -0
- rag_wright/corpus/edgar.py +231 -0
- rag_wright/corpus/gcs_ingestion.py +120 -0
- rag_wright/corpus/http.py +110 -0
- rag_wright/corpus/selection.py +152 -0
- rag_wright/mcp/__init__.py +11 -0
- rag_wright/mcp/compliance_server.py +299 -0
- rag_wright/mcp/intra_document_qa_server.py +170 -0
- rag_wright/mcp/relational_qa_server.py +171 -0
- rag_wright/mcp/session_store.py +64 -0
- rag_wright/mcp/typed_property_retrieval_server.py +191 -0
- rag_wright/models/__init__.py +8 -0
- rag_wright/models/profiles.py +331 -0
- rag_wright/models/seam.py +497 -0
- rag_wright/models/tag_structured.py +285 -0
- rag_wright/models/tracing.py +179 -0
- rag_wright/models/usage.py +102 -0
- rag_wright/okf/__init__.py +11 -0
- rag_wright/okf/compile.py +292 -0
- rag_wright/okf/document.py +47 -0
- rag_wright/okf/enrich.py +176 -0
- rag_wright/okf/links.py +190 -0
- rag_wright/okf/lint.py +105 -0
- rag_wright/ontology/__init__.py +6 -0
- rag_wright/ontology/_generated_template_meta.py +60 -0
- rag_wright/ontology/_generated_vocab.py +52 -0
- rag_wright/ontology/clause_template.py +964 -0
- rag_wright/ontology/codegen.py +84 -0
- rag_wright/ontology/compliance_bridge.ttl +186 -0
- rag_wright/ontology/contract_bridge.ttl +2685 -0
- rag_wright/ontology/contract_taxonomy.py +24 -0
- rag_wright/ontology/derive.py +58 -0
- rag_wright/ontology/loader.py +435 -0
- rag_wright/ontology/packs/ftc_16cfr255.ttl +29 -0
- rag_wright/ontology/registry.py +87 -0
- rag_wright/ontology/template_introspect.py +100 -0
- rag_wright/py.typed +0 -0
- rag_wright/reference/__init__.py +2 -0
- rag_wright/reference/compliance.py +41 -0
- rag_wright/reference/contract_seam.py +123 -0
- rag_wright/skills/__init__.py +7 -0
- rag_wright/skills/claim_extraction/SKILL.md +47 -0
- rag_wright/skills/claim_extraction/__init__.py +1 -0
- rag_wright/skills/claim_extraction/template.py +50 -0
- rag_wright/skills/compliance_judgment/SKILL.md +59 -0
- rag_wright/skills/corpus_ingest/SKILL.md +106 -0
- rag_wright/skills/extraction_semantic_judge/SKILL.md +51 -0
- rag_wright/skills/extraction_semantic_judge/__init__.py +1 -0
- rag_wright/skills/generation/SKILL.md +64 -0
- rag_wright/skills/generation/__init__.py +1 -0
- rag_wright/skills/generic_compliance_judgment/SKILL.md +58 -0
- rag_wright/skills/okf_navigate/SKILL.md +137 -0
- rag_wright/skills/requirement_extraction/SKILL.md +47 -0
- rag_wright/skills/requirement_extraction/__init__.py +1 -0
- rag_wright/skills/requirement_extraction/template.py +50 -0
- rag_wright/skills/rlm/SKILL.md +186 -0
- rag_wright/skills/rlm/__init__.py +31 -0
- rag_wright/skills/rlm/agent.py +292 -0
- rag_wright/skills/span_relevance_judgment/SKILL.md +67 -0
- rag_wright/skills/vision_to_text/SKILL.md +36 -0
- rag_wright/skills/vision_to_text/__init__.py +1 -0
- rag_wright/spans/__init__.py +1 -0
- rag_wright/spans/boundary.py +78 -0
- rag_wright/spans/clause_function_classifier.py +490 -0
- rag_wright/spans/clause_kg_extractor.py +337 -0
- rag_wright/spans/cuad_labels.py +81 -0
- rag_wright/spans/dim_classifier.py +158 -0
- rag_wright/spans/dim_fleet.json +411 -0
- rag_wright/spans/function_classifier.py +77 -0
- rag_wright/spans/function_families.py +62 -0
- rag_wright/spans/hybrid_classifier.py +103 -0
- rag_wright/spans/legalbert_classifier.py +83 -0
- rag_wright/spans/model_capabilities.py +107 -0
- rag_wright/spans/new_function_labels.py +111 -0
- rag_wright/spans/page_map.py +68 -0
- rag_wright/spans/property_extractor.py +365 -0
- rag_wright/spans/property_grounding.py +182 -0
- rag_wright/spans/reclassify.py +77 -0
- rag_wright/spans/scarce_function_labels.py +105 -0
- rag_wright/spans/segment.py +341 -0
- rag_wright/spans/semantic_judge.py +197 -0
- rag_wright/spans/symbolic_validation.py +131 -0
- rag_wright/spans/tag_clause_extractor.py +182 -0
- rag_wright/store/__init__.py +6 -0
- rag_wright/store/arcadedb.py +1135 -0
- rag_wright/store/chunk_text.py +66 -0
- rag_wright/store/seam.py +213 -0
- rag_wright/subgraphs/__init__.py +0 -0
- rag_wright/subgraphs/async_ingestion.py +204 -0
- rag_wright/subgraphs/compliance_check.py +1042 -0
- rag_wright/subgraphs/compliance_ingestion.py +306 -0
- rag_wright/subgraphs/contract_ingestion_pipeline.py +999 -0
- rag_wright/subgraphs/graph_extraction.py +102 -0
- rag_wright/subgraphs/intra_document_qa.py +328 -0
- rag_wright/subgraphs/observability.py +140 -0
- rag_wright/subgraphs/query_constraint_extraction.py +73 -0
- rag_wright/subgraphs/relational_qa.py +165 -0
- rag_wright/subgraphs/requirement_extraction.py +137 -0
- rag_wright/subgraphs/scaffold.py +65 -0
- rag_wright/subgraphs/semantic_chunking.py +183 -0
- rag_wright/subgraphs/typed_clause_extraction.py +172 -0
- rag_wright/subgraphs/typed_property_retrieval.py +278 -0
- rag_wright/util/__init__.py +1 -0
- rag_wright/util/concurrent.py +153 -0
- rag_wright/util/spacy_model.py +45 -0
- rag_wright-0.1.0.dist-info/METADATA +168 -0
- rag_wright-0.1.0.dist-info/RECORD +184 -0
- rag_wright-0.1.0.dist-info/WHEEL +4 -0
- rag_wright-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,130 @@
|
|
|
1
|
+
"""The graph extraction contract and extractor seam (FR-C.6, FR-I.4).
|
|
2
|
+
|
|
3
|
+
Graph extraction is a hybrid stack (FR-C.6): docling-graph contract extraction for schema entities,
|
|
4
|
+
a lightweight NER-plus-dependency path for the bulk, an open-ended language-model escalation for
|
|
5
|
+
hard cases, and later Open Information Extraction (OpenIE). This module is the *contract* those
|
|
6
|
+
extractors conform to, not the extractors themselves (those are the graph-extraction capability,
|
|
7
|
+
T23).
|
|
8
|
+
|
|
9
|
+
The load-bearing part is the seam: `Extractor` is a real interface, and `run_extractors` iterates a
|
|
10
|
+
list of extractors and merges their results. Adding a new extractor (the deferred OpenIE path) is
|
|
11
|
+
just appending an `Extractor` to that list; neither `run_extractors` nor `ExtractionResult` is
|
|
12
|
+
reopened. Every extractor yields an `ExtractionResult` whose facts conform to the ontology (T4) and
|
|
13
|
+
are anchored to the originating `chunk_id` (FR-I.4).
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
from collections.abc import Iterable, Sequence
|
|
19
|
+
from typing import Protocol, runtime_checkable
|
|
20
|
+
|
|
21
|
+
from pydantic import BaseModel, field_validator, model_validator
|
|
22
|
+
|
|
23
|
+
from rag_wright.contracts.identifiers import ChunkId
|
|
24
|
+
from rag_wright.contracts.ontology import ClauseFact, RelationshipFact
|
|
25
|
+
from rag_wright.contracts.provenance import ConfidenceTag
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
class EntityMention(BaseModel):
|
|
29
|
+
"""A pre-resolution entity mention: a surface form, its ontology type, and a confidence tag.
|
|
30
|
+
|
|
31
|
+
Graph extraction produces typed mentions (NER labels); entity resolution (FR-C.7 / T24) later
|
|
32
|
+
maps the surface form to a canonical `entity_id` and creates the canonical `EntityNode`.
|
|
33
|
+
|
|
34
|
+
A mention IS an ontology-conforming graph fact (FR-S.4): it is read from the text, so a spaCy NER
|
|
35
|
+
or contract-extracted mention carries a `confidence` tag like any other fact (ADR-0012). Its
|
|
36
|
+
`chunk_id` provenance is the containing `ExtractionResult.chunk_id` (mentions are anchored by the
|
|
37
|
+
result, not individually provenanced, since resolution collapses many mentions to one node). This
|
|
38
|
+
is deliberately how the spaCy path satisfies "each path produces facts carrying chunk_id +
|
|
39
|
+
confidence" — by emitting confidence-bearing mentions, NOT by inventing edges: co-occurrence of
|
|
40
|
+
two organizations in legal text is frequently non-contractual (a non-compete, a governing-law or
|
|
41
|
+
payment-clause reference), so a proximity edge is a false-edge generator, and nothing downstream
|
|
42
|
+
filters edges (T26 surfaces confidence, it does not gate on it — FR-C.5/FR-Q.3). CONTRACTS_WITH
|
|
43
|
+
comes from signing-party structure (the contract extractor), never proximity (ADR-0012).
|
|
44
|
+
|
|
45
|
+
`text` here is the *same notion* as a `RelationshipFact`'s `source_ref` / `target_ref`: both are
|
|
46
|
+
pre-resolution entity surface forms. Standalone mentions and relationship endpoints are two
|
|
47
|
+
channels for the same entities, so entity resolution (T24) must resolve them as one mention
|
|
48
|
+
stream; an entity appearing as both must resolve to a single node, not a duplicate.
|
|
49
|
+
"""
|
|
50
|
+
|
|
51
|
+
text: str
|
|
52
|
+
entity_type: str # opaque domain entity type (DD-5); the caller/domain pack names it
|
|
53
|
+
confidence: ConfidenceTag
|
|
54
|
+
|
|
55
|
+
@field_validator("text")
|
|
56
|
+
@classmethod
|
|
57
|
+
def _text_non_empty(cls, v: str) -> str:
|
|
58
|
+
if not v.strip():
|
|
59
|
+
raise ValueError("entity mention text must be non-empty")
|
|
60
|
+
return v
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
class ExtractionResult(BaseModel):
|
|
64
|
+
"""What one extractor produces for one chunk: ontology-conforming facts plus typed mentions.
|
|
65
|
+
|
|
66
|
+
Every fact is anchored to `chunk_id`: its provenance must point at this chunk, so the extraction
|
|
67
|
+
result carries the originating `chunk_id` end to end (FR-I.4). Facts already conform to the
|
|
68
|
+
ontology (their type fields are the T4 enums), so a non-ontology fact cannot be built at all.
|
|
69
|
+
"""
|
|
70
|
+
|
|
71
|
+
chunk_id: ChunkId
|
|
72
|
+
entity_mentions: list[EntityMention] = []
|
|
73
|
+
clause_facts: list[ClauseFact] = []
|
|
74
|
+
relationship_facts: list[RelationshipFact] = []
|
|
75
|
+
|
|
76
|
+
@model_validator(mode="after")
|
|
77
|
+
def _facts_anchored_to_chunk(self) -> ExtractionResult:
|
|
78
|
+
for fact in (*self.clause_facts, *self.relationship_facts):
|
|
79
|
+
if fact.provenance.chunk_id != self.chunk_id:
|
|
80
|
+
raise ValueError(
|
|
81
|
+
"every fact in an ExtractionResult must be anchored to the result's chunk_id "
|
|
82
|
+
"(fact provenance chunk_id does not match)"
|
|
83
|
+
)
|
|
84
|
+
return self
|
|
85
|
+
|
|
86
|
+
@classmethod
|
|
87
|
+
def merge(cls, chunk_id: ChunkId, results: Sequence[ExtractionResult]) -> ExtractionResult:
|
|
88
|
+
"""Merge several extractors' results for one chunk into a single result.
|
|
89
|
+
|
|
90
|
+
All results must be for `chunk_id`; a result for another chunk is a defect and is rejected.
|
|
91
|
+
Merging is a union (dedup, if any, is entity resolution's and graph storage's concern).
|
|
92
|
+
"""
|
|
93
|
+
for result in results:
|
|
94
|
+
if result.chunk_id != chunk_id:
|
|
95
|
+
raise ValueError("cannot merge extraction results from different chunks")
|
|
96
|
+
return cls(
|
|
97
|
+
chunk_id=chunk_id,
|
|
98
|
+
entity_mentions=[m for r in results for m in r.entity_mentions],
|
|
99
|
+
clause_facts=[f for r in results for f in r.clause_facts],
|
|
100
|
+
relationship_facts=[f for r in results for f in r.relationship_facts],
|
|
101
|
+
)
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
@runtime_checkable
|
|
105
|
+
class Extractor(Protocol):
|
|
106
|
+
"""The extractor seam. Each extractor in the hybrid stack (FR-C.6) implements this, and the
|
|
107
|
+
graph-extraction capability (T23) iterates over a list of them. The deferred OpenIE path is a
|
|
108
|
+
future `Extractor` added to that list, behind this same contract, with no change here.
|
|
109
|
+
|
|
110
|
+
Note: `@runtime_checkable` makes `isinstance(x, Extractor)` a *presence* check only (it verifies
|
|
111
|
+
`extract` and `name` exist, not their signatures or return type). Signature and output
|
|
112
|
+
conformance are enforced downstream by `ExtractionResult` validation, which is what the
|
|
113
|
+
result-validation tests exercise, not `isinstance`.
|
|
114
|
+
"""
|
|
115
|
+
|
|
116
|
+
name: str
|
|
117
|
+
|
|
118
|
+
def extract(self, chunk_id: ChunkId, text: str) -> ExtractionResult: ...
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def run_extractors(
|
|
122
|
+
extractors: Iterable[Extractor], chunk_id: ChunkId, text: str
|
|
123
|
+
) -> ExtractionResult:
|
|
124
|
+
"""Run every extractor over one chunk and merge into a single anchored `ExtractionResult`.
|
|
125
|
+
|
|
126
|
+
This is the seam the capability drives: registering a new extractor means adding it to
|
|
127
|
+
`extractors`, nothing here changes.
|
|
128
|
+
"""
|
|
129
|
+
results = [extractor.extract(chunk_id, text) for extractor in extractors]
|
|
130
|
+
return ExtractionResult.merge(chunk_id, results)
|
|
@@ -0,0 +1,167 @@
|
|
|
1
|
+
"""The retrieval FUNCTION taxonomy (T57, FR-C.6, ADR-0025).
|
|
2
|
+
|
|
3
|
+
The set of clause types the FUNCTION classifier (T56) routes on. It is a SUPERSET of the 41 CUAD
|
|
4
|
+
`ClauseCategory` (mirrored by value, so there is no drift) plus the three ACORD query families CUAD
|
|
5
|
+
has no class for: Indemnification (14/57 test queries), the indirect/consequential damages waiver,
|
|
6
|
+
and the warranty disclaimer (the bulk of the "Limitation of Liability" family beyond Cap/Uncapped).
|
|
7
|
+
|
|
8
|
+
This is kept DISTINCT from `ClauseCategory` on purpose. `ClauseCategory` is the graph EXTRACTION
|
|
9
|
+
ontology (ADR-0002): a closed vocabulary the knowledge graph conforms to, not reopened here. The
|
|
10
|
+
FUNCTION taxonomy is a RETRIEVAL concern (which clause type a span is routed under). The two serve
|
|
11
|
+
different layers even though 41 labels coincide by value, so extending function routing does not
|
|
12
|
+
reopen the extraction ontology.
|
|
13
|
+
|
|
14
|
+
`FUNCTION_LABELS` is the classifier's full label space and its retrain target (the T56 LegalBERT is
|
|
15
|
+
retrained over these classes plus its own NONE sentinel; NONE is not a function type and is not listed
|
|
16
|
+
here). ADR-0048 step 2 grew it 44 -> 52 with 8 taxonomy-gap functions the full-corpus classifier
|
|
17
|
+
surfaced (curated LLM-assisted with human oversight), plus a curated FOLD alias map (`_FUNCTION_ALIASES`)
|
|
18
|
+
that resolves recurring synonyms to their canonical label.
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
from __future__ import annotations
|
|
22
|
+
|
|
23
|
+
from enum import Enum
|
|
24
|
+
|
|
25
|
+
from pydantic import BaseModel, field_validator
|
|
26
|
+
|
|
27
|
+
from rag_wright.contracts.ontology import ClauseCategory
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
class FunctionClassification(BaseModel):
|
|
31
|
+
"""CAP-REG-2: the ranked FUNCTION_LABELS a span or query is classified into (most relevant first).
|
|
32
|
+
The shared output contract of the LegalBERT `clause_function_classification` (model) and the
|
|
33
|
+
taxonomy-constrained `query_function_classification` (agent_skill)."""
|
|
34
|
+
|
|
35
|
+
labels: list[str]
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
class ExtendedFunction(str, Enum):
|
|
39
|
+
"""The ACORD query families the 41 CUAD `ClauseCategory` has no class for (T57). Values are the
|
|
40
|
+
canonical function labels the classifier emits and the query-decomposer targets."""
|
|
41
|
+
|
|
42
|
+
INDEMNIFICATION = "Indemnification"
|
|
43
|
+
INDIRECT_DAMAGES_WAIVER = "Indirect/Consequential Damages Waiver"
|
|
44
|
+
WARRANTY_DISCLAIMER = "Warranty Disclaimer"
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
class TaxonomyGapFunction(str, Enum):
|
|
48
|
+
"""INGEST-LLM-CLASSIFIER step 2 (ADR-0048): 8 clause functions the full-corpus LLM classifier surfaced as
|
|
49
|
+
recurring OTHER (out-of-taxonomy) clause types across the unified CUAD+ACORD corpus, then curated LLM-assisted
|
|
50
|
+
with human oversight (`data/eval/taxonomy_gaps/curation_proposal.json`, my recommended 8-ADD delta approved).
|
|
51
|
+
Genuinely distinct from the 44 CUAD/ACORD labels: CUAD has no generic Confidentiality, Royalties, Payment
|
|
52
|
+
Terms, Force Majeure, Dispute Resolution, Record Retention, Security Interest, or Condition Precedent class."""
|
|
53
|
+
|
|
54
|
+
CONFIDENTIALITY = "Confidentiality"
|
|
55
|
+
ROYALTIES = "Royalties"
|
|
56
|
+
PAYMENT_TERMS = "Payment Terms"
|
|
57
|
+
DISPUTE_RESOLUTION = "Dispute Resolution"
|
|
58
|
+
RECORD_RETENTION = "Record Retention"
|
|
59
|
+
SECURITY_INTEREST = "Security Interest"
|
|
60
|
+
CONDITION_PRECEDENT = "Condition Precedent"
|
|
61
|
+
FORCE_MAJEURE = "Force Majeure"
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
# The classifier's full label space: the 41 CUAD categories (by value, no drift), then the 3 ACORD
|
|
65
|
+
# extensions, then the 8 ADR-0048 taxonomy-gap additions. Order is stable (CUAD, extension, gap) so a
|
|
66
|
+
# retrain's label<->id map is reproducible. NONE is the off-taxonomy sentinel, not a function -> not listed.
|
|
67
|
+
FUNCTION_LABELS: tuple[str, ...] = tuple(
|
|
68
|
+
[c.value for c in ClauseCategory]
|
|
69
|
+
+ [f.value for f in ExtendedFunction]
|
|
70
|
+
+ [f.value for f in TaxonomyGapFunction]
|
|
71
|
+
)
|
|
72
|
+
FUNCTION_LABEL_SET: frozenset[str] = frozenset(FUNCTION_LABELS)
|
|
73
|
+
|
|
74
|
+
# The no-clause-function sentinel (the classifier's off-taxonomy NONE). NOT a function type, so NOT in
|
|
75
|
+
# FUNCTION_LABELS. A `ClausePropertyRecord` carries it ONLY for a QUERY-constraint record (a query has no clause
|
|
76
|
+
# function -- only its extracted properties matter); a real ingested clause never uses it (the ingest extracts
|
|
77
|
+
# clauses only for canonical functions).
|
|
78
|
+
NO_FUNCTION: str = "NONE"
|
|
79
|
+
|
|
80
|
+
# The classifier was trained on CUAD's label strings, which differ in CASE from the canonical taxonomy for
|
|
81
|
+
# a few labels (e.g. CUAD "Ip Ownership Assignment" vs the canonical "IP Ownership Assignment"). Normalize
|
|
82
|
+
# the classifier output to the canonical label at the boundary (memory: normalize at the boundary), keyed
|
|
83
|
+
# case-insensitively.
|
|
84
|
+
_FUNCTION_BY_CASEFOLD: dict[str, str] = {label.casefold(): label for label in FUNCTION_LABELS}
|
|
85
|
+
|
|
86
|
+
# ADR-0048 step 2: the curated FOLD map -- recurring synonyms/spelling/variant clause types the full-corpus
|
|
87
|
+
# classifier surfaced, each mapped to its canonical `FUNCTION_LABELS` entry (LLM-proposed, human-reconciled:
|
|
88
|
+
# the 3 LLM mis-folds were dropped, the royalty family retargeted to the new Royalties label, and RoFR +
|
|
89
|
+
# Milestone Payment folded rather than dropped). Authored in code (no external map to drift), keyed by
|
|
90
|
+
# canonical target for readability; inverted + casefolded into `_FUNCTION_ALIAS_BY_CASEFOLD` below.
|
|
91
|
+
_FUNCTION_ALIASES: dict[str, tuple[str, ...]] = {
|
|
92
|
+
"Cap On Liability": ("Limitation of Liability", "Limitations of Liability", "Liability Limitation"),
|
|
93
|
+
"Anti-Assignment": ("Assignment", "Assignment of Capacity", "Non-Assignment"),
|
|
94
|
+
"Termination For Convenience": (
|
|
95
|
+
"Termination", "Termination For Cause", "Effect of Termination", "Termination Clause",
|
|
96
|
+
"Cancellation", "Termination Effects", "Rights and Obligations Upon Termination"),
|
|
97
|
+
"Expiration Date": ("Term",),
|
|
98
|
+
"Insurance": ("Insurance Requirement", "Insurance Type", "Insurance Coverage Details"),
|
|
99
|
+
"No-Solicit Of Employees": ("Non-Solicit Of Employees",),
|
|
100
|
+
"No-Solicit Of Customers": ("Non-Solicit Of Customers",),
|
|
101
|
+
"Warranty Duration": ("Warranty Grant", "Product Warranty", "Warranty"),
|
|
102
|
+
"Warranty Disclaimer": (
|
|
103
|
+
"Disclaimer of Warranty", "Disclaimer of Representations and Warranties", "Disclaimer",
|
|
104
|
+
"Liability Disclaimer"),
|
|
105
|
+
"Liquidated Damages": ("Penalty", "Make-Whole Payment"),
|
|
106
|
+
"Indemnification": ("Intellectual Property Indemnification", "Indemnity and Limitation of Liability"),
|
|
107
|
+
"License Grant": (
|
|
108
|
+
"Content License Restrictions", "Use Restrictions", "License Grant Restrictions", "Restriction Of Use"),
|
|
109
|
+
"Notice Period To Terminate Renewal": ("Notice Period To Terminate",),
|
|
110
|
+
"Non-Compete": ("Non-Compete Exception",),
|
|
111
|
+
"Governing Law": ("Jurisdiction",),
|
|
112
|
+
"Audit Rights": ("Inspection", "Inspection Rights"),
|
|
113
|
+
"IP Ownership Assignment": (
|
|
114
|
+
"Ownership", "Domain Name Assignment", "Intellectual Property Rights", "Patents",
|
|
115
|
+
"Intellectual Property Assignment"),
|
|
116
|
+
"Revenue/Profit Sharing": ("Revenue Sharing", "Payment/Revenue Sharing"),
|
|
117
|
+
"Competitive Restriction Exception": (
|
|
118
|
+
"Definition of Class C Breaches", "Other Restriction Exception", "Other Restriction"),
|
|
119
|
+
"Rofr/Rofo/Rofn": ("Right of First Refusal",),
|
|
120
|
+
# the royalty family consolidates into the new Royalties label; Milestone Payment -> the new Payment Terms
|
|
121
|
+
"Royalties": ("Royalty Grant", "Royalty", "Royalty Obligation", "Royalty Payment"),
|
|
122
|
+
"Payment Terms": ("Milestone Payment",),
|
|
123
|
+
}
|
|
124
|
+
_FUNCTION_ALIAS_BY_CASEFOLD: dict[str, str] = {
|
|
125
|
+
alias.casefold(): canon for canon, aliases in _FUNCTION_ALIASES.items() for alias in aliases
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def canonical_function(label: str) -> str | None:
|
|
130
|
+
"""Map a function label to its canonical `FUNCTION_LABELS` entry, or None if it matches none. Case-
|
|
131
|
+
insensitive, and resolves the ADR-0048 curated FOLD aliases (e.g. 'Limitation of Liability' -> 'Cap On
|
|
132
|
+
Liability', 'Royalty Grant' -> 'Royalties'). Exact/cased match wins over an alias."""
|
|
133
|
+
key = label.strip().casefold()
|
|
134
|
+
return _FUNCTION_BY_CASEFOLD.get(key) or _FUNCTION_ALIAS_BY_CASEFOLD.get(key)
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
class FunctionConfidence(str, Enum):
|
|
138
|
+
"""INGEST-LLM-CLASSIFIER (ADR-0048): the LLM clause classifier's coarse confidence in a function assignment.
|
|
139
|
+
Ordinal, not a float -- LLMs are not calibrated on numeric self-confidence; a floor (>= medium) filters weak
|
|
140
|
+
labels, so a clause with one clear function stays single while a genuinely mixed clause keeps 2-3."""
|
|
141
|
+
|
|
142
|
+
HIGH = "high"
|
|
143
|
+
MEDIUM = "medium"
|
|
144
|
+
LOW = "low"
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
class FunctionScore(BaseModel):
|
|
148
|
+
"""INGEST-LLM-CLASSIFIER (ADR-0048): one function a clause is classified into, with coarse confidence. In a
|
|
149
|
+
ranked list the first is the PRIMARY (the label kept on `Clause.function` for the query legs). `function` is
|
|
150
|
+
normalized to its canonical `FUNCTION_LABELS` entry at the boundary; a non-canonical label (incl. the NONE
|
|
151
|
+
sentinel) is rejected (strict contract; normalize upstream)."""
|
|
152
|
+
|
|
153
|
+
function: str
|
|
154
|
+
confidence: FunctionConfidence
|
|
155
|
+
|
|
156
|
+
@field_validator("function")
|
|
157
|
+
@classmethod
|
|
158
|
+
def _canonicalize(cls, v: str) -> str:
|
|
159
|
+
canon = canonical_function(v)
|
|
160
|
+
if canon is None:
|
|
161
|
+
raise ValueError(f"function must be a canonical FUNCTION_LABELS label, got {v!r}")
|
|
162
|
+
return canon
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
def primary_function(scores: list[FunctionScore]) -> str | None:
|
|
166
|
+
"""The PRIMARY (highest-ranked) function of a ranked `FunctionScore` list (primary first), or None if empty."""
|
|
167
|
+
return scores[0].function if scores else None
|
|
@@ -0,0 +1,91 @@
|
|
|
1
|
+
"""KG-5e (FR-Q): the query-side FUNCTION ROUTER, driven by the granite-extracted DIMENSIONS.
|
|
2
|
+
|
|
3
|
+
KG-5c found that query-side function routing is the dominant retrieval lever (the oracle->real gap is
|
|
4
|
+
-0.22 recall@20, dwarfing every reranker lever) and that the clause-trained LegalBERT classifier is
|
|
5
|
+
out-of-distribution on short query text. This routes instead from the query's typed DIMENSIONS -- which
|
|
6
|
+
KG-5b showed the granite extraction gets right even where `clause_type` is unreliable -- through a
|
|
7
|
+
corpus-derived `dimension -> function` co-occurrence prior. No extra LLM call, no clause-trained model:
|
|
8
|
+
the typed extraction we already compute builds the pool.
|
|
9
|
+
|
|
10
|
+
The prior is built from a HELD-OUT corpus (CUAD) so no ACORD eval data enters the router; the two corpora
|
|
11
|
+
share the property-dimension schema and the CUAD-type function taxonomy, so the prior transfers.
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
import math
|
|
17
|
+
from collections import defaultdict
|
|
18
|
+
from typing import Iterable
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def build_cooccurrence(clause_rows: Iterable[tuple[str, str, str]]) -> dict[str, dict[str, int]]:
|
|
22
|
+
"""`dimension -> {function -> clause-count}` from `(clause_id, function, dimension)` typed-edge rows.
|
|
23
|
+
|
|
24
|
+
Counted once per (clause, dimension, function) so a clause with several values for one dimension does not
|
|
25
|
+
over-weight its function. An empty/`NONE` function is dropped (not a routable target). Pure -- no store.
|
|
26
|
+
"""
|
|
27
|
+
seen: set[tuple[str, str, str]] = set()
|
|
28
|
+
cooc: dict[str, dict[str, int]] = defaultdict(lambda: defaultdict(int))
|
|
29
|
+
for clause_id, function, dimension in clause_rows:
|
|
30
|
+
if not function or function == "NONE" or not dimension:
|
|
31
|
+
continue
|
|
32
|
+
key = (clause_id, dimension, function)
|
|
33
|
+
if key in seen:
|
|
34
|
+
continue
|
|
35
|
+
seen.add(key)
|
|
36
|
+
cooc[dimension][function] += 1
|
|
37
|
+
return {d: dict(fs) for d, fs in cooc.items()}
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def _function_marginal(cooc: dict[str, dict[str, int]]) -> dict[str, float]:
|
|
41
|
+
"""P(function) proxy: each function's share of all edges in the prior -- the base rate that lift/PMI
|
|
42
|
+
divide out so a distinctive dimension beats a merely-common function."""
|
|
43
|
+
tot: dict[str, float] = defaultdict(float)
|
|
44
|
+
grand = 0.0
|
|
45
|
+
for col in cooc.values():
|
|
46
|
+
for fn, c in col.items():
|
|
47
|
+
tot[fn] += c
|
|
48
|
+
grand += c
|
|
49
|
+
return {fn: c / grand for fn, c in tot.items()} if grand else {}
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def route_functions(
|
|
53
|
+
query_dimensions: Iterable[str],
|
|
54
|
+
cooc: dict[str, dict[str, int]],
|
|
55
|
+
*,
|
|
56
|
+
k: int,
|
|
57
|
+
score: str = "conditional",
|
|
58
|
+
min_support: int = 1,
|
|
59
|
+
) -> list[str]:
|
|
60
|
+
"""The top-`k` functions for a query, scored by summing a per-dimension signal over the query's dimensions:
|
|
61
|
+
|
|
62
|
+
- ``conditional`` (default): ``P(function | dimension)`` -- biased toward high-frequency functions.
|
|
63
|
+
- ``lift``: ``P(function|dimension) / P(function)`` -- corrects for the function base rate (a dimension
|
|
64
|
+
routes to the function it is DISTINCTIVE of, not merely the most common one that has it).
|
|
65
|
+
- ``pmi``: ``log(P(function|dimension) / P(function))`` -- the log-odds form of lift.
|
|
66
|
+
|
|
67
|
+
`min_support` drops sparse `(dimension, function)` evidence (< that many clauses) so lift/PMI are not blown
|
|
68
|
+
up by a single-clause coincidence. A dimension with no corpus signal contributes nothing; no dimensions
|
|
69
|
+
(or none seen) -> ``[]``. Ties break on the function name so the routing is deterministic.
|
|
70
|
+
"""
|
|
71
|
+
marginal = _function_marginal(cooc) if score in ("lift", "pmi") else {}
|
|
72
|
+
total_score: dict[str, float] = defaultdict(float)
|
|
73
|
+
for d in query_dimensions:
|
|
74
|
+
col = cooc.get(d)
|
|
75
|
+
if not col:
|
|
76
|
+
continue
|
|
77
|
+
total = sum(col.values())
|
|
78
|
+
if not total:
|
|
79
|
+
continue
|
|
80
|
+
for fn, c in col.items():
|
|
81
|
+
if c < min_support:
|
|
82
|
+
continue
|
|
83
|
+
p_f_given_d = c / total
|
|
84
|
+
if score == "conditional":
|
|
85
|
+
total_score[fn] += p_f_given_d
|
|
86
|
+
else:
|
|
87
|
+
p_f = marginal.get(fn, 0.0)
|
|
88
|
+
if p_f <= 0:
|
|
89
|
+
continue
|
|
90
|
+
total_score[fn] += p_f_given_d / p_f if score == "lift" else math.log(p_f_given_d / p_f)
|
|
91
|
+
return [fn for fn, _ in sorted(total_score.items(), key=lambda x: (-x[1], x[0]))[:k]]
|
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
"""The serve-side highlight response contract (CU-A1 / CU-C2, ADR-0029).
|
|
2
|
+
|
|
3
|
+
What the pipeline returns for a query about a known contract: the SET of spans that pertain (possibly empty ->
|
|
4
|
+
"not present"), each with its exact document location so an app can highlight it, plus provenance for citation.
|
|
5
|
+
`HighlightResult` also carries the presence/absence and out-of-taxonomy/low-confidence signals the app/agent
|
|
6
|
+
routes on. Offsets mirror `SpanRecord` (document-absolute char offsets; optional page/bbox).
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from pydantic import BaseModel, model_validator
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class HighlightSpan(BaseModel):
|
|
15
|
+
"""One span to highlight, with its citation (location + parent-clause reference)."""
|
|
16
|
+
|
|
17
|
+
model_config = {"frozen": True}
|
|
18
|
+
|
|
19
|
+
span_id: str
|
|
20
|
+
contract_id: str
|
|
21
|
+
function: str # the clause type this span was matched under
|
|
22
|
+
clause_ref: str # human-readable parent-clause reference (heading/number or parent_chunk_id)
|
|
23
|
+
text: str
|
|
24
|
+
doc_start: int | None = None # document-absolute char offsets (the highlight range)
|
|
25
|
+
doc_end: int | None = None
|
|
26
|
+
page: int | None = None # optional PDF-overlay location (FIRST page; == pages[0] when known)
|
|
27
|
+
pages: list[int] = [] # issue 0032: ALL source pages this span overlaps (page-level click-through)
|
|
28
|
+
bbox: tuple[float, float, float, float] | None = None
|
|
29
|
+
extracted_value: str | None = None # for value-type categories: the pinpointed value within the span
|
|
30
|
+
confidence: float = 1.0
|
|
31
|
+
|
|
32
|
+
@model_validator(mode="after")
|
|
33
|
+
def _check_offsets(self) -> "HighlightSpan":
|
|
34
|
+
if self.doc_start is not None and self.doc_end is not None and self.doc_end < self.doc_start:
|
|
35
|
+
raise ValueError(f"doc_end ({self.doc_end}) must be >= doc_start ({self.doc_start})")
|
|
36
|
+
return self
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
class SpanLocation(BaseModel):
|
|
40
|
+
"""Where one span sits in the ORIGINAL document, for a citation PREVIEW (EP-REF-1b): a location + the text
|
|
41
|
+
to confirm it landed, plus the clause ids extracted from it (so an answer's `citation_id` -- a span id OR a
|
|
42
|
+
clause id, two different spaces -- resolves either way). Lighter than `HighlightSpan` (no function / clause
|
|
43
|
+
ref / extracted value / confidence): a preview answers "show me this span", not "which spans pertain". `pages`
|
|
44
|
+
is a LIST (a span can cross a page break; the first page is where the preview opens) and `bbox` is best-effort
|
|
45
|
+
-- a page is nearly always known and a rectangle usually is, so a missing rectangle never costs the page."""
|
|
46
|
+
|
|
47
|
+
span_id: str
|
|
48
|
+
clause_ids: list[str] = []
|
|
49
|
+
pages: list[int] = []
|
|
50
|
+
bbox: tuple[float, float, float, float] | None = None
|
|
51
|
+
doc_start: int | None = None
|
|
52
|
+
doc_end: int | None = None
|
|
53
|
+
text: str = ""
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
class HighlightResult(BaseModel):
|
|
57
|
+
"""The full response to one query about one contract."""
|
|
58
|
+
|
|
59
|
+
model_config = {"frozen": True}
|
|
60
|
+
|
|
61
|
+
query: str
|
|
62
|
+
contract_id: str
|
|
63
|
+
clause_types: list[str] = [] # the types searched (from QueryIntent)
|
|
64
|
+
intent: str = "highlight"
|
|
65
|
+
spans: list[HighlightSpan] = [] # the pertaining set; empty => not present
|
|
66
|
+
present: bool = False # spans is non-empty
|
|
67
|
+
in_taxonomy: bool = True
|
|
68
|
+
low_confidence: bool = False # out-of-taxonomy semantic fallback used
|
|
69
|
+
|
|
70
|
+
@model_validator(mode="after")
|
|
71
|
+
def _check_present(self) -> "HighlightResult":
|
|
72
|
+
if self.present != bool(self.spans):
|
|
73
|
+
raise ValueError("`present` must equal whether `spans` is non-empty")
|
|
74
|
+
return self
|
|
@@ -0,0 +1,153 @@
|
|
|
1
|
+
"""Shared identifier contracts: `chunk_id` (FR-S.2) and `entity_id` (FR-S.3).
|
|
2
|
+
|
|
3
|
+
These schemes are load-bearing and fixed here before anything is built. A re-chunk that changes
|
|
4
|
+
a `chunk_id` breaks the link between a chunk and its extracted graph nodes, and a non-canonical
|
|
5
|
+
`entity_id` fragments the graph across surface-form variants. Changing either scheme is an
|
|
6
|
+
ask-first change (SPEC.md section 14; CLAUDE.md boundaries).
|
|
7
|
+
|
|
8
|
+
Both identifiers are frozen Pydantic models, so they are immutable and hashable and can serve as
|
|
9
|
+
dictionary keys and graph-node identity directly.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import hashlib
|
|
15
|
+
import re
|
|
16
|
+
|
|
17
|
+
from pydantic import BaseModel, ConfigDict, field_validator
|
|
18
|
+
|
|
19
|
+
_SHA256_HEX = re.compile(r"^[0-9a-f]{64}$")
|
|
20
|
+
_SOURCE_DOC_ID = re.compile(r"^[A-Za-z0-9._-]+$")
|
|
21
|
+
_SOURCE_DOC_UNSAFE = re.compile(r"[^A-Za-z0-9._-]+")
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def canonical_source_doc_id(raw: str) -> str:
|
|
25
|
+
"""The ONE canonical filename/title -> `source_doc_id` slug (FR-S.2; HYG-1).
|
|
26
|
+
|
|
27
|
+
Every ingestion path MUST derive a `source_doc_id` through this function so the same document gets the
|
|
28
|
+
same id everywhere. Any run of characters outside the delimiter-safe set ``[A-Za-z0-9._-]`` (notably
|
|
29
|
+
spaces, ``&``, commas) collapses to a single ``_``; leading/trailing ``_`` are stripped. Existing safe
|
|
30
|
+
delimiters (``-``, ``.``, ``_`` -- e.g. inside ``EX-10.1`` / ``10-Q``) are preserved. Idempotent on an
|
|
31
|
+
already-canonical id.
|
|
32
|
+
|
|
33
|
+
The ``_`` replacement (never ``-``) is the fix for the HYG-1 divergence: two ingestion paths slugged the
|
|
34
|
+
same title with different characters (``FLEET_MAINTENANCE`` vs ``FLEET-MAINTENANCE``), breaking the
|
|
35
|
+
cross-graph join. An empty result raises rather than silently colliding every empty title into one id.
|
|
36
|
+
"""
|
|
37
|
+
slug = _SOURCE_DOC_UNSAFE.sub("_", raw).strip("_") if isinstance(raw, str) else ""
|
|
38
|
+
if not slug:
|
|
39
|
+
raise ValueError(
|
|
40
|
+
f"source_doc_id slug is empty for {raw!r}; supply a non-empty, sluggable document id"
|
|
41
|
+
)
|
|
42
|
+
return slug
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
class ChunkId(BaseModel):
|
|
46
|
+
"""The stable identifier for a chunk (FR-S.2).
|
|
47
|
+
|
|
48
|
+
Scheme: source-document identifier, chunk index, and a content hash of the chunk text. The
|
|
49
|
+
content hash is what makes the identifier change when (and only when) the chunk content
|
|
50
|
+
changes, so an unchanged document re-chunks to the same ids (the content-hash gate in FR-I.1
|
|
51
|
+
and FR-I.5 relies on this).
|
|
52
|
+
"""
|
|
53
|
+
|
|
54
|
+
model_config = ConfigDict(frozen=True)
|
|
55
|
+
|
|
56
|
+
source_doc_id: str
|
|
57
|
+
chunk_index: int
|
|
58
|
+
content_hash: str # lowercase hex SHA-256 digest of the chunk content
|
|
59
|
+
|
|
60
|
+
@field_validator("source_doc_id")
|
|
61
|
+
@classmethod
|
|
62
|
+
def _delimiter_safe_source(cls, v: str) -> str:
|
|
63
|
+
v = v.strip()
|
|
64
|
+
if not _SOURCE_DOC_ID.match(v):
|
|
65
|
+
raise ValueError(
|
|
66
|
+
"source_doc_id must be non-empty and use only [A-Za-z0-9._-], so the ':'-delimited "
|
|
67
|
+
"value string stays unambiguous for provenance and citation lookup; assign a "
|
|
68
|
+
"delimiter-safe id upstream (slugify the filename if needed)"
|
|
69
|
+
)
|
|
70
|
+
return v
|
|
71
|
+
|
|
72
|
+
@field_validator("chunk_index")
|
|
73
|
+
@classmethod
|
|
74
|
+
def _nonnegative_index(cls, v: int) -> int:
|
|
75
|
+
if v < 0:
|
|
76
|
+
raise ValueError("chunk_index must be >= 0")
|
|
77
|
+
return v
|
|
78
|
+
|
|
79
|
+
@field_validator("content_hash")
|
|
80
|
+
@classmethod
|
|
81
|
+
def _valid_sha256(cls, v: str) -> str:
|
|
82
|
+
v = v.strip().lower()
|
|
83
|
+
if not _SHA256_HEX.match(v):
|
|
84
|
+
raise ValueError("content_hash must be a 64-character lowercase hex SHA-256 digest")
|
|
85
|
+
return v
|
|
86
|
+
|
|
87
|
+
@classmethod
|
|
88
|
+
def of(cls, source_doc_id: str, chunk_index: int, content: str | bytes) -> ChunkId:
|
|
89
|
+
"""Build a `ChunkId`, computing the content hash deterministically (SHA-256).
|
|
90
|
+
|
|
91
|
+
This is where determinism lives (RAC-1): identical `(source_doc_id, chunk_index, content)`
|
|
92
|
+
always yield an identical `ChunkId`.
|
|
93
|
+
"""
|
|
94
|
+
data = content.encode("utf-8") if isinstance(content, str) else content
|
|
95
|
+
return cls(
|
|
96
|
+
source_doc_id=source_doc_id,
|
|
97
|
+
chunk_index=chunk_index,
|
|
98
|
+
content_hash=hashlib.sha256(data).hexdigest(),
|
|
99
|
+
)
|
|
100
|
+
|
|
101
|
+
@property
|
|
102
|
+
def value(self) -> str:
|
|
103
|
+
"""The canonical string form: ``<source_doc_id>:<chunk_index>:<content_hash>``.
|
|
104
|
+
|
|
105
|
+
Because ``source_doc_id`` is constrained to a delimiter-safe character set, this string is
|
|
106
|
+
safe to string-match and to parse back with ``rsplit(":", 2)`` for provenance and citation
|
|
107
|
+
lookup. Identity itself remains field-based (frozen-model equality and hashing), not
|
|
108
|
+
string-based.
|
|
109
|
+
"""
|
|
110
|
+
return f"{self.source_doc_id}:{self.chunk_index}:{self.content_hash}"
|
|
111
|
+
|
|
112
|
+
def __str__(self) -> str:
|
|
113
|
+
return self.value
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
class EntityId(BaseModel):
|
|
117
|
+
"""The canonical entity identifier (FR-S.3): an opaque canonical-registry id string.
|
|
118
|
+
|
|
119
|
+
The engine is domain-agnostic (DD-4, ADR-0067/0117), so the FORMAT of a canonical id is owned by
|
|
120
|
+
the resolver / domain pack, NOT by this contract. The SEC pack resolves to a 10-digit zero-padded
|
|
121
|
+
EDGAR Central Index Key (CIK), e.g. ``"0000320193"`` (shaped in ``corpus/edgar.normalize_cik``); a
|
|
122
|
+
generic pack uses an exact-normalized surface-form key; another domain uses its own scheme. This
|
|
123
|
+
contract's only invariant is therefore the domain-neutral one: a non-empty string. That is still a
|
|
124
|
+
real invariant -- every downstream holder of an `EntityId` can trust it is a present, non-blank id --
|
|
125
|
+
while the format check lives at the one boundary that knows the domain (the resolver/loader), where
|
|
126
|
+
the world's mess actually arrives, per the "normalize at the boundary" rule.
|
|
127
|
+
"""
|
|
128
|
+
|
|
129
|
+
model_config = ConfigDict(frozen=True)
|
|
130
|
+
|
|
131
|
+
value: str # an opaque canonical id; its FORMAT is the resolver/pack's concern, not this contract's
|
|
132
|
+
|
|
133
|
+
@field_validator("value", mode="before")
|
|
134
|
+
@classmethod
|
|
135
|
+
def _nonempty(cls, v: object) -> str:
|
|
136
|
+
if not isinstance(v, str) or not v.strip():
|
|
137
|
+
raise ValueError(
|
|
138
|
+
"EntityId.value must be a non-empty string. The canonical-id FORMAT is owned by the "
|
|
139
|
+
"resolver / domain pack (e.g. corpus/edgar.normalize_cik for SEC CIKs), not this contract."
|
|
140
|
+
)
|
|
141
|
+
return v
|
|
142
|
+
|
|
143
|
+
@classmethod
|
|
144
|
+
def of(cls, value: str) -> EntityId:
|
|
145
|
+
"""Build an `EntityId` from an already-canonical id string.
|
|
146
|
+
|
|
147
|
+
This does not normalize or format-check beyond non-emptiness; shaping the raw domain form into
|
|
148
|
+
the canonical id is the resolver / domain pack's job (e.g. `corpus/edgar.normalize_cik`).
|
|
149
|
+
"""
|
|
150
|
+
return cls(value=value)
|
|
151
|
+
|
|
152
|
+
def __str__(self) -> str:
|
|
153
|
+
return self.value
|