rag-wright 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- rag_wright/__init__.py +13 -0
- rag_wright/api/__init__.py +33 -0
- rag_wright/api/config.py +59 -0
- rag_wright/api/discover.py +70 -0
- rag_wright/api/documents.py +39 -0
- rag_wright/api/ids.py +31 -0
- rag_wright/api/invoke.py +99 -0
- rag_wright/api/kg.py +61 -0
- rag_wright/api/mcp.py +94 -0
- rag_wright/api/usage.py +30 -0
- rag_wright/api/workspace.py +85 -0
- rag_wright/capabilities/__init__.py +8 -0
- rag_wright/capabilities/answer_generator.py +427 -0
- rag_wright/capabilities/ard.py +286 -0
- rag_wright/capabilities/assertion_extraction.py +79 -0
- rag_wright/capabilities/chunk_read.py +58 -0
- rag_wright/capabilities/chunk_write.py +163 -0
- rag_wright/capabilities/claim_extraction.py +153 -0
- rag_wright/capabilities/clause_exception_linking.py +117 -0
- rag_wright/capabilities/compliance_judgment.py +322 -0
- rag_wright/capabilities/compliance_store.py +87 -0
- rag_wright/capabilities/contract_kg_serve.py +156 -0
- rag_wright/capabilities/contract_kg_store.py +251 -0
- rag_wright/capabilities/dg_extraction.py +585 -0
- rag_wright/capabilities/disambiguation.py +163 -0
- rag_wright/capabilities/document_parse.py +87 -0
- rag_wright/capabilities/document_scope.py +49 -0
- rag_wright/capabilities/embedding.py +164 -0
- rag_wright/capabilities/embedding_profiles.py +43 -0
- rag_wright/capabilities/entity_resolution.py +154 -0
- rag_wright/capabilities/fusion.py +64 -0
- rag_wright/capabilities/graph_extraction.py +243 -0
- rag_wright/capabilities/graph_query.py +73 -0
- rag_wright/capabilities/graph_storage.py +111 -0
- rag_wright/capabilities/highlight_serve.py +142 -0
- rag_wright/capabilities/hybrid_search.py +65 -0
- rag_wright/capabilities/invoke.py +31 -0
- rag_wright/capabilities/jev_decision.py +38 -0
- rag_wright/capabilities/manifests.py +872 -0
- rag_wright/capabilities/okf_navigate.py +456 -0
- rag_wright/capabilities/parsing.py +286 -0
- rag_wright/capabilities/property_boosted_retrieval.py +125 -0
- rag_wright/capabilities/query_function_classifier.py +94 -0
- rag_wright/capabilities/query_understanding.py +109 -0
- rag_wright/capabilities/registry.py +262 -0
- rag_wright/capabilities/remote_encoders.py +94 -0
- rag_wright/capabilities/requirement_extraction.py +247 -0
- rag_wright/capabilities/reranking.py +123 -0
- rag_wright/capabilities/retrieval_core.py +126 -0
- rag_wright/capabilities/rlm_chunking.py +808 -0
- rag_wright/capabilities/rlm_synthesis.py +316 -0
- rag_wright/capabilities/scan_quality.py +136 -0
- rag_wright/capabilities/span_relevance_judgment.py +191 -0
- rag_wright/capabilities/vision_to_text.py +85 -0
- rag_wright/capabilities/vlm_ocr.py +85 -0
- rag_wright/contracts/__init__.py +6 -0
- rag_wright/contracts/chunk.py +79 -0
- rag_wright/contracts/compliance.py +303 -0
- rag_wright/contracts/contract_meta.py +27 -0
- rag_wright/contracts/extraction.py +130 -0
- rag_wright/contracts/function.py +167 -0
- rag_wright/contracts/function_routing.py +91 -0
- rag_wright/contracts/highlight.py +74 -0
- rag_wright/contracts/identifiers.py +153 -0
- rag_wright/contracts/jurisdiction.py +96 -0
- rag_wright/contracts/ontology.py +142 -0
- rag_wright/contracts/property.py +201 -0
- rag_wright/contracts/provenance.py +78 -0
- rag_wright/contracts/query_intent.py +53 -0
- rag_wright/contracts/span.py +76 -0
- rag_wright/contracts/value_match.py +84 -0
- rag_wright/corpus/__init__.py +0 -0
- rag_wright/corpus/canonicalize.py +116 -0
- rag_wright/corpus/cuad.py +153 -0
- rag_wright/corpus/cuad_ingestion.py +72 -0
- rag_wright/corpus/document_parser.py +299 -0
- rag_wright/corpus/edgar.py +231 -0
- rag_wright/corpus/gcs_ingestion.py +120 -0
- rag_wright/corpus/http.py +110 -0
- rag_wright/corpus/selection.py +152 -0
- rag_wright/mcp/__init__.py +11 -0
- rag_wright/mcp/compliance_server.py +299 -0
- rag_wright/mcp/intra_document_qa_server.py +170 -0
- rag_wright/mcp/relational_qa_server.py +171 -0
- rag_wright/mcp/session_store.py +64 -0
- rag_wright/mcp/typed_property_retrieval_server.py +191 -0
- rag_wright/models/__init__.py +8 -0
- rag_wright/models/profiles.py +331 -0
- rag_wright/models/seam.py +497 -0
- rag_wright/models/tag_structured.py +285 -0
- rag_wright/models/tracing.py +179 -0
- rag_wright/models/usage.py +102 -0
- rag_wright/okf/__init__.py +11 -0
- rag_wright/okf/compile.py +292 -0
- rag_wright/okf/document.py +47 -0
- rag_wright/okf/enrich.py +176 -0
- rag_wright/okf/links.py +190 -0
- rag_wright/okf/lint.py +105 -0
- rag_wright/ontology/__init__.py +6 -0
- rag_wright/ontology/_generated_template_meta.py +60 -0
- rag_wright/ontology/_generated_vocab.py +52 -0
- rag_wright/ontology/clause_template.py +964 -0
- rag_wright/ontology/codegen.py +84 -0
- rag_wright/ontology/compliance_bridge.ttl +186 -0
- rag_wright/ontology/contract_bridge.ttl +2685 -0
- rag_wright/ontology/contract_taxonomy.py +24 -0
- rag_wright/ontology/derive.py +58 -0
- rag_wright/ontology/loader.py +435 -0
- rag_wright/ontology/packs/ftc_16cfr255.ttl +29 -0
- rag_wright/ontology/registry.py +87 -0
- rag_wright/ontology/template_introspect.py +100 -0
- rag_wright/py.typed +0 -0
- rag_wright/reference/__init__.py +2 -0
- rag_wright/reference/compliance.py +41 -0
- rag_wright/reference/contract_seam.py +123 -0
- rag_wright/skills/__init__.py +7 -0
- rag_wright/skills/claim_extraction/SKILL.md +47 -0
- rag_wright/skills/claim_extraction/__init__.py +1 -0
- rag_wright/skills/claim_extraction/template.py +50 -0
- rag_wright/skills/compliance_judgment/SKILL.md +59 -0
- rag_wright/skills/corpus_ingest/SKILL.md +106 -0
- rag_wright/skills/extraction_semantic_judge/SKILL.md +51 -0
- rag_wright/skills/extraction_semantic_judge/__init__.py +1 -0
- rag_wright/skills/generation/SKILL.md +64 -0
- rag_wright/skills/generation/__init__.py +1 -0
- rag_wright/skills/generic_compliance_judgment/SKILL.md +58 -0
- rag_wright/skills/okf_navigate/SKILL.md +137 -0
- rag_wright/skills/requirement_extraction/SKILL.md +47 -0
- rag_wright/skills/requirement_extraction/__init__.py +1 -0
- rag_wright/skills/requirement_extraction/template.py +50 -0
- rag_wright/skills/rlm/SKILL.md +186 -0
- rag_wright/skills/rlm/__init__.py +31 -0
- rag_wright/skills/rlm/agent.py +292 -0
- rag_wright/skills/span_relevance_judgment/SKILL.md +67 -0
- rag_wright/skills/vision_to_text/SKILL.md +36 -0
- rag_wright/skills/vision_to_text/__init__.py +1 -0
- rag_wright/spans/__init__.py +1 -0
- rag_wright/spans/boundary.py +78 -0
- rag_wright/spans/clause_function_classifier.py +490 -0
- rag_wright/spans/clause_kg_extractor.py +337 -0
- rag_wright/spans/cuad_labels.py +81 -0
- rag_wright/spans/dim_classifier.py +158 -0
- rag_wright/spans/dim_fleet.json +411 -0
- rag_wright/spans/function_classifier.py +77 -0
- rag_wright/spans/function_families.py +62 -0
- rag_wright/spans/hybrid_classifier.py +103 -0
- rag_wright/spans/legalbert_classifier.py +83 -0
- rag_wright/spans/model_capabilities.py +107 -0
- rag_wright/spans/new_function_labels.py +111 -0
- rag_wright/spans/page_map.py +68 -0
- rag_wright/spans/property_extractor.py +365 -0
- rag_wright/spans/property_grounding.py +182 -0
- rag_wright/spans/reclassify.py +77 -0
- rag_wright/spans/scarce_function_labels.py +105 -0
- rag_wright/spans/segment.py +341 -0
- rag_wright/spans/semantic_judge.py +197 -0
- rag_wright/spans/symbolic_validation.py +131 -0
- rag_wright/spans/tag_clause_extractor.py +182 -0
- rag_wright/store/__init__.py +6 -0
- rag_wright/store/arcadedb.py +1135 -0
- rag_wright/store/chunk_text.py +66 -0
- rag_wright/store/seam.py +213 -0
- rag_wright/subgraphs/__init__.py +0 -0
- rag_wright/subgraphs/async_ingestion.py +204 -0
- rag_wright/subgraphs/compliance_check.py +1042 -0
- rag_wright/subgraphs/compliance_ingestion.py +306 -0
- rag_wright/subgraphs/contract_ingestion_pipeline.py +999 -0
- rag_wright/subgraphs/graph_extraction.py +102 -0
- rag_wright/subgraphs/intra_document_qa.py +328 -0
- rag_wright/subgraphs/observability.py +140 -0
- rag_wright/subgraphs/query_constraint_extraction.py +73 -0
- rag_wright/subgraphs/relational_qa.py +165 -0
- rag_wright/subgraphs/requirement_extraction.py +137 -0
- rag_wright/subgraphs/scaffold.py +65 -0
- rag_wright/subgraphs/semantic_chunking.py +183 -0
- rag_wright/subgraphs/typed_clause_extraction.py +172 -0
- rag_wright/subgraphs/typed_property_retrieval.py +278 -0
- rag_wright/util/__init__.py +1 -0
- rag_wright/util/concurrent.py +153 -0
- rag_wright/util/spacy_model.py +45 -0
- rag_wright-0.1.0.dist-info/METADATA +168 -0
- rag_wright-0.1.0.dist-info/RECORD +184 -0
- rag_wright-0.1.0.dist-info/WHEEL +4 -0
- rag_wright-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
"""Fusion (FR-Q.4, T27): union and deduplicate the two evidence streams on `chunk_id`, capped.
|
|
2
|
+
|
|
3
|
+
Combines the reranked retrieval top set (T22) and the graph-cited chunks (T26) into one deduplicated,
|
|
4
|
+
capped evidence set for synthesis/generation. This is deliberately NOT a score fusion: the graph
|
|
5
|
+
returns an answer with cited chunks, not a comparable ranked list, so there is no common score to fuse.
|
|
6
|
+
It is a deterministic union — retrieval chunks first (in their reranked order), then the graph-cited
|
|
7
|
+
chunks (in order of appearance), each `chunk_id` kept once with a record of which stream(s) surfaced it.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
from pydantic import BaseModel
|
|
13
|
+
|
|
14
|
+
from rag_wright.capabilities.graph_query import GraphAnswer
|
|
15
|
+
from rag_wright.capabilities.reranking import RerankResult
|
|
16
|
+
|
|
17
|
+
DEFAULT_UNION_CAP = 20 # the fused evidence set is capped before synthesis (§16.7)
|
|
18
|
+
|
|
19
|
+
RETRIEVAL = "retrieval"
|
|
20
|
+
GRAPH = "graph"
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class FusedChunk(BaseModel):
|
|
24
|
+
"""One chunk in the fused evidence set: its id and which stream(s) surfaced it (no score — this is
|
|
25
|
+
a union, not a score fusion)."""
|
|
26
|
+
|
|
27
|
+
chunk_id: str
|
|
28
|
+
sources: list[str] # subset of {"retrieval", "graph"}, sorted
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
class FusionResult(BaseModel):
|
|
32
|
+
"""The fused, deduplicated, capped evidence set for synthesis (T28) / generation (T29)."""
|
|
33
|
+
|
|
34
|
+
chunks: list[FusedChunk]
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def fuse(
|
|
38
|
+
reranked: RerankResult, graph: GraphAnswer, *, cap: int = DEFAULT_UNION_CAP
|
|
39
|
+
) -> FusionResult:
|
|
40
|
+
"""Union the reranked candidates and the graph-cited chunks on `chunk_id`, capped, deterministic.
|
|
41
|
+
|
|
42
|
+
Order is deterministic: reranked candidates first (in their order), then graph-cited chunks (first
|
|
43
|
+
appearance). A chunk surfaced by both streams appears once, tagged with both sources. The union is
|
|
44
|
+
then cut to `cap`.
|
|
45
|
+
"""
|
|
46
|
+
sources: dict[str, set[str]] = {}
|
|
47
|
+
order: list[str] = []
|
|
48
|
+
|
|
49
|
+
def _add(chunk_id: str, source: str) -> None:
|
|
50
|
+
if chunk_id not in sources:
|
|
51
|
+
sources[chunk_id] = set()
|
|
52
|
+
order.append(chunk_id)
|
|
53
|
+
sources[chunk_id].add(source)
|
|
54
|
+
|
|
55
|
+
for candidate in reranked.candidates:
|
|
56
|
+
_add(candidate.chunk_id, RETRIEVAL)
|
|
57
|
+
for evidence in graph.evidence:
|
|
58
|
+
for chunk_id in evidence.chunk_ids:
|
|
59
|
+
_add(chunk_id, GRAPH)
|
|
60
|
+
|
|
61
|
+
chunks = [FusedChunk(chunk_id=chunk_id, sources=sorted(sources[chunk_id])) for chunk_id in order[:cap]]
|
|
62
|
+
return FusionResult(chunks=chunks)
|
|
63
|
+
|
|
64
|
+
|
|
@@ -0,0 +1,243 @@
|
|
|
1
|
+
"""Graph extraction (FR-C.6, FR-I.4): the GP-1B docling-graph party/relational extractor over parsed chunks.
|
|
2
|
+
|
|
3
|
+
Re-backed per ADR-0035. The capability's job is unchanged -- a chunk -> ontology-conforming graph facts
|
|
4
|
+
(`ExtractionResult`) anchored to `chunk_id` with a confidence tag (FR-S.4), behind the T5 `Extractor` seam --
|
|
5
|
+
but the *implementation* is now the **GP-1B docling-graph extractor** (granite-4.2-8b), the entity/relational
|
|
6
|
+
extractor that populated Leg C at real recall 0.991. The earlier T23-27 hybrid stack (spaCy NER +
|
|
7
|
+
Pydantic-contract extraction + LLM escalation) is retired: Leg B (clause facts / inter-corpus recall) is served
|
|
8
|
+
by the typed Clause KG (typed_clause_extraction), and Leg C (party-to-party relational) by GP-1B, so the hybrid
|
|
9
|
+
stack -- built to probe inter-corpus recall -- no longer earns its keep.
|
|
10
|
+
|
|
11
|
+
`DoclingGraphExtractor` extracts the signing parties from the chunk text via docling-graph
|
|
12
|
+
(`dg_extraction.extract_parties`) and emits ORGANIZATION mentions + structural `CONTRACTS_WITH` facts between
|
|
13
|
+
them (`parties_to_extraction`), EXTRACTED. Resolution to canonical ids (EDGAR CIK) and the graph write are
|
|
14
|
+
downstream (`entity_resolution` -> `write_graph`), unchanged. The docling-graph call is a raw-SDK call, so the
|
|
15
|
+
LG-2c subgraph wraps it in `raw_llm_span` (the observability contract); `extract_fn` is dependency-injected so
|
|
16
|
+
the extractor is hermetically testable with no docling-graph / no network.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
from __future__ import annotations
|
|
20
|
+
|
|
21
|
+
import asyncio
|
|
22
|
+
import os
|
|
23
|
+
import re
|
|
24
|
+
from typing import Any, Callable
|
|
25
|
+
|
|
26
|
+
from pydantic import BaseModel
|
|
27
|
+
|
|
28
|
+
from rag_wright.capabilities.registry import CapabilityRegistry
|
|
29
|
+
from rag_wright.contracts.extraction import (
|
|
30
|
+
EntityMention,
|
|
31
|
+
ExtractionResult,
|
|
32
|
+
Extractor,
|
|
33
|
+
run_extractors,
|
|
34
|
+
)
|
|
35
|
+
from rag_wright.contracts.identifiers import ChunkId
|
|
36
|
+
from rag_wright.contracts.ontology import RelationshipFact
|
|
37
|
+
from rag_wright.contracts.provenance import ConfidenceTag, Provenance
|
|
38
|
+
# DD-5 (ADR-0066/0117): this is a CONTRACT-domain builder (reference pack). The engine contracts are
|
|
39
|
+
# taxonomy-free; the reference pack's entity/edge values live in ontology/contract_taxonomy.
|
|
40
|
+
from rag_wright.ontology.contract_taxonomy import AFFILIATE_OF, CONTRACTS_WITH, ORGANIZATION
|
|
41
|
+
|
|
42
|
+
DEFAULT_EXTRACT_CONCURRENCY = 4 # in-flight chunk extractions (backpressure); GPU/network-bound
|
|
43
|
+
# The adopted graph-extraction model (GP-1B): granite-4.2-8b via OpenRouter; config-driven (SPEC §17).
|
|
44
|
+
# Precedence parallels models.profiles.model_for: the role-specific env (RAG_GRAPH_EXTRACT_MODEL) > the
|
|
45
|
+
# all-roles knob (RAG_MODEL_ALL) > the built-in default. The caller's `graph_extract_model` arg wins over all
|
|
46
|
+
# (it is passed explicitly). So `RAG_MODEL_ALL=<id>` now covers party+affiliation extraction too.
|
|
47
|
+
DEFAULT_GRAPH_EXTRACT_MODEL = (
|
|
48
|
+
os.getenv("RAG_GRAPH_EXTRACT_MODEL") or os.getenv("RAG_MODEL_ALL") or "qwen3.8-27b-modal-or")
|
|
49
|
+
|
|
50
|
+
# extract_fn: contract/chunk text -> a `ContractParties` (docling-graph output) or None when nothing extracted.
|
|
51
|
+
PartyExtractFn = Callable[[str], Any]
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def parties_to_extraction(chunk_id: ChunkId, parties: list[str]) -> ExtractionResult:
|
|
55
|
+
"""Known signing parties -> ORGANIZATION mentions + a `CONTRACTS_WITH` fact between each pair, EXTRACTED.
|
|
56
|
+
|
|
57
|
+
Strips + dedups (order-stable); a lone party yields a mention but no edge. ADR-0012: the parties are the
|
|
58
|
+
actual signatories, so `CONTRACTS_WITH` is structural, not a proximity guess. This is the fact shape both
|
|
59
|
+
the GP-1B extractor (below) and the no-LLM CUAD-`Parties` path share.
|
|
60
|
+
"""
|
|
61
|
+
provenance = Provenance.of(chunk_id)
|
|
62
|
+
names = list(dict.fromkeys(p.strip() for p in parties if p.strip()))
|
|
63
|
+
mentions = [
|
|
64
|
+
EntityMention(text=name, entity_type=ORGANIZATION, confidence=ConfidenceTag.EXTRACTED)
|
|
65
|
+
for name in names
|
|
66
|
+
]
|
|
67
|
+
relationships = [
|
|
68
|
+
RelationshipFact(
|
|
69
|
+
provenance=provenance, confidence=ConfidenceTag.EXTRACTED,
|
|
70
|
+
source_ref=names[i], relationship_type=CONTRACTS_WITH, target_ref=names[j],
|
|
71
|
+
)
|
|
72
|
+
for i in range(len(names))
|
|
73
|
+
for j in range(i + 1, len(names))
|
|
74
|
+
]
|
|
75
|
+
return ExtractionResult(chunk_id=chunk_id, entity_mentions=mentions, relationship_facts=relationships)
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
# --- issue 0027: corporate AFFILIATION extraction (AFFILIATE_OF) ---------------------------------
|
|
79
|
+
# The reference pack declares the `AFFILIATE_OF` edge and both `write_graph` and `graph_query` handle it, but
|
|
80
|
+
# nothing ever PRODUCED the edge, so corporate affiliation was unanswerable. Affiliation is stated in the text
|
|
81
|
+
# ("Acme Holdings Ltd, an affiliate of Acme Corp"), so -- unlike the structural `CONTRACTS_WITH` -- it needs a
|
|
82
|
+
# text-reading extraction. This runs once per contract on the preamble (like party extraction), gated by a lexical
|
|
83
|
+
# pre-filter so contracts that state no affiliation cost no LLM call. Entities are NOT merged (an affiliate is a
|
|
84
|
+
# separate legal entity); only the edge between the two org nodes is added.
|
|
85
|
+
|
|
86
|
+
_AFFIL_PREAMBLE_CHARS = 8000 # affiliations, like parties, are named in the preamble; bound the LLM input
|
|
87
|
+
|
|
88
|
+
# Lexical pre-filter: no affiliation cue in the text -> no LLM call. A miss is a false-negative (missed
|
|
89
|
+
# affiliation); a spurious hit just costs a call that returns nothing. Prompt-engineering overlay (ADR-0066: the
|
|
90
|
+
# relationship TYPE is ontology-declared; these cue words are mechanism, not the closed vocabulary).
|
|
91
|
+
_AFFILIATION_CUE_RE = re.compile(
|
|
92
|
+
r"\b(affiliate|affiliated|subsidiar|parent\s+compan|wholly[\s-]?owned|under\s+common\s+control|"
|
|
93
|
+
r"a\s+division\s+of|owned\s+by)\b", re.IGNORECASE)
|
|
94
|
+
|
|
95
|
+
_AFFILIATION_PROMPT = (
|
|
96
|
+
"From this contract text, extract statements of CORPORATE AFFILIATION -- where one organization is stated to "
|
|
97
|
+
"be an affiliate, subsidiary, parent, division of, or under common control with ANOTHER organization. For each, "
|
|
98
|
+
"give the organization and the organization it is affiliated with. Do NOT treat two organizations merely "
|
|
99
|
+
"signing the same contract as an affiliation. If none is stated, return no items.\n\nTEXT:\n{text}")
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
class Affiliation(BaseModel):
|
|
103
|
+
"""One corporate-affiliation statement (issue 0027): `organization` is stated to be an affiliate/subsidiary/
|
|
104
|
+
parent of / under common control with `affiliate_of`. Both are ORGANIZATION surface forms."""
|
|
105
|
+
|
|
106
|
+
organization: str = ""
|
|
107
|
+
affiliate_of: str = ""
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
class Affiliations(BaseModel):
|
|
111
|
+
"""The tag-parse output of the affiliation extraction: zero or more `Affiliation` pairs."""
|
|
112
|
+
|
|
113
|
+
affiliations: list[Affiliation] = []
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def affiliations_to_extraction(chunk_id: ChunkId, affiliations: list) -> ExtractionResult:
|
|
117
|
+
"""Corporate-affiliation pairs -> ORGANIZATION mentions for BOTH orgs (so both endpoints resolve to nodes) +
|
|
118
|
+
an `AFFILIATE_OF` fact between them, EXTRACTED (issue 0027). Entities are NOT merged; only the edge is added.
|
|
119
|
+
Each `affiliations` item is a 2-tuple/list `(organization, affiliate_of)`. Strips/dedups; a pair with an empty
|
|
120
|
+
or self-referential side is dropped."""
|
|
121
|
+
provenance = Provenance.of(chunk_id)
|
|
122
|
+
mentions: list[EntityMention] = []
|
|
123
|
+
facts: list[RelationshipFact] = []
|
|
124
|
+
seen: set[str] = set()
|
|
125
|
+
for pair in affiliations:
|
|
126
|
+
org = (pair[0] or "").strip()
|
|
127
|
+
affil_of = (pair[1] or "").strip()
|
|
128
|
+
if not org or not affil_of or org.lower() == affil_of.lower():
|
|
129
|
+
continue
|
|
130
|
+
for name in (org, affil_of):
|
|
131
|
+
if name.lower() not in seen:
|
|
132
|
+
seen.add(name.lower())
|
|
133
|
+
mentions.append(EntityMention(
|
|
134
|
+
text=name, entity_type=ORGANIZATION, confidence=ConfidenceTag.EXTRACTED))
|
|
135
|
+
facts.append(RelationshipFact(
|
|
136
|
+
provenance=provenance, confidence=ConfidenceTag.EXTRACTED,
|
|
137
|
+
source_ref=org, relationship_type=AFFILIATE_OF, target_ref=affil_of))
|
|
138
|
+
return ExtractionResult(chunk_id=chunk_id, entity_mentions=mentions, relationship_facts=facts)
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
async def aextract_affiliations(text: str, *, model_id: str = DEFAULT_GRAPH_EXTRACT_MODEL) -> list[tuple[str, str]]:
|
|
142
|
+
"""Extract corporate-affiliation pairs from `text` (issue 0027). LEXICAL PRE-FILTER first: no affiliation cue
|
|
143
|
+
word -> no LLM call (near-zero added ingestion cost, since most contracts state none). On a cue hit, a lean
|
|
144
|
+
client-side tag-parse extraction over the preamble (ADR-0045). Degrades to no affiliations on any parse
|
|
145
|
+
failure (never raised)."""
|
|
146
|
+
if not _AFFILIATION_CUE_RE.search(text or ""):
|
|
147
|
+
return []
|
|
148
|
+
from rag_wright.models.tag_structured import build_tag_structured
|
|
149
|
+
|
|
150
|
+
try:
|
|
151
|
+
out = await build_tag_structured(model_id, Affiliations, label="affiliations").ainvoke(
|
|
152
|
+
_AFFILIATION_PROMPT.format(text=text[:_AFFIL_PREAMBLE_CHARS]))
|
|
153
|
+
except Exception: # noqa: BLE001 - a persistent client-side parse failure -> no affiliations, not a crash
|
|
154
|
+
return []
|
|
155
|
+
if out is None:
|
|
156
|
+
return []
|
|
157
|
+
return [(a.organization, a.affiliate_of) for a in out.affiliations
|
|
158
|
+
if (a.organization or "").strip() and (a.affiliate_of or "").strip()]
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
class DoclingGraphExtractor:
|
|
162
|
+
"""GP-1B: docling-graph party extraction (granite-4.2-8b) -- the adopted graph extractor (ADR-0035).
|
|
163
|
+
|
|
164
|
+
Extracts the signing parties from the chunk text via the injected `extract_fn` (docling-graph
|
|
165
|
+
`extract_parties`), then emits ORGANIZATION mentions + structural `CONTRACTS_WITH` facts between them. A
|
|
166
|
+
chunk that yields no parties returns an empty `ExtractionResult` (never raised). `extract_fn` is injected so
|
|
167
|
+
the extractor is hermetic in tests (a stub returning a `ContractParties`-shaped object with `.parties[].name`).
|
|
168
|
+
"""
|
|
169
|
+
|
|
170
|
+
name = "docling_graph"
|
|
171
|
+
|
|
172
|
+
def __init__(self, extract_fn: PartyExtractFn) -> None:
|
|
173
|
+
self._extract_fn = extract_fn
|
|
174
|
+
|
|
175
|
+
def extract(self, chunk_id: ChunkId, text: str) -> ExtractionResult:
|
|
176
|
+
extracted = self._extract_fn(text)
|
|
177
|
+
names = [p.name for p in extracted.parties] if extracted is not None else []
|
|
178
|
+
return parties_to_extraction(chunk_id, names)
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
def production_extract_fn(*, model_id: str = DEFAULT_GRAPH_EXTRACT_MODEL) -> PartyExtractFn:
|
|
182
|
+
"""Bind the real docling-graph party extractor to `(text) -> ContractParties | None` (granite via OpenRouter).
|
|
183
|
+
Lazy import so the module stays import-light and hermetic (tests inject a stub instead)."""
|
|
184
|
+
from rag_wright.capabilities.dg_extraction import default_extraction_model, extract_parties
|
|
185
|
+
|
|
186
|
+
model = default_extraction_model("graph-extract", model_id) # ADR-0100: profile-routed (RAG_SERVING/pin)
|
|
187
|
+
return lambda text: extract_parties(text, model)
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
def aproduction_extract_fn(*, model_id: str = DEFAULT_GRAPH_EXTRACT_MODEL):
|
|
191
|
+
"""ASYNC-B2c (ADR-0057): the async twin of `production_extract_fn` -- party extraction on the async
|
|
192
|
+
docling-graph seam (`aextract_parties`, true wall-clock deadline). Returns an async `(text) -> ContractParties
|
|
193
|
+
| None`."""
|
|
194
|
+
from rag_wright.capabilities.dg_extraction import aextract_parties, default_extraction_model
|
|
195
|
+
|
|
196
|
+
model = default_extraction_model("graph-extract", model_id) # ADR-0100: profile-routed (RAG_SERVING/pin)
|
|
197
|
+
|
|
198
|
+
async def _afn(text: str):
|
|
199
|
+
return await aextract_parties(text, model)
|
|
200
|
+
|
|
201
|
+
return _afn
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
def default_extractors() -> list[Extractor]:
|
|
205
|
+
"""The default extractor: the GP-1B docling-graph party extractor (the live wiring; ADR-0035)."""
|
|
206
|
+
return [DoclingGraphExtractor(production_extract_fn())]
|
|
207
|
+
|
|
208
|
+
|
|
209
|
+
async def extract_chunks(
|
|
210
|
+
items: list[tuple[ChunkId, str]],
|
|
211
|
+
*,
|
|
212
|
+
extractors: list[Extractor],
|
|
213
|
+
max_concurrency: int = DEFAULT_EXTRACT_CONCURRENCY,
|
|
214
|
+
) -> list[ExtractionResult]:
|
|
215
|
+
"""Extract facts from chunks concurrently (FR-I.6): each chunk's extractor stack runs in a thread,
|
|
216
|
+
bounded by a semaphore so bulk mode saturates the GPU/provider without unbounded in-flight work."""
|
|
217
|
+
semaphore = asyncio.Semaphore(max_concurrency)
|
|
218
|
+
|
|
219
|
+
async def _one(chunk_id: ChunkId, text: str) -> ExtractionResult:
|
|
220
|
+
async with semaphore: # backpressure
|
|
221
|
+
return await asyncio.to_thread(run_extractors, extractors, chunk_id, text)
|
|
222
|
+
|
|
223
|
+
return list(await asyncio.gather(*(_one(chunk_id, text) for chunk_id, text in items)))
|
|
224
|
+
|
|
225
|
+
|
|
226
|
+
def extract_chunks_sync(
|
|
227
|
+
items: list[tuple[ChunkId, str]],
|
|
228
|
+
*,
|
|
229
|
+
extractors: list[Extractor],
|
|
230
|
+
max_concurrency: int = DEFAULT_EXTRACT_CONCURRENCY,
|
|
231
|
+
) -> list[ExtractionResult]:
|
|
232
|
+
"""Synchronous convenience for callers not already in an event loop."""
|
|
233
|
+
return asyncio.run(extract_chunks(items, extractors=extractors, max_concurrency=max_concurrency))
|
|
234
|
+
|
|
235
|
+
|
|
236
|
+
def register_graph_extraction(registry: CapabilityRegistry) -> None:
|
|
237
|
+
"""Register graph extraction under FR-C.6 (`graph_extraction`, a `subgraph`; GP-1B docling-graph, ADR-0035)."""
|
|
238
|
+
registry.register(
|
|
239
|
+
"graph_extraction",
|
|
240
|
+
contract=ExtractionResult,
|
|
241
|
+
kind="subgraph", # multi-step extractor workflow (CAP-REG-1)
|
|
242
|
+
display_name="Graph extraction (GP-1B docling-graph party/relational)",
|
|
243
|
+
)
|
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
"""Graph query (FR-C.5, FR-Q.3, T26): answer relational/multi-hop questions by graph traversal.
|
|
2
|
+
|
|
3
|
+
Traverses the knowledge graph (T25) from a start entity over relationship edges and returns the answer
|
|
4
|
+
shaped as EVIDENCE for fusion (T27) — cited `chunk_id`s, `entity_id`s, and confidence tags — treated as
|
|
5
|
+
evidence to verify, not truth (SPEC §8/§14), never a final ranked list. It **surfaces** confidence on
|
|
6
|
+
every path; it does **not** filter or down-weight edges by confidence (FR-C.5/FR-Q.3). Acting on
|
|
7
|
+
confidence — being confidence-aware, abstaining — is the answer generator's job (FR-Q.6 / T29).
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
from typing import Optional
|
|
13
|
+
|
|
14
|
+
from pydantic import BaseModel
|
|
15
|
+
|
|
16
|
+
from rag_wright.capabilities.document_scope import validate_documents
|
|
17
|
+
from rag_wright.store.seam import Store
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class GraphEvidence(BaseModel):
|
|
21
|
+
"""One reached entity as a cited evidence item (not a scored result)."""
|
|
22
|
+
|
|
23
|
+
entity_id: str # the reached entity (a candidate answer)
|
|
24
|
+
name: str
|
|
25
|
+
hops: int
|
|
26
|
+
path_entity_ids: list[str] # the entity_id chain from start to target (the traversal evidence)
|
|
27
|
+
chunk_ids: list[str] # the chunks the path's edges were extracted from (no claim without a citation)
|
|
28
|
+
confidences: list[str] # the confidence tag of each edge on the path (surfaced, not filtered)
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
class GraphAnswer(BaseModel):
|
|
32
|
+
"""The graph query capability's output: candidate answers as evidence for fusion (FR-C.5, FR-Q.3)."""
|
|
33
|
+
|
|
34
|
+
start_entity_id: str
|
|
35
|
+
relationship_type: str
|
|
36
|
+
evidence: list[GraphEvidence] # evidence for fusion (T27), NOT a final ranked list
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def graph_query(
|
|
40
|
+
start_entity_id: str,
|
|
41
|
+
*,
|
|
42
|
+
store: Store,
|
|
43
|
+
relationship_type: str,
|
|
44
|
+
max_hops: int = 1,
|
|
45
|
+
documents: Optional[list[str]] = None,
|
|
46
|
+
) -> GraphAnswer:
|
|
47
|
+
"""Traverse `relationship_type` (a generic edge-type string -- the CALLER names it; no domain default) from
|
|
48
|
+
`start_entity_id` up to `max_hops` and return cited evidence.
|
|
49
|
+
|
|
50
|
+
Every edge on every path is surfaced with its `chunk_id` and confidence tag; no edge is dropped or
|
|
51
|
+
re-weighted by confidence here (that is the generator's job, FR-Q.6). The evidence is unranked.
|
|
52
|
+
|
|
53
|
+
`documents` (issue 0031): scope the traversal to a workspace's source documents -- EVERY edge on a path
|
|
54
|
+
must belong to one of them (so a multi-hop path cannot route through an out-of-scope document). `None` =
|
|
55
|
+
the whole graph; an unknown id RAISES (`UnknownDocumentError`); `[]` = scope-to-nothing (no evidence).
|
|
56
|
+
"""
|
|
57
|
+
validate_documents(store, documents) # reject an unknown document BEFORE the traversal (issue 0031)
|
|
58
|
+
rows = store.graph_neighbors(
|
|
59
|
+
start_entity_id, relationship_type=relationship_type, max_hops=max_hops, documents=documents
|
|
60
|
+
)
|
|
61
|
+
evidence = [
|
|
62
|
+
GraphEvidence(
|
|
63
|
+
entity_id=row["target_id"], name=row["target_name"], hops=row["hops"],
|
|
64
|
+
path_entity_ids=row["path_entity_ids"], chunk_ids=row["path_chunk_ids"],
|
|
65
|
+
confidences=row["path_confidences"],
|
|
66
|
+
)
|
|
67
|
+
for row in rows
|
|
68
|
+
]
|
|
69
|
+
return GraphAnswer(
|
|
70
|
+
start_entity_id=start_entity_id, relationship_type=relationship_type, evidence=evidence
|
|
71
|
+
)
|
|
72
|
+
|
|
73
|
+
|
|
@@ -0,0 +1,111 @@
|
|
|
1
|
+
"""Graph storage (FR-I.4, FR-I.5, T25): write resolved entities + relationships into the store, gated.
|
|
2
|
+
|
|
3
|
+
A seam-bound ingestion pipeline step (no SPEC section-5 slug, so it registers nothing and authors no
|
|
4
|
+
ARD manifest — like chunk write, T20). It maps a document's T24 `ResolutionResult` to store-level
|
|
5
|
+
nodes and edges and writes them through the `Store` seam in ONE transaction (FR-S.1: a chunk and its
|
|
6
|
+
extracted entities land together), content-hash gated so an unchanged document does no graph work
|
|
7
|
+
(FR-I.5). The graph is the relationship layer only (SPEC §8): entity nodes carry id/name/type and edges
|
|
8
|
+
carry the relationship — no heavy structured data.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import json
|
|
14
|
+
from dataclasses import dataclass
|
|
15
|
+
from pathlib import Path
|
|
16
|
+
from typing import Literal, Optional
|
|
17
|
+
|
|
18
|
+
from rag_wright.capabilities.entity_resolution import ResolutionResult
|
|
19
|
+
from rag_wright.corpus.canonicalize import normalize_entity_name
|
|
20
|
+
from rag_wright.store.seam import GraphEdge, GraphNode, Store
|
|
21
|
+
|
|
22
|
+
GraphWriteStatus = Literal["written", "skipped"]
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
@dataclass(frozen=True)
|
|
26
|
+
class GraphWriteResult:
|
|
27
|
+
"""The outcome of writing one document's graph."""
|
|
28
|
+
|
|
29
|
+
source_doc_id: str
|
|
30
|
+
status: GraphWriteStatus
|
|
31
|
+
node_count: int
|
|
32
|
+
edge_count: int
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _node_key(entity_id: Optional[str], surface_key: str) -> str:
|
|
36
|
+
"""The vertex identity: the resolver's canonical id when linked, else an `UNLINKED:<key>` surrogate (so a
|
|
37
|
+
canonical-id lookup matches only linked entities, and an unlinked ref lands on the same node as its cluster)."""
|
|
38
|
+
return entity_id if entity_id else f"UNLINKED:{surface_key}"
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def to_graph(resolution: ResolutionResult) -> tuple[list[GraphNode], list[GraphEdge]]:
|
|
42
|
+
"""Map a `ResolutionResult` to store nodes + edges. A relationship endpoint that has no standalone
|
|
43
|
+
entity (a ref-only endpoint) gets a minimal node so every edge connects to a vertex."""
|
|
44
|
+
nodes: dict[str, GraphNode] = {}
|
|
45
|
+
for entity in resolution.entities:
|
|
46
|
+
key = _node_key(entity.entity_id, entity.key)
|
|
47
|
+
nodes[key] = GraphNode(
|
|
48
|
+
node_key=key, entity_id=entity.entity_id or "", name=entity.representative,
|
|
49
|
+
entity_type=entity.entity_type, confidence=entity.confidence.value,
|
|
50
|
+
chunk_id=entity.chunk_ids[0] if entity.chunk_ids else "",
|
|
51
|
+
)
|
|
52
|
+
|
|
53
|
+
edges: list[GraphEdge] = []
|
|
54
|
+
for rel in resolution.relationships:
|
|
55
|
+
source_key = _node_key(rel.source_id, normalize_entity_name(rel.source_ref))
|
|
56
|
+
target_key = _node_key(rel.target_id, normalize_entity_name(rel.target_ref))
|
|
57
|
+
for key, resolved_id, ref in (
|
|
58
|
+
(source_key, rel.source_id, rel.source_ref),
|
|
59
|
+
(target_key, rel.target_id, rel.target_ref),
|
|
60
|
+
):
|
|
61
|
+
nodes.setdefault( # ref-only endpoint: minimal node (type UNKNOWN -- a bare ref carries no type; DD-5)
|
|
62
|
+
key,
|
|
63
|
+
GraphNode(
|
|
64
|
+
node_key=key, entity_id=resolved_id or "", name=ref,
|
|
65
|
+
entity_type="", confidence=rel.confidence.value,
|
|
66
|
+
chunk_id=rel.chunk_id,
|
|
67
|
+
),
|
|
68
|
+
)
|
|
69
|
+
edges.append(
|
|
70
|
+
GraphEdge(
|
|
71
|
+
source_key=source_key, target_key=target_key,
|
|
72
|
+
relationship_type=rel.relationship_type, confidence=rel.confidence.value,
|
|
73
|
+
chunk_id=rel.chunk_id,
|
|
74
|
+
)
|
|
75
|
+
)
|
|
76
|
+
return list(nodes.values()), edges
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
class GraphWriter:
|
|
80
|
+
"""Writes a document's resolved graph to the store, content-hash gated (FR-I.5). A checkpoint per
|
|
81
|
+
document keyed by content hash makes an unchanged document a no-op; the store holds the graph."""
|
|
82
|
+
|
|
83
|
+
def __init__(self, store: Store, *, checkpoint_dir: Path) -> None:
|
|
84
|
+
self._store = store
|
|
85
|
+
self._checkpoints = Path(checkpoint_dir) / "graph_checkpoints"
|
|
86
|
+
self._checkpoints.mkdir(parents=True, exist_ok=True)
|
|
87
|
+
|
|
88
|
+
def write_document(
|
|
89
|
+
self, source_doc_id: str, content_hash: str, resolution: ResolutionResult
|
|
90
|
+
) -> GraphWriteResult:
|
|
91
|
+
"""Write the document's nodes/edges (one transaction), unless an unchanged run already did."""
|
|
92
|
+
checkpoint = self._load(source_doc_id)
|
|
93
|
+
if checkpoint is not None and checkpoint["content_hash"] == content_hash:
|
|
94
|
+
return GraphWriteResult(source_doc_id, "skipped", 0, 0) # content-hash gate: no graph work
|
|
95
|
+
|
|
96
|
+
nodes, edges = to_graph(resolution)
|
|
97
|
+
self._store.write_graph(nodes, edges)
|
|
98
|
+
self._save(source_doc_id, content_hash)
|
|
99
|
+
return GraphWriteResult(source_doc_id, "written", len(nodes), len(edges))
|
|
100
|
+
|
|
101
|
+
def _path(self, source_doc_id: str) -> Path:
|
|
102
|
+
return self._checkpoints / f"{source_doc_id}.json"
|
|
103
|
+
|
|
104
|
+
def _load(self, source_doc_id: str) -> Optional[dict]:
|
|
105
|
+
path = self._path(source_doc_id)
|
|
106
|
+
return json.loads(path.read_text(encoding="utf-8")) if path.exists() else None
|
|
107
|
+
|
|
108
|
+
def _save(self, source_doc_id: str, content_hash: str) -> None:
|
|
109
|
+
self._path(source_doc_id).write_text(
|
|
110
|
+
json.dumps({"content_hash": content_hash, "status": "complete"}), encoding="utf-8"
|
|
111
|
+
)
|
|
@@ -0,0 +1,142 @@
|
|
|
1
|
+
"""CU-C2: serve + citation -- turn a `QueryIntent` about a KNOWN contract into a `HighlightResult`.
|
|
2
|
+
|
|
3
|
+
Given `(query, contract_id, QueryIntent)`, this is the back half of the CUAD front door:
|
|
4
|
+
|
|
5
|
+
in-taxonomy, intent=highlight -> typed within-contract filter -> the span SET (empty => "not present")
|
|
6
|
+
in-taxonomy, intent=extract -> the set, then field-extract `value_to_extract` from each matched span
|
|
7
|
+
in-taxonomy, intent=discriminate -> the set; the value_condition discriminator + (b) rerank is STUBBED
|
|
8
|
+
(passthrough hook -- the set is small: a handful of same-type spans)
|
|
9
|
+
out-of-taxonomy (in_taxonomy=False) -> semantic fallback over ALL the contract's spans + low-confidence flag
|
|
10
|
+
|
|
11
|
+
Every returned span carries its exact document location (`doc_start`/`doc_end` from `SpanRecord`, CU-B2) so an
|
|
12
|
+
app can highlight it. `structured_factory`/`embedder`/`store` are injected so the routing is tested
|
|
13
|
+
hermetically; the field-extraction LLM calls run concurrently (the standing eval/serve concurrency rule).
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import asyncio
|
|
19
|
+
import json
|
|
20
|
+
|
|
21
|
+
from pydantic import BaseModel
|
|
22
|
+
|
|
23
|
+
from rag_wright.contracts.highlight import HighlightResult, HighlightSpan
|
|
24
|
+
from rag_wright.contracts.query_intent import QueryIntent
|
|
25
|
+
from rag_wright.models.profiles import ModelRole, model_for
|
|
26
|
+
from rag_wright.models.tag_structured import build_tag_structured # ADR-0045: LLM-agnostic client-side output
|
|
27
|
+
|
|
28
|
+
DEFAULT_FALLBACK_K = 5
|
|
29
|
+
_EXTRACT_CONCURRENCY = 8
|
|
30
|
+
|
|
31
|
+
_EXTRACT_PROMPT = (
|
|
32
|
+
"From the following contract clause, extract {value}. Return ONLY the value as written in the clause; "
|
|
33
|
+
"if the clause does not state it, return null.\n\nClause:\n{text}"
|
|
34
|
+
)
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
class _Extracted(BaseModel):
|
|
38
|
+
value: str | None = None
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def _decode_bbox(raw) -> tuple[float, float, float, float] | None:
|
|
42
|
+
"""issue 0032: the store keeps bbox as a JSON `[l,t,r,b]` string (best-effort); decode to a tuple or None."""
|
|
43
|
+
if not raw:
|
|
44
|
+
return None
|
|
45
|
+
try:
|
|
46
|
+
vals = json.loads(raw) if isinstance(raw, str) else raw
|
|
47
|
+
return (float(vals[0]), float(vals[1]), float(vals[2]), float(vals[3])) if vals else None
|
|
48
|
+
except (ValueError, TypeError, IndexError):
|
|
49
|
+
return None
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def _to_span(row: dict, *, confidence: float, extracted_value: str | None = None) -> HighlightSpan:
|
|
53
|
+
pages = [int(p) for p in (row.get("pages") or [])] # issue 0032: source page(s) for the citation highlight
|
|
54
|
+
return HighlightSpan(
|
|
55
|
+
span_id=row["span_id"],
|
|
56
|
+
contract_id=row["contract_id"],
|
|
57
|
+
function=row.get("function", ""),
|
|
58
|
+
clause_ref=row.get("parent_chunk_id", ""), # heading/number not extracted yet -> the clause id
|
|
59
|
+
text=row["text"],
|
|
60
|
+
doc_start=row.get("doc_start"),
|
|
61
|
+
doc_end=row.get("doc_end"),
|
|
62
|
+
page=(pages[0] if pages else row.get("page")), # FIRST page (== pages[0] when known)
|
|
63
|
+
pages=pages,
|
|
64
|
+
bbox=_decode_bbox(row.get("bbox")),
|
|
65
|
+
extracted_value=extracted_value,
|
|
66
|
+
confidence=confidence,
|
|
67
|
+
)
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def _cosine(a: list[float], b: list[float]) -> float:
|
|
71
|
+
dot = sum(x * y for x, y in zip(a, b))
|
|
72
|
+
na = sum(x * x for x in a) ** 0.5
|
|
73
|
+
nb = sum(y * y for y in b) ** 0.5
|
|
74
|
+
return dot / (na * nb) if na and nb else 0.0
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def _field_extract(spans: list[HighlightSpan], value_to_extract: str, factory, model_id: str) -> list[HighlightSpan]:
|
|
78
|
+
"""One structured call per matched span (concurrent) to pinpoint `value_to_extract` within it."""
|
|
79
|
+
|
|
80
|
+
def _one(sp: HighlightSpan) -> HighlightSpan:
|
|
81
|
+
try:
|
|
82
|
+
out = factory(model_id, _Extracted).invoke(
|
|
83
|
+
_EXTRACT_PROMPT.format(value=value_to_extract, text=sp.text)
|
|
84
|
+
)
|
|
85
|
+
value = None if out is None else out.value
|
|
86
|
+
except Exception: # noqa: BLE001 - a persistent client-side parse failure -> no value (best-effort extract)
|
|
87
|
+
value = None
|
|
88
|
+
return sp.model_copy(update={"extracted_value": value})
|
|
89
|
+
|
|
90
|
+
async def _run() -> list[HighlightSpan]:
|
|
91
|
+
sem = asyncio.Semaphore(_EXTRACT_CONCURRENCY)
|
|
92
|
+
|
|
93
|
+
async def _guarded(sp: HighlightSpan) -> HighlightSpan:
|
|
94
|
+
async with sem:
|
|
95
|
+
return await asyncio.to_thread(_one, sp)
|
|
96
|
+
|
|
97
|
+
return await asyncio.gather(*[_guarded(sp) for sp in spans])
|
|
98
|
+
|
|
99
|
+
return asyncio.run(_run())
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def serve_highlight(
|
|
103
|
+
query: str,
|
|
104
|
+
contract_id: str,
|
|
105
|
+
intent: QueryIntent,
|
|
106
|
+
*,
|
|
107
|
+
store,
|
|
108
|
+
embedder=None,
|
|
109
|
+
structured_factory=build_tag_structured,
|
|
110
|
+
model_id: str | None = None,
|
|
111
|
+
fallback_k: int = DEFAULT_FALLBACK_K,
|
|
112
|
+
) -> HighlightResult:
|
|
113
|
+
"""Serve a `QueryIntent` against one contract into a cited `HighlightResult`. In-taxonomy routes through
|
|
114
|
+
the typed within-contract filter; out-of-taxonomy falls back to contract-scoped semantic ranking with a
|
|
115
|
+
low-confidence flag. See the module docstring for the four branches."""
|
|
116
|
+
model_id = model_id or model_for(ModelRole.STRUCTURED_REASONING)
|
|
117
|
+
low_confidence = False
|
|
118
|
+
|
|
119
|
+
if intent.in_taxonomy:
|
|
120
|
+
rows = store.spans_by_contract(contract_id, intent.clause_types)
|
|
121
|
+
spans = [_to_span(r, confidence=intent.confidence) for r in rows]
|
|
122
|
+
if intent.intent == "extract" and intent.value_to_extract and spans:
|
|
123
|
+
spans = _field_extract(spans, intent.value_to_extract, structured_factory, model_id)
|
|
124
|
+
# intent == "discriminate": the value_condition discriminator + (b) rerank is stubbed -> passthrough.
|
|
125
|
+
else:
|
|
126
|
+
rows = store.all_spans_by_contract(contract_id)
|
|
127
|
+
query_vec = embedder.encode_dense(query) if embedder is not None else None
|
|
128
|
+
if query_vec is not None:
|
|
129
|
+
rows = sorted(rows, key=lambda r: _cosine(query_vec, r["dense"]), reverse=True)
|
|
130
|
+
spans = [_to_span(r, confidence=intent.confidence) for r in rows[:fallback_k]]
|
|
131
|
+
low_confidence = True
|
|
132
|
+
|
|
133
|
+
return HighlightResult(
|
|
134
|
+
query=query,
|
|
135
|
+
contract_id=contract_id,
|
|
136
|
+
clause_types=intent.clause_types,
|
|
137
|
+
intent=intent.intent,
|
|
138
|
+
spans=spans,
|
|
139
|
+
present=bool(spans),
|
|
140
|
+
in_taxonomy=intent.in_taxonomy,
|
|
141
|
+
low_confidence=low_confidence,
|
|
142
|
+
)
|