rag-wright 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- rag_wright/__init__.py +13 -0
- rag_wright/api/__init__.py +33 -0
- rag_wright/api/config.py +59 -0
- rag_wright/api/discover.py +70 -0
- rag_wright/api/documents.py +39 -0
- rag_wright/api/ids.py +31 -0
- rag_wright/api/invoke.py +99 -0
- rag_wright/api/kg.py +61 -0
- rag_wright/api/mcp.py +94 -0
- rag_wright/api/usage.py +30 -0
- rag_wright/api/workspace.py +85 -0
- rag_wright/capabilities/__init__.py +8 -0
- rag_wright/capabilities/answer_generator.py +427 -0
- rag_wright/capabilities/ard.py +286 -0
- rag_wright/capabilities/assertion_extraction.py +79 -0
- rag_wright/capabilities/chunk_read.py +58 -0
- rag_wright/capabilities/chunk_write.py +163 -0
- rag_wright/capabilities/claim_extraction.py +153 -0
- rag_wright/capabilities/clause_exception_linking.py +117 -0
- rag_wright/capabilities/compliance_judgment.py +322 -0
- rag_wright/capabilities/compliance_store.py +87 -0
- rag_wright/capabilities/contract_kg_serve.py +156 -0
- rag_wright/capabilities/contract_kg_store.py +251 -0
- rag_wright/capabilities/dg_extraction.py +585 -0
- rag_wright/capabilities/disambiguation.py +163 -0
- rag_wright/capabilities/document_parse.py +87 -0
- rag_wright/capabilities/document_scope.py +49 -0
- rag_wright/capabilities/embedding.py +164 -0
- rag_wright/capabilities/embedding_profiles.py +43 -0
- rag_wright/capabilities/entity_resolution.py +154 -0
- rag_wright/capabilities/fusion.py +64 -0
- rag_wright/capabilities/graph_extraction.py +243 -0
- rag_wright/capabilities/graph_query.py +73 -0
- rag_wright/capabilities/graph_storage.py +111 -0
- rag_wright/capabilities/highlight_serve.py +142 -0
- rag_wright/capabilities/hybrid_search.py +65 -0
- rag_wright/capabilities/invoke.py +31 -0
- rag_wright/capabilities/jev_decision.py +38 -0
- rag_wright/capabilities/manifests.py +872 -0
- rag_wright/capabilities/okf_navigate.py +456 -0
- rag_wright/capabilities/parsing.py +286 -0
- rag_wright/capabilities/property_boosted_retrieval.py +125 -0
- rag_wright/capabilities/query_function_classifier.py +94 -0
- rag_wright/capabilities/query_understanding.py +109 -0
- rag_wright/capabilities/registry.py +262 -0
- rag_wright/capabilities/remote_encoders.py +94 -0
- rag_wright/capabilities/requirement_extraction.py +247 -0
- rag_wright/capabilities/reranking.py +123 -0
- rag_wright/capabilities/retrieval_core.py +126 -0
- rag_wright/capabilities/rlm_chunking.py +808 -0
- rag_wright/capabilities/rlm_synthesis.py +316 -0
- rag_wright/capabilities/scan_quality.py +136 -0
- rag_wright/capabilities/span_relevance_judgment.py +191 -0
- rag_wright/capabilities/vision_to_text.py +85 -0
- rag_wright/capabilities/vlm_ocr.py +85 -0
- rag_wright/contracts/__init__.py +6 -0
- rag_wright/contracts/chunk.py +79 -0
- rag_wright/contracts/compliance.py +303 -0
- rag_wright/contracts/contract_meta.py +27 -0
- rag_wright/contracts/extraction.py +130 -0
- rag_wright/contracts/function.py +167 -0
- rag_wright/contracts/function_routing.py +91 -0
- rag_wright/contracts/highlight.py +74 -0
- rag_wright/contracts/identifiers.py +153 -0
- rag_wright/contracts/jurisdiction.py +96 -0
- rag_wright/contracts/ontology.py +142 -0
- rag_wright/contracts/property.py +201 -0
- rag_wright/contracts/provenance.py +78 -0
- rag_wright/contracts/query_intent.py +53 -0
- rag_wright/contracts/span.py +76 -0
- rag_wright/contracts/value_match.py +84 -0
- rag_wright/corpus/__init__.py +0 -0
- rag_wright/corpus/canonicalize.py +116 -0
- rag_wright/corpus/cuad.py +153 -0
- rag_wright/corpus/cuad_ingestion.py +72 -0
- rag_wright/corpus/document_parser.py +299 -0
- rag_wright/corpus/edgar.py +231 -0
- rag_wright/corpus/gcs_ingestion.py +120 -0
- rag_wright/corpus/http.py +110 -0
- rag_wright/corpus/selection.py +152 -0
- rag_wright/mcp/__init__.py +11 -0
- rag_wright/mcp/compliance_server.py +299 -0
- rag_wright/mcp/intra_document_qa_server.py +170 -0
- rag_wright/mcp/relational_qa_server.py +171 -0
- rag_wright/mcp/session_store.py +64 -0
- rag_wright/mcp/typed_property_retrieval_server.py +191 -0
- rag_wright/models/__init__.py +8 -0
- rag_wright/models/profiles.py +331 -0
- rag_wright/models/seam.py +497 -0
- rag_wright/models/tag_structured.py +285 -0
- rag_wright/models/tracing.py +179 -0
- rag_wright/models/usage.py +102 -0
- rag_wright/okf/__init__.py +11 -0
- rag_wright/okf/compile.py +292 -0
- rag_wright/okf/document.py +47 -0
- rag_wright/okf/enrich.py +176 -0
- rag_wright/okf/links.py +190 -0
- rag_wright/okf/lint.py +105 -0
- rag_wright/ontology/__init__.py +6 -0
- rag_wright/ontology/_generated_template_meta.py +60 -0
- rag_wright/ontology/_generated_vocab.py +52 -0
- rag_wright/ontology/clause_template.py +964 -0
- rag_wright/ontology/codegen.py +84 -0
- rag_wright/ontology/compliance_bridge.ttl +186 -0
- rag_wright/ontology/contract_bridge.ttl +2685 -0
- rag_wright/ontology/contract_taxonomy.py +24 -0
- rag_wright/ontology/derive.py +58 -0
- rag_wright/ontology/loader.py +435 -0
- rag_wright/ontology/packs/ftc_16cfr255.ttl +29 -0
- rag_wright/ontology/registry.py +87 -0
- rag_wright/ontology/template_introspect.py +100 -0
- rag_wright/py.typed +0 -0
- rag_wright/reference/__init__.py +2 -0
- rag_wright/reference/compliance.py +41 -0
- rag_wright/reference/contract_seam.py +123 -0
- rag_wright/skills/__init__.py +7 -0
- rag_wright/skills/claim_extraction/SKILL.md +47 -0
- rag_wright/skills/claim_extraction/__init__.py +1 -0
- rag_wright/skills/claim_extraction/template.py +50 -0
- rag_wright/skills/compliance_judgment/SKILL.md +59 -0
- rag_wright/skills/corpus_ingest/SKILL.md +106 -0
- rag_wright/skills/extraction_semantic_judge/SKILL.md +51 -0
- rag_wright/skills/extraction_semantic_judge/__init__.py +1 -0
- rag_wright/skills/generation/SKILL.md +64 -0
- rag_wright/skills/generation/__init__.py +1 -0
- rag_wright/skills/generic_compliance_judgment/SKILL.md +58 -0
- rag_wright/skills/okf_navigate/SKILL.md +137 -0
- rag_wright/skills/requirement_extraction/SKILL.md +47 -0
- rag_wright/skills/requirement_extraction/__init__.py +1 -0
- rag_wright/skills/requirement_extraction/template.py +50 -0
- rag_wright/skills/rlm/SKILL.md +186 -0
- rag_wright/skills/rlm/__init__.py +31 -0
- rag_wright/skills/rlm/agent.py +292 -0
- rag_wright/skills/span_relevance_judgment/SKILL.md +67 -0
- rag_wright/skills/vision_to_text/SKILL.md +36 -0
- rag_wright/skills/vision_to_text/__init__.py +1 -0
- rag_wright/spans/__init__.py +1 -0
- rag_wright/spans/boundary.py +78 -0
- rag_wright/spans/clause_function_classifier.py +490 -0
- rag_wright/spans/clause_kg_extractor.py +337 -0
- rag_wright/spans/cuad_labels.py +81 -0
- rag_wright/spans/dim_classifier.py +158 -0
- rag_wright/spans/dim_fleet.json +411 -0
- rag_wright/spans/function_classifier.py +77 -0
- rag_wright/spans/function_families.py +62 -0
- rag_wright/spans/hybrid_classifier.py +103 -0
- rag_wright/spans/legalbert_classifier.py +83 -0
- rag_wright/spans/model_capabilities.py +107 -0
- rag_wright/spans/new_function_labels.py +111 -0
- rag_wright/spans/page_map.py +68 -0
- rag_wright/spans/property_extractor.py +365 -0
- rag_wright/spans/property_grounding.py +182 -0
- rag_wright/spans/reclassify.py +77 -0
- rag_wright/spans/scarce_function_labels.py +105 -0
- rag_wright/spans/segment.py +341 -0
- rag_wright/spans/semantic_judge.py +197 -0
- rag_wright/spans/symbolic_validation.py +131 -0
- rag_wright/spans/tag_clause_extractor.py +182 -0
- rag_wright/store/__init__.py +6 -0
- rag_wright/store/arcadedb.py +1135 -0
- rag_wright/store/chunk_text.py +66 -0
- rag_wright/store/seam.py +213 -0
- rag_wright/subgraphs/__init__.py +0 -0
- rag_wright/subgraphs/async_ingestion.py +204 -0
- rag_wright/subgraphs/compliance_check.py +1042 -0
- rag_wright/subgraphs/compliance_ingestion.py +306 -0
- rag_wright/subgraphs/contract_ingestion_pipeline.py +999 -0
- rag_wright/subgraphs/graph_extraction.py +102 -0
- rag_wright/subgraphs/intra_document_qa.py +328 -0
- rag_wright/subgraphs/observability.py +140 -0
- rag_wright/subgraphs/query_constraint_extraction.py +73 -0
- rag_wright/subgraphs/relational_qa.py +165 -0
- rag_wright/subgraphs/requirement_extraction.py +137 -0
- rag_wright/subgraphs/scaffold.py +65 -0
- rag_wright/subgraphs/semantic_chunking.py +183 -0
- rag_wright/subgraphs/typed_clause_extraction.py +172 -0
- rag_wright/subgraphs/typed_property_retrieval.py +278 -0
- rag_wright/util/__init__.py +1 -0
- rag_wright/util/concurrent.py +153 -0
- rag_wright/util/spacy_model.py +45 -0
- rag_wright-0.1.0.dist-info/METADATA +168 -0
- rag_wright-0.1.0.dist-info/RECORD +184 -0
- rag_wright-0.1.0.dist-info/WHEEL +4 -0
- rag_wright-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,286 @@
|
|
|
1
|
+
"""Parsing capability (FR-C.1): source documents -> a clean structured representation, parsed once.
|
|
2
|
+
|
|
3
|
+
Docling's `DocumentConverter` turns a PDF, Office file, or scan into a `DoclingDocument` (reading
|
|
4
|
+
order, headings, sections, tables, OCR). Parsing is content-hash gated so a document is parsed once
|
|
5
|
+
and reused by chunking, embedding, and extraction: the structured representation is cached as JSON
|
|
6
|
+
(`DoclingDocument.save_as_json` / `load_from_json`) keyed by the source's content hash, so re-parsing
|
|
7
|
+
unchanged content is a cache hit and changed content re-parses.
|
|
8
|
+
|
|
9
|
+
The `DocumentConverter` sits behind a small `Parser` seam so the capability's cache/gate logic is
|
|
10
|
+
tested hermetically with a stub, and the real (model-loading) parse is exercised opt-in (`-m parse`).
|
|
11
|
+
Grounded against `docling.document_converter.DocumentConverter.convert` and
|
|
12
|
+
`docling_core.types.doc.document.DoclingDocument` (framework graph).
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import hashlib
|
|
18
|
+
import logging
|
|
19
|
+
import re
|
|
20
|
+
from pathlib import Path
|
|
21
|
+
from typing import Protocol, runtime_checkable
|
|
22
|
+
|
|
23
|
+
from docling_core.types.doc.document import DoclingDocument
|
|
24
|
+
from pydantic import BaseModel, field_validator
|
|
25
|
+
|
|
26
|
+
from rag_wright.contracts.identifiers import canonical_source_doc_id
|
|
27
|
+
|
|
28
|
+
# Reused from the ChunkId scheme (T1): the delimiter-safe charset for a source_doc_id, so the id is
|
|
29
|
+
# citation/provenance-safe and consistent with `chunk_id`.
|
|
30
|
+
_SAFE = re.compile(r"[^A-Za-z0-9._-]+")
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
@runtime_checkable
|
|
34
|
+
class Parser(Protocol):
|
|
35
|
+
"""The document-conversion seam: turn a source path into a `DoclingDocument`."""
|
|
36
|
+
|
|
37
|
+
def convert(self, source: Path) -> DoclingDocument: ...
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
class ParsedDocument(BaseModel):
|
|
41
|
+
"""The parsing capability's contract: a handle to the cached structured representation.
|
|
42
|
+
|
|
43
|
+
The full `DoclingDocument` lives in the parse manifest at `manifest_path` (loaded via
|
|
44
|
+
`load_document`); this record carries the identity and the content hash the pipeline gates on.
|
|
45
|
+
"""
|
|
46
|
+
|
|
47
|
+
model_config = {"frozen": True}
|
|
48
|
+
|
|
49
|
+
source_doc_id: str
|
|
50
|
+
content_hash: str # sha256 hex of the source bytes
|
|
51
|
+
manifest_path: str
|
|
52
|
+
|
|
53
|
+
@field_validator("source_doc_id")
|
|
54
|
+
@classmethod
|
|
55
|
+
def _safe_source_doc_id(cls, v: str) -> str:
|
|
56
|
+
if not v or _SAFE.search(v):
|
|
57
|
+
raise ValueError("source_doc_id must be non-empty and use only [A-Za-z0-9._-]")
|
|
58
|
+
return v
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
class DoclingParser:
|
|
62
|
+
"""The real parser: Docling's `DocumentConverter`. Constructed lazily so importing the capability
|
|
63
|
+
(and the hermetic tests) does not load Docling's models."""
|
|
64
|
+
|
|
65
|
+
def __init__(self) -> None:
|
|
66
|
+
from docling.document_converter import DocumentConverter
|
|
67
|
+
|
|
68
|
+
self._converter = DocumentConverter()
|
|
69
|
+
|
|
70
|
+
def convert(self, source: Path) -> DoclingDocument:
|
|
71
|
+
return self._converter.convert(source).document
|
|
72
|
+
|
|
73
|
+
def parse_range(self, source: Path, page_range: tuple[int, int]) -> DoclingDocument:
|
|
74
|
+
"""PARSE-3: parse only pages `page_range` (1-based, inclusive) -- lets the tiered path fast-parse the
|
|
75
|
+
born-digital pages and VLM only the degraded ones, then concatenate, instead of VLM-ing the whole doc."""
|
|
76
|
+
return self._converter.convert(source, page_range=page_range).document
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
class TieredOCRReport(BaseModel):
|
|
80
|
+
"""0009-WIRE: which pages the tiered parser escalated to the VLM, and which remained unreadable even after
|
|
81
|
+
the VLM (genuine info loss -> the caller should flag PARTIAL / needs-rescan)."""
|
|
82
|
+
|
|
83
|
+
escalated_pages: list[int] = []
|
|
84
|
+
unreadable_pages: list[int] = []
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
class TieredOCRParser:
|
|
88
|
+
"""0009-WIRE: fast OCR -> scan-quality gate -> VLM escalation for degraded pages -> PARTIAL for what the VLM
|
|
89
|
+
still cannot read. A `Parser`, so it drops into `parse(..., parser=TieredOCRParser())` unchanged.
|
|
90
|
+
|
|
91
|
+
Benchmark (docs/eval/ocr_benchmark.md): fast OCR is perfect on readable scans and worthless on a heavily
|
|
92
|
+
degraded one (char_sim ~0.01); a VLM reads the degraded-but-readable scan (Gemma-4 0.991). So: run the cheap
|
|
93
|
+
fast parse, and ONLY when a page's OCR is untrustworthy re-parse via the VLM (whole-document escalation --
|
|
94
|
+
the VLM reads good pages fine too, so this is safe and keeps the common readable case at zero VLM cost).
|
|
95
|
+
`fast` and `vlm` are `Parser`s (injectable); the last run's `report` is exposed for PARTIAL reporting."""
|
|
96
|
+
|
|
97
|
+
def __init__(self, *, fast: Parser | None = None, vlm: Parser | None = None) -> None:
|
|
98
|
+
self._fast = fast
|
|
99
|
+
self._vlm = vlm
|
|
100
|
+
self.report = TieredOCRReport()
|
|
101
|
+
|
|
102
|
+
def convert(self, source: Path) -> DoclingDocument:
|
|
103
|
+
from rag_wright.capabilities.scan_quality import ScanQuality, assess_document
|
|
104
|
+
|
|
105
|
+
fast = self._fast or DoclingParser()
|
|
106
|
+
fast_doc = fast.convert(source)
|
|
107
|
+
# 0009-GATE-CAL: fold in IMAGE metrics (blur/faintness) -- the strong signal a text-only gate misses when
|
|
108
|
+
# the fast OCR is garbled-but-common-word. The VLM re-check below is text-only (the image stays blurry).
|
|
109
|
+
assessed = assess_document(fast_doc, page_images=_render_gray_pages(source))
|
|
110
|
+
degraded = sorted(pg for pg, a in assessed.items() if a.quality is not ScanQuality.READABLE)
|
|
111
|
+
# PARSE-1: a page with a usable NATIVE text layer (born-digital) is authoritative -- the OCR word-hit gate
|
|
112
|
+
# false-positives on legitimately sparse born-digital pages (a signature/joinder page: names, titles,
|
|
113
|
+
# page numbers), which triggered an unnecessary whole-document VLM escalation (~minutes on OpenRouter) on
|
|
114
|
+
# real contracts. So never OCR-escalate a page whose text layer we can read directly; only genuinely
|
|
115
|
+
# image-only pages (no text layer) stay in the escalation set.
|
|
116
|
+
born_digital = _text_layer_pages(source)
|
|
117
|
+
degraded = [pg for pg in degraded if pg not in born_digital]
|
|
118
|
+
if not degraded: # readable scan OR every "degraded" page was actually born-digital -> no VLM cost
|
|
119
|
+
self.report = TieredOCRReport()
|
|
120
|
+
return fast_doc
|
|
121
|
+
|
|
122
|
+
vlm = self._vlm if self._vlm is not None else (_default_vlm_parser() if _vlm_available() else None)
|
|
123
|
+
if vlm is None: # GRACEFUL DEGRADE: no VLM configured -> cannot escalate; flag PARTIAL, keep the fast doc
|
|
124
|
+
self.report = TieredOCRReport(escalated_pages=[], unreadable_pages=degraded)
|
|
125
|
+
_log_unreadable(source, degraded, "no VLM configured (set OPENROUTER_API_KEY)")
|
|
126
|
+
return fast_doc
|
|
127
|
+
try:
|
|
128
|
+
# PARSE-3: escalate ONLY the degraded pages to the VLM (per-page), then concatenate with the
|
|
129
|
+
# fast-parsed good pages -- never the whole document. A large born-digital doc with one genuine
|
|
130
|
+
# image-only page used to VLM all 60+ pages (~30 min) and blow the 600s parse deadline.
|
|
131
|
+
vlm_doc = _escalate_degraded_pages(source, fast, vlm, degraded)
|
|
132
|
+
except Exception as exc: # noqa: BLE001 - a VLM failure must not sink the parse; flag PARTIAL, keep fast doc
|
|
133
|
+
self.report = TieredOCRReport(escalated_pages=degraded, unreadable_pages=degraded)
|
|
134
|
+
_log_unreadable(source, degraded, f"VLM escalation failed: {exc!r}")
|
|
135
|
+
return fast_doc
|
|
136
|
+
|
|
137
|
+
vlm_assessed = assess_document(vlm_doc)
|
|
138
|
+
unreadable = sorted(pg for pg in degraded
|
|
139
|
+
if vlm_assessed.get(pg) is None or vlm_assessed[pg].quality is not ScanQuality.READABLE)
|
|
140
|
+
self.report = TieredOCRReport(escalated_pages=degraded, unreadable_pages=unreadable)
|
|
141
|
+
if unreadable: # even the VLM could not read these -> surface, never silently ingest gibberish
|
|
142
|
+
_log_unreadable(source, unreadable, "unreadable even after VLM escalation")
|
|
143
|
+
return vlm_doc
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
def _default_vlm_parser() -> Parser:
|
|
147
|
+
from rag_wright.capabilities.vlm_ocr import VlmOCRParser
|
|
148
|
+
|
|
149
|
+
return VlmOCRParser()
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def _page_count(source: Path) -> int:
|
|
153
|
+
"""Total page count of a PDF (pypdfium2, no parse). 0 for a non-PDF or any read error -> the caller falls
|
|
154
|
+
back to whole-document escalation."""
|
|
155
|
+
try:
|
|
156
|
+
import pypdfium2 as pdfium
|
|
157
|
+
|
|
158
|
+
return len(pdfium.PdfDocument(str(source)))
|
|
159
|
+
except Exception: # noqa: BLE001 - best-effort; unknown page count -> whole-doc fallback
|
|
160
|
+
return 0
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
def _escalation_runs(n_pages: int, degraded: set[int]) -> list[tuple[int, int, bool]]:
|
|
164
|
+
"""PARSE-3: partition pages 1..n_pages into CONTIGUOUS runs, each tagged `is_vlm` (a degraded page -> VLM,
|
|
165
|
+
else fast). So a 63-page doc with page 31 degraded yields [(1,30,False),(31,31,True),(32,63,False)] -- the VLM
|
|
166
|
+
touches only page 31. Returns `[(start, end, is_vlm)]`, 1-based inclusive, covering every page in order."""
|
|
167
|
+
runs: list[tuple[int, int, bool]] = []
|
|
168
|
+
start = 1
|
|
169
|
+
while start <= n_pages:
|
|
170
|
+
is_vlm = start in degraded
|
|
171
|
+
end = start
|
|
172
|
+
while end + 1 <= n_pages and ((end + 1) in degraded) == is_vlm:
|
|
173
|
+
end += 1
|
|
174
|
+
runs.append((start, end, is_vlm))
|
|
175
|
+
start = end + 1
|
|
176
|
+
return runs
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
def _escalate_degraded_pages(source: Path, fast: Parser, vlm: Parser, degraded: list[int]) -> DoclingDocument:
|
|
180
|
+
"""PARSE-3: build the escalated document by parsing each contiguous page-run with the right parser (fast for
|
|
181
|
+
born-digital pages, VLM for the degraded ones) and concatenating -- so VLM cost scales with the number of
|
|
182
|
+
DEGRADED pages, not the document length. Falls back to a whole-document VLM parse when the page count is
|
|
183
|
+
unknown or a parser has no `parse_range` (e.g. a non-PDF, or an injected stub) -- preserving the prior
|
|
184
|
+
behavior for those cases."""
|
|
185
|
+
n_pages = _page_count(source)
|
|
186
|
+
if not n_pages or not hasattr(fast, "parse_range") or not hasattr(vlm, "parse_range"):
|
|
187
|
+
return vlm.convert(source) # fallback: whole-document VLM (unknown page count / no page-range support)
|
|
188
|
+
runs = _escalation_runs(n_pages, set(degraded))
|
|
189
|
+
subdocs = [
|
|
190
|
+
(vlm if is_vlm else fast).parse_range(source, (start, end)) for start, end, is_vlm in runs
|
|
191
|
+
]
|
|
192
|
+
return subdocs[0] if len(subdocs) == 1 else DoclingDocument.concatenate(subdocs)
|
|
193
|
+
|
|
194
|
+
|
|
195
|
+
_MIN_TEXT_LAYER_CHARS = 30 # a PDF page with >= this many directly-extractable chars has a real, authoritative
|
|
196
|
+
# text layer (born-digital). Set LOW on purpose (PARSE-2, doc3): a true image-only scan
|
|
197
|
+
# page extracts ~0 chars, but a SPARSE born-digital page -- a schedule, an exhibit
|
|
198
|
+
# divider, a signature page (doc3 pages 52-58 = 91-179 chars) -- extracts only tens.
|
|
199
|
+
# The old 200 threshold mislabeled those sparse-but-real pages as scans, so a single one
|
|
200
|
+
# flagged by the OCR gate triggered a WHOLE-DOCUMENT VLM escalation that, on a 63-page
|
|
201
|
+
# doc, blew the 600s parse deadline. A real text layer of any size is authoritative.
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
def _text_layer_pages(source: Path, *, min_chars: int = _MIN_TEXT_LAYER_CHARS) -> set[int]:
|
|
205
|
+
"""PARSE-1: the 1-based page numbers of `source` that carry a usable NATIVE text layer (born-digital) -- read
|
|
206
|
+
DIRECTLY from the PDF (pypdfium2, no OCR). A page here is authoritative and must never be OCR-quality-assessed
|
|
207
|
+
or VLM-escalated. Best-effort: a non-PDF or any read error -> empty set (no override -> the tiered OCR path is
|
|
208
|
+
unchanged), so a scan / text / office source is never affected."""
|
|
209
|
+
try:
|
|
210
|
+
import pypdfium2 as pdfium
|
|
211
|
+
|
|
212
|
+
pdf = pdfium.PdfDocument(str(source))
|
|
213
|
+
pages: set[int] = set()
|
|
214
|
+
for i in range(len(pdf)):
|
|
215
|
+
text = pdf[i].get_textpage().get_text_range()
|
|
216
|
+
if len(text.strip()) >= min_chars:
|
|
217
|
+
pages.add(i + 1)
|
|
218
|
+
return pages
|
|
219
|
+
except Exception: # noqa: BLE001 - the text-layer probe is an optional authority signal; never fail the parse
|
|
220
|
+
return set()
|
|
221
|
+
|
|
222
|
+
|
|
223
|
+
def _render_gray_pages(source: Path, dpi: int = 200) -> dict:
|
|
224
|
+
"""Render each PDF page to a grayscale image {page_no(1-based): ndarray} for the scan-quality gate's image
|
|
225
|
+
metrics. 200 DPI to match the validated Laplacian/dark_frac thresholds. Best-effort: a non-PDF or any render
|
|
226
|
+
error -> {} (the gate falls back to text-only), so a text/office source never breaks the parse."""
|
|
227
|
+
try:
|
|
228
|
+
import numpy as np
|
|
229
|
+
import pypdfium2 as pdfium
|
|
230
|
+
|
|
231
|
+
pdf = pdfium.PdfDocument(str(source))
|
|
232
|
+
return {i + 1: np.asarray(pdf[i].render(scale=dpi / 72.0).to_pil().convert("L")) for i in range(len(pdf))}
|
|
233
|
+
except Exception: # noqa: BLE001 - image metrics are an optional gate signal; never fail the parse over them
|
|
234
|
+
return {}
|
|
235
|
+
|
|
236
|
+
|
|
237
|
+
def _vlm_available() -> bool:
|
|
238
|
+
"""The escalation VLM is usable only if an OpenRouter key is configured. Absent -> graceful degrade."""
|
|
239
|
+
import os
|
|
240
|
+
|
|
241
|
+
return bool(os.environ.get("OPENROUTER_API_KEY"))
|
|
242
|
+
|
|
243
|
+
|
|
244
|
+
def _log_unreadable(source: Path, pages: list[int], why: str) -> None:
|
|
245
|
+
"""Surface unreadable/degraded pages (never silently ingest gibberish -- 0006-C / ENG-1 applied to OCR)."""
|
|
246
|
+
logging.getLogger(__name__).warning(
|
|
247
|
+
"[ocr] %s: pages %s could not be read (%s) -- flagged PARTIAL / needs-rescan", getattr(source, "name", source),
|
|
248
|
+
pages, why)
|
|
249
|
+
|
|
250
|
+
|
|
251
|
+
def _source_doc_id(source: Path) -> str:
|
|
252
|
+
"""A delimiter-safe id from the file stem via the ONE canonical slug (HYG-1)."""
|
|
253
|
+
return canonical_source_doc_id(source.stem)
|
|
254
|
+
|
|
255
|
+
|
|
256
|
+
def _content_hash(source: Path) -> str:
|
|
257
|
+
return hashlib.sha256(source.read_bytes()).hexdigest()
|
|
258
|
+
|
|
259
|
+
|
|
260
|
+
def parse(source: Path, *, cache_dir: Path, parser: Parser) -> ParsedDocument:
|
|
261
|
+
"""Parse `source` into the cached structured representation, parsed once (content-hash gated).
|
|
262
|
+
|
|
263
|
+
If a manifest for this content hash already exists, it is reused (no re-parse); otherwise the
|
|
264
|
+
source is converted and the `DoclingDocument` is cached as JSON.
|
|
265
|
+
"""
|
|
266
|
+
source_doc_id = _source_doc_id(source)
|
|
267
|
+
content_hash = _content_hash(source)
|
|
268
|
+
cache_dir.mkdir(parents=True, exist_ok=True)
|
|
269
|
+
manifest_path = cache_dir / f"{source_doc_id}.{content_hash[:16]}.json"
|
|
270
|
+
|
|
271
|
+
if not manifest_path.exists(): # the content-hash gate: parse once
|
|
272
|
+
document = parser.convert(source)
|
|
273
|
+
document.save_as_json(manifest_path)
|
|
274
|
+
|
|
275
|
+
return ParsedDocument(
|
|
276
|
+
source_doc_id=source_doc_id,
|
|
277
|
+
content_hash=content_hash,
|
|
278
|
+
manifest_path=str(manifest_path),
|
|
279
|
+
)
|
|
280
|
+
|
|
281
|
+
|
|
282
|
+
def load_document(parsed: ParsedDocument) -> DoclingDocument:
|
|
283
|
+
"""Load the full structured representation from the parse manifest."""
|
|
284
|
+
return DoclingDocument.load_from_json(parsed.manifest_path)
|
|
285
|
+
|
|
286
|
+
|
|
@@ -0,0 +1,125 @@
|
|
|
1
|
+
"""SPAN-CLAUSE-RERANK (b) (FR-Q, ADR-0033): property-boosted typed retrieval over the CUAD-full KG.
|
|
2
|
+
|
|
3
|
+
The KG-5 V4 adopted shape, bound to CUAD-full via the operative-span `edge.span_id` join (SPAN-CLAUSE-RERANK):
|
|
4
|
+
BGE base pool (`store.span_hybrid_search`, function-filtered, bounded) -> join each span to its clause's
|
|
5
|
+
typed (dimension, value) props (`store.span_properties`, the edge.span_id join) -> `typed_constraint_match_rank`
|
|
6
|
+
(STABLE sort: constraint-match count primary, so the BGE pool order is the tiebreak) -> top-k cited spans.
|
|
7
|
+
|
|
8
|
+
The routed `functions` and typed `constraints` come from the query front-door (the MS1-6 A100 path: LegalBERT +
|
|
9
|
+
granite routing, granite constraint-extraction); they are passed in so this capability is a pure store+embedder
|
|
10
|
+
composition, hermetically testable with fakes. `store` / `embedder` are the seams (local or the A100 adapters).
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
from typing import Any, Iterable
|
|
16
|
+
|
|
17
|
+
from pydantic import BaseModel
|
|
18
|
+
|
|
19
|
+
from rag_wright.capabilities.retrieval_core import typed_constraint_match_rank
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
class RankedSpan(BaseModel):
|
|
23
|
+
"""One property-boosted, cited retrieval result (FR-Q.6): the span citation + text, its function, the
|
|
24
|
+
constraint-match score, the query constraints it satisfied, and its 1-based rank. A PURE retrieval contract --
|
|
25
|
+
relevance is a separate judgement (issue 0023): the `typed_property_retrieval` subgraph composes a RankedSpan
|
|
26
|
+
with a `RelevanceVerdict` into a `JudgedSpan`, rather than growing a verdict field here."""
|
|
27
|
+
|
|
28
|
+
span_id: str
|
|
29
|
+
text: str
|
|
30
|
+
function: str
|
|
31
|
+
match_score: float
|
|
32
|
+
matched: list[tuple[str, str]]
|
|
33
|
+
rank: int
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def _select_with_dense_floor(reranked_ids: list[str], dense_floor: list[str], k: int) -> list[str]:
|
|
37
|
+
"""Take the first `k` reranked ids (fusion/constraint order), but RESERVE slots so every dense-floor span
|
|
38
|
+
survives into the returned `k` (issue 0041): a non-floor span is skipped when the remaining slots are needed
|
|
39
|
+
for floor spans not yet included, so a strong dense match the RRF fusion buried is never dropped. A floor span
|
|
40
|
+
keeps its reranked position where a slot is free; constraint-matching spans (high in the reranked order) are
|
|
41
|
+
reached before slots run low, so the floor only displaces the weak, non-matching tail."""
|
|
42
|
+
floor = list(dict.fromkeys(dense_floor)) # dedup, keep dense order
|
|
43
|
+
result: list[str] = []
|
|
44
|
+
for sid in reranked_ids:
|
|
45
|
+
if len(result) >= k:
|
|
46
|
+
break
|
|
47
|
+
if sid in result:
|
|
48
|
+
continue
|
|
49
|
+
pending = [g for g in floor if g not in result and g != sid]
|
|
50
|
+
if sid in floor or (k - len(result)) > len(pending):
|
|
51
|
+
result.append(sid)
|
|
52
|
+
# else: skip this non-floor span, reserving the slot for a still-pending floor span
|
|
53
|
+
for g in floor: # safety net: any floor span the reranked pass didn't reach (should not happen)
|
|
54
|
+
if len(result) >= k:
|
|
55
|
+
break
|
|
56
|
+
if g not in result:
|
|
57
|
+
result.append(g)
|
|
58
|
+
return result[:k]
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def property_boosted_retrieval(
|
|
62
|
+
query: str,
|
|
63
|
+
*,
|
|
64
|
+
store: Any,
|
|
65
|
+
embedder: Any,
|
|
66
|
+
functions: Iterable[str],
|
|
67
|
+
constraints: Iterable[tuple[str, str]],
|
|
68
|
+
k: int = 8,
|
|
69
|
+
pool_k: int = 30,
|
|
70
|
+
documents: list[str] | None = None,
|
|
71
|
+
dense_floor_n: int = 3,
|
|
72
|
+
match_count_fn: Any,
|
|
73
|
+
) -> list[RankedSpan]:
|
|
74
|
+
"""Retrieve the top-`k` cited spans for `query`, property-boosted by the typed constraints. Pool =
|
|
75
|
+
`span_hybrid_search` (BGE RRF) over each routed function (deduped, BGE order preserved); rerank =
|
|
76
|
+
constraint-match primary + BGE tiebreak via the stable `typed_constraint_match_rank`.
|
|
77
|
+
|
|
78
|
+
Issue 0031: `documents` scopes the pool to a workspace's source documents IN THE STORE (`contract_id IN
|
|
79
|
+
[...]`), so out-of-scope spans are never pooled or reranked. `None` = whole index; `[]` = no results.
|
|
80
|
+
|
|
81
|
+
Issue 0041: DENSE FLOOR. RRF equal-weights the dense and sparse legs, so a short query on a ubiquitous token
|
|
82
|
+
('...terms?') lets the sparse leg crowd the strong dense match out of the pool entirely -> the answer span is
|
|
83
|
+
never returned and the product abstains. The top-`dense_floor_n` PURE-DENSE spans are unioned into the pool
|
|
84
|
+
and GUARANTEED into the returned `k` (`_select_with_dense_floor`): fusion still decides order, dense guarantees
|
|
85
|
+
membership. Kept small (default 3) so it recovers the buried dense match without displacing the working
|
|
86
|
+
queries' RRF/constraint results (it only fills non-matching tail slots). `dense_floor_n=0` disables it."""
|
|
87
|
+
constraints = set(constraints)
|
|
88
|
+
dense, sparse = embedder.encode_dense(query), embedder.encode_sparse(query)
|
|
89
|
+
ordered: list[str] = []
|
|
90
|
+
function_of: dict[str, str] = {}
|
|
91
|
+
seen: set[str] = set()
|
|
92
|
+
for f in (list(functions) or [None]): # None -> no function filter (whole-index pool)
|
|
93
|
+
for h in store.span_hybrid_search(dense, sparse, k=pool_k, function=f, documents=documents):
|
|
94
|
+
sid = h["span_id"]
|
|
95
|
+
if sid not in seen:
|
|
96
|
+
seen.add(sid)
|
|
97
|
+
ordered.append(sid)
|
|
98
|
+
function_of[sid] = h.get("function", "") or (f or "")
|
|
99
|
+
dense_floor: list[str] = [] # issue 0041: the top-N pure-dense spans, guaranteed into the returned k
|
|
100
|
+
if dense_floor_n:
|
|
101
|
+
for h in store.span_dense_search(dense, k=dense_floor_n, documents=documents):
|
|
102
|
+
sid = h["span_id"]
|
|
103
|
+
dense_floor.append(sid)
|
|
104
|
+
if sid not in seen: # pool it (for props + rerank) if the RRF leg missed it
|
|
105
|
+
seen.add(sid)
|
|
106
|
+
ordered.append(sid)
|
|
107
|
+
function_of[sid] = h.get("function", "")
|
|
108
|
+
if not ordered:
|
|
109
|
+
return []
|
|
110
|
+
from rag_wright.capabilities.contract_kg_store import ContractKGStore # EP-REF-1a-ii: typed reads via the domain store
|
|
111
|
+
props = ContractKGStore(store).span_properties(ordered)
|
|
112
|
+
ranked = typed_constraint_match_rank(
|
|
113
|
+
constraints, [(sid, props[sid]) for sid in ordered], match_count_fn=match_count_fn).ranked
|
|
114
|
+
top_ids = _select_with_dense_floor([r.clause_id for r in ranked], dense_floor, k)
|
|
115
|
+
texts = store.span_texts(top_ids)
|
|
116
|
+
score_of = {r.clause_id: r.match_score for r in ranked}
|
|
117
|
+
return [
|
|
118
|
+
RankedSpan(
|
|
119
|
+
span_id=sid, text=texts.get(sid, ""), function=function_of.get(sid, ""),
|
|
120
|
+
match_score=score_of.get(sid, 0.0), matched=sorted(constraints & props.get(sid, set())), rank=i,
|
|
121
|
+
)
|
|
122
|
+
for i, sid in enumerate(top_ids, 1)
|
|
123
|
+
]
|
|
124
|
+
|
|
125
|
+
|
|
@@ -0,0 +1,94 @@
|
|
|
1
|
+
"""KG-5e lever (b): the taxonomy-constrained query->function classifier -- a SINGLE structured granite call.
|
|
2
|
+
|
|
3
|
+
KG-5c/5e proved that query-side function routing is the dominant retrieval lever and that the LegalBERT
|
|
4
|
+
classifier (trained on CLAUSE spans) is out-of-distribution on short query text. This routes instead with an
|
|
5
|
+
LLM that reads the query in-distribution, its output FORCED to the closed FUNCTION taxonomy (structured output,
|
|
6
|
+
then normalized at the boundary via `canonical_function`, dropping anything off-taxonomy -- the "strict
|
|
7
|
+
contracts, normalize at the boundary" rule). Unlike granite's free-text `clause_type` (KG-5b, unreliable), the
|
|
8
|
+
label space here is closed and ranked.
|
|
9
|
+
|
|
10
|
+
`route_query` is the KG-5e (b) front door: TWO separate granite calls -- one for the typed constraints, one
|
|
11
|
+
for the function -- on the hypothesis that granite does one task at a time better than both in one prompt (the
|
|
12
|
+
fallback, if this underperforms, is to unify into a single prompt). The structured factory is injectable so
|
|
13
|
+
the normalization is tested hermetically (no LLM, no network).
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
from pydantic import BaseModel
|
|
19
|
+
|
|
20
|
+
from rag_wright.contracts.function import FUNCTION_LABELS, canonical_function
|
|
21
|
+
from rag_wright.models.tag_structured import build_tag_structured # ADR-0045: LLM-agnostic client-side output
|
|
22
|
+
|
|
23
|
+
_PROMPT = (
|
|
24
|
+
"You match a legal question about a contract to clause types from a FIXED taxonomy. Return the EXACT "
|
|
25
|
+
"taxonomy labels the question is about, MOST RELEVANT FIRST -- usually 1, at most 3, and only more than "
|
|
26
|
+
"one when the question genuinely spans multiple types. Use only labels from the taxonomy; if none apply, "
|
|
27
|
+
"return an empty list.\n\nTaxonomy:\n{taxonomy}\n\nQuestion: {query}"
|
|
28
|
+
)
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
class _FunctionChoice(BaseModel):
|
|
32
|
+
"""The LOOSE schema the LLM fills (ranked labels); normalized to canonical `FUNCTION_LABELS` at the boundary."""
|
|
33
|
+
|
|
34
|
+
clause_types: list[str] = []
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def _taxonomy_block() -> str:
|
|
38
|
+
return "\n".join(f"- {label}" for label in FUNCTION_LABELS)
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def classify_query_functions(
|
|
42
|
+
query: str, model_id: str, *, k: int = 3, structured_factory=build_tag_structured
|
|
43
|
+
) -> list[str]:
|
|
44
|
+
"""The top-`k` canonical FUNCTION_LABELS a query is about, ranked, via one structured call. Off-taxonomy or
|
|
45
|
+
unmappable labels are dropped; a failed structured emit (None or a persistent parse failure) -> ``[]``
|
|
46
|
+
(degrades to no routing, never a hard error). Order preserved, deduped, truncated to `k`."""
|
|
47
|
+
try:
|
|
48
|
+
raw = structured_factory(model_id, _FunctionChoice).invoke(
|
|
49
|
+
_PROMPT.format(taxonomy=_taxonomy_block(), query=query)
|
|
50
|
+
)
|
|
51
|
+
except Exception: # noqa: BLE001 - client-side tag parse gave up -> no routing (degrade, never a hard error)
|
|
52
|
+
return []
|
|
53
|
+
if raw is None:
|
|
54
|
+
return []
|
|
55
|
+
out: list[str] = []
|
|
56
|
+
for label in raw.clause_types:
|
|
57
|
+
mapped = canonical_function(label)
|
|
58
|
+
if mapped is not None and mapped not in out:
|
|
59
|
+
out.append(mapped)
|
|
60
|
+
return out[:k]
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def route_query(
|
|
64
|
+
query: str, *, extract_model, function_model_id: str, k: int = 3
|
|
65
|
+
) -> tuple[list[tuple[str, str]], list[str]]:
|
|
66
|
+
"""The KG-5e (b) query front door: TWO separate granite calls -> (typed constraints, ranked functions).
|
|
67
|
+
|
|
68
|
+
Call 1 extracts the typed (dimension, value) constraints (docling-graph + `clause_template`); call 2
|
|
69
|
+
classifies the function (this module). Two tasks, two calls -- if this underperforms a unified prompt, fold
|
|
70
|
+
them. Kept as a single wrapper so production and the eval share one shape.
|
|
71
|
+
"""
|
|
72
|
+
from rag_wright.capabilities.dg_extraction import extract_clause
|
|
73
|
+
from rag_wright.contracts.identifiers import ChunkId
|
|
74
|
+
from rag_wright.spans.clause_kg_extractor import clause_to_record
|
|
75
|
+
|
|
76
|
+
constraints: list[tuple[str, str]] = []
|
|
77
|
+
clause = extract_clause(query, extract_model)
|
|
78
|
+
if clause is not None:
|
|
79
|
+
rec = clause_to_record(clause, chunk_id=ChunkId.of("q", 0, query), function="Cap On Liability")
|
|
80
|
+
constraints = [(a.dimension.value, a.value) for a in rec.assertions]
|
|
81
|
+
functions = classify_query_functions(query, function_model_id, k=k)
|
|
82
|
+
return constraints, functions
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def register_query_function_classification(registry) -> None:
|
|
86
|
+
"""CAP-REG-2: register `query_function_classification` (agent_skill; taxonomy-constrained LLM classifier)."""
|
|
87
|
+
from rag_wright.contracts.function import FunctionClassification
|
|
88
|
+
|
|
89
|
+
registry.register(
|
|
90
|
+
"query_function_classification",
|
|
91
|
+
contract=FunctionClassification,
|
|
92
|
+
kind="agent_skill",
|
|
93
|
+
display_name="Query function classification",
|
|
94
|
+
)
|
|
@@ -0,0 +1,109 @@
|
|
|
1
|
+
"""CU-C1: NL->type query understanding -- the front door of the CUAD highlight pipeline.
|
|
2
|
+
|
|
3
|
+
A natural-language question about a KNOWN contract is parsed in ONE structured LLM call into a `QueryIntent`:
|
|
4
|
+
which clause type(s) the user asks about (mapped to the FUNCTION taxonomy), what they want done (highlight /
|
|
5
|
+
extract a value / discriminate among same-type clauses), and whether the ask is in-taxonomy.
|
|
6
|
+
|
|
7
|
+
This is a user-required MVP front door (robust natural-language handling, NOT templates), layered ON TOP of the
|
|
8
|
+
registered capabilities. It is intentionally NOT registered under a canonical capability slug: NL->type is not
|
|
9
|
+
an FR-C/FR-Q capability in the (closed) spec catalog, and registering one would invent a requirement the spec
|
|
10
|
+
does not state. It is ordinary tested software the compiled query graph can call as glue.
|
|
11
|
+
|
|
12
|
+
The strict `QueryIntent` validator rejects non-taxonomy labels, so the LLM emits a LOOSE schema and this module
|
|
13
|
+
NORMALIZES at the boundary (canonicalize labels, drop unmappable ones, derive `in_taxonomy`) -- the "strict
|
|
14
|
+
contracts, normalize at the boundary" rule. Multi-type is allowed. Out-of-taxonomy (nothing maps) ->
|
|
15
|
+
`in_taxonomy=False`, `clause_types=[]` -> the serve stage does semantic fallback + a low-confidence flag.
|
|
16
|
+
`structured_factory` is injectable so the mapping logic is tested hermetically (no LLM, no network).
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
from __future__ import annotations
|
|
20
|
+
|
|
21
|
+
from pydantic import BaseModel
|
|
22
|
+
|
|
23
|
+
from rag_wright.contracts.function import FUNCTION_LABELS, canonical_function
|
|
24
|
+
from rag_wright.contracts.query_intent import QueryIntent
|
|
25
|
+
from rag_wright.models.profiles import ModelRole, model_for
|
|
26
|
+
from rag_wright.models.seam import build_model
|
|
27
|
+
from rag_wright.models.tag_structured import build_tag_structured # ADR-0045: LLM-agnostic client-side output
|
|
28
|
+
|
|
29
|
+
_INTENTS = ("highlight", "extract", "discriminate")
|
|
30
|
+
|
|
31
|
+
# Two-step (reason -> emit), the sanctioned pattern for a model that cannot combine reasoning with a forced
|
|
32
|
+
# structured call (CLAUDE.md standing rule; ADR-0006 Qwen precedent; ADR-0032). Step 1 reasons in free text
|
|
33
|
+
# (a GENERAL-model strength -- no forced tool, so no thinking-mode tool rejection); step 2 emits the schema
|
|
34
|
+
# from that reasoning (the profile disables thinking on this forced call for the models that need it, e.g.
|
|
35
|
+
# Gemma). This makes NL->type work on the cheap GENERAL model, not just DeepSeek Pro (benchmarked in CU-D2).
|
|
36
|
+
_REASON_PROMPT = (
|
|
37
|
+
"You match a user's natural-language question about a SINGLE known contract to clause types from a fixed "
|
|
38
|
+
"taxonomy. Reason briefly about which type(s) the question concerns and what the user wants, then END "
|
|
39
|
+
"with EXACTLY these three lines:\n"
|
|
40
|
+
"TYPES: <comma-separated EXACT taxonomy labels the question is about, or NONE if the concept is absent "
|
|
41
|
+
"from the taxonomy>\n"
|
|
42
|
+
"INTENT: <highlight to locate the clause | extract if the user asks for a specific value inside it | "
|
|
43
|
+
"discriminate if the user wants the one clause matching a condition among several of the same type>\n"
|
|
44
|
+
"VALUE: <the value to extract or the selecting condition, or NONE>\n\n"
|
|
45
|
+
"Taxonomy:\n{taxonomy}\n\nQuestion: {query}"
|
|
46
|
+
)
|
|
47
|
+
_EMIT_PROMPT = (
|
|
48
|
+
"Convert this analysis into the structured intent. clause_types = the EXACT labels listed after TYPES "
|
|
49
|
+
"(empty list if TYPES is NONE). intent = the word after INTENT. value_to_extract = the VALUE when "
|
|
50
|
+
"intent is extract, else null; value_condition = the VALUE when intent is discriminate, else null. "
|
|
51
|
+
"in_taxonomy = false iff TYPES is NONE.\n\nAnalysis:\n{reasoning}"
|
|
52
|
+
)
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
class _RawIntent(BaseModel):
|
|
56
|
+
"""The LOOSE schema the LLM fills; normalized into the strict `QueryIntent` at the boundary."""
|
|
57
|
+
|
|
58
|
+
clause_types: list[str] = []
|
|
59
|
+
intent: str = "highlight"
|
|
60
|
+
value_to_extract: str | None = None
|
|
61
|
+
value_condition: str | None = None
|
|
62
|
+
in_taxonomy: bool = True
|
|
63
|
+
confidence: float = 1.0
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def _taxonomy_block() -> str:
|
|
67
|
+
return "\n".join(f"- {label}" for label in FUNCTION_LABELS)
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def understand_query(
|
|
71
|
+
query: str,
|
|
72
|
+
*,
|
|
73
|
+
reason_factory=build_model,
|
|
74
|
+
structured_factory=build_tag_structured,
|
|
75
|
+
model_id: str | None = None,
|
|
76
|
+
) -> QueryIntent:
|
|
77
|
+
"""Parse a natural-language question into a `QueryIntent` via the two-step reason->emit (see the prompts
|
|
78
|
+
above): step 1 reasons in free text, step 2 emits the schema from that reasoning. Then boundary
|
|
79
|
+
normalization: `in_taxonomy` is DERIVED from what actually maps (a mapping miss degrades gracefully to
|
|
80
|
+
semantic fallback, never a hard error); a failed emit (None) degrades to out-of-taxonomy low-confidence.
|
|
81
|
+
Defaults to the GENERAL model (Gemma): the two-step ties/beats DeepSeek Pro on NL->type at ~3x less
|
|
82
|
+
latency and cost, and is not throttled (benchmarked CU-D2 / ADR-0032). Factories are injectable for
|
|
83
|
+
hermetic tests."""
|
|
84
|
+
model_id = model_id or model_for(ModelRole.GENERAL)
|
|
85
|
+
reasoning = reason_factory(model_id).invoke(
|
|
86
|
+
_REASON_PROMPT.format(taxonomy=_taxonomy_block(), query=query)
|
|
87
|
+
).content
|
|
88
|
+
try:
|
|
89
|
+
raw = structured_factory(model_id, _RawIntent).invoke(_EMIT_PROMPT.format(reasoning=reasoning))
|
|
90
|
+
except Exception: # noqa: BLE001 - a persistent client-side parse failure degrades like a None emit
|
|
91
|
+
raw = None
|
|
92
|
+
if raw is None: # the emit failed -> out-of-taxonomy, low confidence (never a hard error)
|
|
93
|
+
return QueryIntent(clause_types=[], intent="highlight", in_taxonomy=False, confidence=0.0)
|
|
94
|
+
canon: list[str] = []
|
|
95
|
+
for label in raw.clause_types:
|
|
96
|
+
mapped = canonical_function(label)
|
|
97
|
+
if mapped is not None and mapped not in canon:
|
|
98
|
+
canon.append(mapped)
|
|
99
|
+
in_taxonomy = bool(canon) # ground truth = did anything map; the LLM's own flag is advisory only
|
|
100
|
+
intent = raw.intent if raw.intent in _INTENTS else "highlight"
|
|
101
|
+
confidence = min(1.0, max(0.0, raw.confidence))
|
|
102
|
+
return QueryIntent(
|
|
103
|
+
clause_types=canon,
|
|
104
|
+
intent=intent,
|
|
105
|
+
value_to_extract=raw.value_to_extract if intent == "extract" else None,
|
|
106
|
+
value_condition=raw.value_condition if intent == "discriminate" else None,
|
|
107
|
+
in_taxonomy=in_taxonomy,
|
|
108
|
+
confidence=confidence,
|
|
109
|
+
)
|