rag-wright 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (184) hide show
  1. rag_wright/__init__.py +13 -0
  2. rag_wright/api/__init__.py +33 -0
  3. rag_wright/api/config.py +59 -0
  4. rag_wright/api/discover.py +70 -0
  5. rag_wright/api/documents.py +39 -0
  6. rag_wright/api/ids.py +31 -0
  7. rag_wright/api/invoke.py +99 -0
  8. rag_wright/api/kg.py +61 -0
  9. rag_wright/api/mcp.py +94 -0
  10. rag_wright/api/usage.py +30 -0
  11. rag_wright/api/workspace.py +85 -0
  12. rag_wright/capabilities/__init__.py +8 -0
  13. rag_wright/capabilities/answer_generator.py +427 -0
  14. rag_wright/capabilities/ard.py +286 -0
  15. rag_wright/capabilities/assertion_extraction.py +79 -0
  16. rag_wright/capabilities/chunk_read.py +58 -0
  17. rag_wright/capabilities/chunk_write.py +163 -0
  18. rag_wright/capabilities/claim_extraction.py +153 -0
  19. rag_wright/capabilities/clause_exception_linking.py +117 -0
  20. rag_wright/capabilities/compliance_judgment.py +322 -0
  21. rag_wright/capabilities/compliance_store.py +87 -0
  22. rag_wright/capabilities/contract_kg_serve.py +156 -0
  23. rag_wright/capabilities/contract_kg_store.py +251 -0
  24. rag_wright/capabilities/dg_extraction.py +585 -0
  25. rag_wright/capabilities/disambiguation.py +163 -0
  26. rag_wright/capabilities/document_parse.py +87 -0
  27. rag_wright/capabilities/document_scope.py +49 -0
  28. rag_wright/capabilities/embedding.py +164 -0
  29. rag_wright/capabilities/embedding_profiles.py +43 -0
  30. rag_wright/capabilities/entity_resolution.py +154 -0
  31. rag_wright/capabilities/fusion.py +64 -0
  32. rag_wright/capabilities/graph_extraction.py +243 -0
  33. rag_wright/capabilities/graph_query.py +73 -0
  34. rag_wright/capabilities/graph_storage.py +111 -0
  35. rag_wright/capabilities/highlight_serve.py +142 -0
  36. rag_wright/capabilities/hybrid_search.py +65 -0
  37. rag_wright/capabilities/invoke.py +31 -0
  38. rag_wright/capabilities/jev_decision.py +38 -0
  39. rag_wright/capabilities/manifests.py +872 -0
  40. rag_wright/capabilities/okf_navigate.py +456 -0
  41. rag_wright/capabilities/parsing.py +286 -0
  42. rag_wright/capabilities/property_boosted_retrieval.py +125 -0
  43. rag_wright/capabilities/query_function_classifier.py +94 -0
  44. rag_wright/capabilities/query_understanding.py +109 -0
  45. rag_wright/capabilities/registry.py +262 -0
  46. rag_wright/capabilities/remote_encoders.py +94 -0
  47. rag_wright/capabilities/requirement_extraction.py +247 -0
  48. rag_wright/capabilities/reranking.py +123 -0
  49. rag_wright/capabilities/retrieval_core.py +126 -0
  50. rag_wright/capabilities/rlm_chunking.py +808 -0
  51. rag_wright/capabilities/rlm_synthesis.py +316 -0
  52. rag_wright/capabilities/scan_quality.py +136 -0
  53. rag_wright/capabilities/span_relevance_judgment.py +191 -0
  54. rag_wright/capabilities/vision_to_text.py +85 -0
  55. rag_wright/capabilities/vlm_ocr.py +85 -0
  56. rag_wright/contracts/__init__.py +6 -0
  57. rag_wright/contracts/chunk.py +79 -0
  58. rag_wright/contracts/compliance.py +303 -0
  59. rag_wright/contracts/contract_meta.py +27 -0
  60. rag_wright/contracts/extraction.py +130 -0
  61. rag_wright/contracts/function.py +167 -0
  62. rag_wright/contracts/function_routing.py +91 -0
  63. rag_wright/contracts/highlight.py +74 -0
  64. rag_wright/contracts/identifiers.py +153 -0
  65. rag_wright/contracts/jurisdiction.py +96 -0
  66. rag_wright/contracts/ontology.py +142 -0
  67. rag_wright/contracts/property.py +201 -0
  68. rag_wright/contracts/provenance.py +78 -0
  69. rag_wright/contracts/query_intent.py +53 -0
  70. rag_wright/contracts/span.py +76 -0
  71. rag_wright/contracts/value_match.py +84 -0
  72. rag_wright/corpus/__init__.py +0 -0
  73. rag_wright/corpus/canonicalize.py +116 -0
  74. rag_wright/corpus/cuad.py +153 -0
  75. rag_wright/corpus/cuad_ingestion.py +72 -0
  76. rag_wright/corpus/document_parser.py +299 -0
  77. rag_wright/corpus/edgar.py +231 -0
  78. rag_wright/corpus/gcs_ingestion.py +120 -0
  79. rag_wright/corpus/http.py +110 -0
  80. rag_wright/corpus/selection.py +152 -0
  81. rag_wright/mcp/__init__.py +11 -0
  82. rag_wright/mcp/compliance_server.py +299 -0
  83. rag_wright/mcp/intra_document_qa_server.py +170 -0
  84. rag_wright/mcp/relational_qa_server.py +171 -0
  85. rag_wright/mcp/session_store.py +64 -0
  86. rag_wright/mcp/typed_property_retrieval_server.py +191 -0
  87. rag_wright/models/__init__.py +8 -0
  88. rag_wright/models/profiles.py +331 -0
  89. rag_wright/models/seam.py +497 -0
  90. rag_wright/models/tag_structured.py +285 -0
  91. rag_wright/models/tracing.py +179 -0
  92. rag_wright/models/usage.py +102 -0
  93. rag_wright/okf/__init__.py +11 -0
  94. rag_wright/okf/compile.py +292 -0
  95. rag_wright/okf/document.py +47 -0
  96. rag_wright/okf/enrich.py +176 -0
  97. rag_wright/okf/links.py +190 -0
  98. rag_wright/okf/lint.py +105 -0
  99. rag_wright/ontology/__init__.py +6 -0
  100. rag_wright/ontology/_generated_template_meta.py +60 -0
  101. rag_wright/ontology/_generated_vocab.py +52 -0
  102. rag_wright/ontology/clause_template.py +964 -0
  103. rag_wright/ontology/codegen.py +84 -0
  104. rag_wright/ontology/compliance_bridge.ttl +186 -0
  105. rag_wright/ontology/contract_bridge.ttl +2685 -0
  106. rag_wright/ontology/contract_taxonomy.py +24 -0
  107. rag_wright/ontology/derive.py +58 -0
  108. rag_wright/ontology/loader.py +435 -0
  109. rag_wright/ontology/packs/ftc_16cfr255.ttl +29 -0
  110. rag_wright/ontology/registry.py +87 -0
  111. rag_wright/ontology/template_introspect.py +100 -0
  112. rag_wright/py.typed +0 -0
  113. rag_wright/reference/__init__.py +2 -0
  114. rag_wright/reference/compliance.py +41 -0
  115. rag_wright/reference/contract_seam.py +123 -0
  116. rag_wright/skills/__init__.py +7 -0
  117. rag_wright/skills/claim_extraction/SKILL.md +47 -0
  118. rag_wright/skills/claim_extraction/__init__.py +1 -0
  119. rag_wright/skills/claim_extraction/template.py +50 -0
  120. rag_wright/skills/compliance_judgment/SKILL.md +59 -0
  121. rag_wright/skills/corpus_ingest/SKILL.md +106 -0
  122. rag_wright/skills/extraction_semantic_judge/SKILL.md +51 -0
  123. rag_wright/skills/extraction_semantic_judge/__init__.py +1 -0
  124. rag_wright/skills/generation/SKILL.md +64 -0
  125. rag_wright/skills/generation/__init__.py +1 -0
  126. rag_wright/skills/generic_compliance_judgment/SKILL.md +58 -0
  127. rag_wright/skills/okf_navigate/SKILL.md +137 -0
  128. rag_wright/skills/requirement_extraction/SKILL.md +47 -0
  129. rag_wright/skills/requirement_extraction/__init__.py +1 -0
  130. rag_wright/skills/requirement_extraction/template.py +50 -0
  131. rag_wright/skills/rlm/SKILL.md +186 -0
  132. rag_wright/skills/rlm/__init__.py +31 -0
  133. rag_wright/skills/rlm/agent.py +292 -0
  134. rag_wright/skills/span_relevance_judgment/SKILL.md +67 -0
  135. rag_wright/skills/vision_to_text/SKILL.md +36 -0
  136. rag_wright/skills/vision_to_text/__init__.py +1 -0
  137. rag_wright/spans/__init__.py +1 -0
  138. rag_wright/spans/boundary.py +78 -0
  139. rag_wright/spans/clause_function_classifier.py +490 -0
  140. rag_wright/spans/clause_kg_extractor.py +337 -0
  141. rag_wright/spans/cuad_labels.py +81 -0
  142. rag_wright/spans/dim_classifier.py +158 -0
  143. rag_wright/spans/dim_fleet.json +411 -0
  144. rag_wright/spans/function_classifier.py +77 -0
  145. rag_wright/spans/function_families.py +62 -0
  146. rag_wright/spans/hybrid_classifier.py +103 -0
  147. rag_wright/spans/legalbert_classifier.py +83 -0
  148. rag_wright/spans/model_capabilities.py +107 -0
  149. rag_wright/spans/new_function_labels.py +111 -0
  150. rag_wright/spans/page_map.py +68 -0
  151. rag_wright/spans/property_extractor.py +365 -0
  152. rag_wright/spans/property_grounding.py +182 -0
  153. rag_wright/spans/reclassify.py +77 -0
  154. rag_wright/spans/scarce_function_labels.py +105 -0
  155. rag_wright/spans/segment.py +341 -0
  156. rag_wright/spans/semantic_judge.py +197 -0
  157. rag_wright/spans/symbolic_validation.py +131 -0
  158. rag_wright/spans/tag_clause_extractor.py +182 -0
  159. rag_wright/store/__init__.py +6 -0
  160. rag_wright/store/arcadedb.py +1135 -0
  161. rag_wright/store/chunk_text.py +66 -0
  162. rag_wright/store/seam.py +213 -0
  163. rag_wright/subgraphs/__init__.py +0 -0
  164. rag_wright/subgraphs/async_ingestion.py +204 -0
  165. rag_wright/subgraphs/compliance_check.py +1042 -0
  166. rag_wright/subgraphs/compliance_ingestion.py +306 -0
  167. rag_wright/subgraphs/contract_ingestion_pipeline.py +999 -0
  168. rag_wright/subgraphs/graph_extraction.py +102 -0
  169. rag_wright/subgraphs/intra_document_qa.py +328 -0
  170. rag_wright/subgraphs/observability.py +140 -0
  171. rag_wright/subgraphs/query_constraint_extraction.py +73 -0
  172. rag_wright/subgraphs/relational_qa.py +165 -0
  173. rag_wright/subgraphs/requirement_extraction.py +137 -0
  174. rag_wright/subgraphs/scaffold.py +65 -0
  175. rag_wright/subgraphs/semantic_chunking.py +183 -0
  176. rag_wright/subgraphs/typed_clause_extraction.py +172 -0
  177. rag_wright/subgraphs/typed_property_retrieval.py +278 -0
  178. rag_wright/util/__init__.py +1 -0
  179. rag_wright/util/concurrent.py +153 -0
  180. rag_wright/util/spacy_model.py +45 -0
  181. rag_wright-0.1.0.dist-info/METADATA +168 -0
  182. rag_wright-0.1.0.dist-info/RECORD +184 -0
  183. rag_wright-0.1.0.dist-info/WHEEL +4 -0
  184. rag_wright-0.1.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,286 @@
1
+ """Parsing capability (FR-C.1): source documents -> a clean structured representation, parsed once.
2
+
3
+ Docling's `DocumentConverter` turns a PDF, Office file, or scan into a `DoclingDocument` (reading
4
+ order, headings, sections, tables, OCR). Parsing is content-hash gated so a document is parsed once
5
+ and reused by chunking, embedding, and extraction: the structured representation is cached as JSON
6
+ (`DoclingDocument.save_as_json` / `load_from_json`) keyed by the source's content hash, so re-parsing
7
+ unchanged content is a cache hit and changed content re-parses.
8
+
9
+ The `DocumentConverter` sits behind a small `Parser` seam so the capability's cache/gate logic is
10
+ tested hermetically with a stub, and the real (model-loading) parse is exercised opt-in (`-m parse`).
11
+ Grounded against `docling.document_converter.DocumentConverter.convert` and
12
+ `docling_core.types.doc.document.DoclingDocument` (framework graph).
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ import hashlib
18
+ import logging
19
+ import re
20
+ from pathlib import Path
21
+ from typing import Protocol, runtime_checkable
22
+
23
+ from docling_core.types.doc.document import DoclingDocument
24
+ from pydantic import BaseModel, field_validator
25
+
26
+ from rag_wright.contracts.identifiers import canonical_source_doc_id
27
+
28
+ # Reused from the ChunkId scheme (T1): the delimiter-safe charset for a source_doc_id, so the id is
29
+ # citation/provenance-safe and consistent with `chunk_id`.
30
+ _SAFE = re.compile(r"[^A-Za-z0-9._-]+")
31
+
32
+
33
+ @runtime_checkable
34
+ class Parser(Protocol):
35
+ """The document-conversion seam: turn a source path into a `DoclingDocument`."""
36
+
37
+ def convert(self, source: Path) -> DoclingDocument: ...
38
+
39
+
40
+ class ParsedDocument(BaseModel):
41
+ """The parsing capability's contract: a handle to the cached structured representation.
42
+
43
+ The full `DoclingDocument` lives in the parse manifest at `manifest_path` (loaded via
44
+ `load_document`); this record carries the identity and the content hash the pipeline gates on.
45
+ """
46
+
47
+ model_config = {"frozen": True}
48
+
49
+ source_doc_id: str
50
+ content_hash: str # sha256 hex of the source bytes
51
+ manifest_path: str
52
+
53
+ @field_validator("source_doc_id")
54
+ @classmethod
55
+ def _safe_source_doc_id(cls, v: str) -> str:
56
+ if not v or _SAFE.search(v):
57
+ raise ValueError("source_doc_id must be non-empty and use only [A-Za-z0-9._-]")
58
+ return v
59
+
60
+
61
+ class DoclingParser:
62
+ """The real parser: Docling's `DocumentConverter`. Constructed lazily so importing the capability
63
+ (and the hermetic tests) does not load Docling's models."""
64
+
65
+ def __init__(self) -> None:
66
+ from docling.document_converter import DocumentConverter
67
+
68
+ self._converter = DocumentConverter()
69
+
70
+ def convert(self, source: Path) -> DoclingDocument:
71
+ return self._converter.convert(source).document
72
+
73
+ def parse_range(self, source: Path, page_range: tuple[int, int]) -> DoclingDocument:
74
+ """PARSE-3: parse only pages `page_range` (1-based, inclusive) -- lets the tiered path fast-parse the
75
+ born-digital pages and VLM only the degraded ones, then concatenate, instead of VLM-ing the whole doc."""
76
+ return self._converter.convert(source, page_range=page_range).document
77
+
78
+
79
+ class TieredOCRReport(BaseModel):
80
+ """0009-WIRE: which pages the tiered parser escalated to the VLM, and which remained unreadable even after
81
+ the VLM (genuine info loss -> the caller should flag PARTIAL / needs-rescan)."""
82
+
83
+ escalated_pages: list[int] = []
84
+ unreadable_pages: list[int] = []
85
+
86
+
87
+ class TieredOCRParser:
88
+ """0009-WIRE: fast OCR -> scan-quality gate -> VLM escalation for degraded pages -> PARTIAL for what the VLM
89
+ still cannot read. A `Parser`, so it drops into `parse(..., parser=TieredOCRParser())` unchanged.
90
+
91
+ Benchmark (docs/eval/ocr_benchmark.md): fast OCR is perfect on readable scans and worthless on a heavily
92
+ degraded one (char_sim ~0.01); a VLM reads the degraded-but-readable scan (Gemma-4 0.991). So: run the cheap
93
+ fast parse, and ONLY when a page's OCR is untrustworthy re-parse via the VLM (whole-document escalation --
94
+ the VLM reads good pages fine too, so this is safe and keeps the common readable case at zero VLM cost).
95
+ `fast` and `vlm` are `Parser`s (injectable); the last run's `report` is exposed for PARTIAL reporting."""
96
+
97
+ def __init__(self, *, fast: Parser | None = None, vlm: Parser | None = None) -> None:
98
+ self._fast = fast
99
+ self._vlm = vlm
100
+ self.report = TieredOCRReport()
101
+
102
+ def convert(self, source: Path) -> DoclingDocument:
103
+ from rag_wright.capabilities.scan_quality import ScanQuality, assess_document
104
+
105
+ fast = self._fast or DoclingParser()
106
+ fast_doc = fast.convert(source)
107
+ # 0009-GATE-CAL: fold in IMAGE metrics (blur/faintness) -- the strong signal a text-only gate misses when
108
+ # the fast OCR is garbled-but-common-word. The VLM re-check below is text-only (the image stays blurry).
109
+ assessed = assess_document(fast_doc, page_images=_render_gray_pages(source))
110
+ degraded = sorted(pg for pg, a in assessed.items() if a.quality is not ScanQuality.READABLE)
111
+ # PARSE-1: a page with a usable NATIVE text layer (born-digital) is authoritative -- the OCR word-hit gate
112
+ # false-positives on legitimately sparse born-digital pages (a signature/joinder page: names, titles,
113
+ # page numbers), which triggered an unnecessary whole-document VLM escalation (~minutes on OpenRouter) on
114
+ # real contracts. So never OCR-escalate a page whose text layer we can read directly; only genuinely
115
+ # image-only pages (no text layer) stay in the escalation set.
116
+ born_digital = _text_layer_pages(source)
117
+ degraded = [pg for pg in degraded if pg not in born_digital]
118
+ if not degraded: # readable scan OR every "degraded" page was actually born-digital -> no VLM cost
119
+ self.report = TieredOCRReport()
120
+ return fast_doc
121
+
122
+ vlm = self._vlm if self._vlm is not None else (_default_vlm_parser() if _vlm_available() else None)
123
+ if vlm is None: # GRACEFUL DEGRADE: no VLM configured -> cannot escalate; flag PARTIAL, keep the fast doc
124
+ self.report = TieredOCRReport(escalated_pages=[], unreadable_pages=degraded)
125
+ _log_unreadable(source, degraded, "no VLM configured (set OPENROUTER_API_KEY)")
126
+ return fast_doc
127
+ try:
128
+ # PARSE-3: escalate ONLY the degraded pages to the VLM (per-page), then concatenate with the
129
+ # fast-parsed good pages -- never the whole document. A large born-digital doc with one genuine
130
+ # image-only page used to VLM all 60+ pages (~30 min) and blow the 600s parse deadline.
131
+ vlm_doc = _escalate_degraded_pages(source, fast, vlm, degraded)
132
+ except Exception as exc: # noqa: BLE001 - a VLM failure must not sink the parse; flag PARTIAL, keep fast doc
133
+ self.report = TieredOCRReport(escalated_pages=degraded, unreadable_pages=degraded)
134
+ _log_unreadable(source, degraded, f"VLM escalation failed: {exc!r}")
135
+ return fast_doc
136
+
137
+ vlm_assessed = assess_document(vlm_doc)
138
+ unreadable = sorted(pg for pg in degraded
139
+ if vlm_assessed.get(pg) is None or vlm_assessed[pg].quality is not ScanQuality.READABLE)
140
+ self.report = TieredOCRReport(escalated_pages=degraded, unreadable_pages=unreadable)
141
+ if unreadable: # even the VLM could not read these -> surface, never silently ingest gibberish
142
+ _log_unreadable(source, unreadable, "unreadable even after VLM escalation")
143
+ return vlm_doc
144
+
145
+
146
+ def _default_vlm_parser() -> Parser:
147
+ from rag_wright.capabilities.vlm_ocr import VlmOCRParser
148
+
149
+ return VlmOCRParser()
150
+
151
+
152
+ def _page_count(source: Path) -> int:
153
+ """Total page count of a PDF (pypdfium2, no parse). 0 for a non-PDF or any read error -> the caller falls
154
+ back to whole-document escalation."""
155
+ try:
156
+ import pypdfium2 as pdfium
157
+
158
+ return len(pdfium.PdfDocument(str(source)))
159
+ except Exception: # noqa: BLE001 - best-effort; unknown page count -> whole-doc fallback
160
+ return 0
161
+
162
+
163
+ def _escalation_runs(n_pages: int, degraded: set[int]) -> list[tuple[int, int, bool]]:
164
+ """PARSE-3: partition pages 1..n_pages into CONTIGUOUS runs, each tagged `is_vlm` (a degraded page -> VLM,
165
+ else fast). So a 63-page doc with page 31 degraded yields [(1,30,False),(31,31,True),(32,63,False)] -- the VLM
166
+ touches only page 31. Returns `[(start, end, is_vlm)]`, 1-based inclusive, covering every page in order."""
167
+ runs: list[tuple[int, int, bool]] = []
168
+ start = 1
169
+ while start <= n_pages:
170
+ is_vlm = start in degraded
171
+ end = start
172
+ while end + 1 <= n_pages and ((end + 1) in degraded) == is_vlm:
173
+ end += 1
174
+ runs.append((start, end, is_vlm))
175
+ start = end + 1
176
+ return runs
177
+
178
+
179
+ def _escalate_degraded_pages(source: Path, fast: Parser, vlm: Parser, degraded: list[int]) -> DoclingDocument:
180
+ """PARSE-3: build the escalated document by parsing each contiguous page-run with the right parser (fast for
181
+ born-digital pages, VLM for the degraded ones) and concatenating -- so VLM cost scales with the number of
182
+ DEGRADED pages, not the document length. Falls back to a whole-document VLM parse when the page count is
183
+ unknown or a parser has no `parse_range` (e.g. a non-PDF, or an injected stub) -- preserving the prior
184
+ behavior for those cases."""
185
+ n_pages = _page_count(source)
186
+ if not n_pages or not hasattr(fast, "parse_range") or not hasattr(vlm, "parse_range"):
187
+ return vlm.convert(source) # fallback: whole-document VLM (unknown page count / no page-range support)
188
+ runs = _escalation_runs(n_pages, set(degraded))
189
+ subdocs = [
190
+ (vlm if is_vlm else fast).parse_range(source, (start, end)) for start, end, is_vlm in runs
191
+ ]
192
+ return subdocs[0] if len(subdocs) == 1 else DoclingDocument.concatenate(subdocs)
193
+
194
+
195
+ _MIN_TEXT_LAYER_CHARS = 30 # a PDF page with >= this many directly-extractable chars has a real, authoritative
196
+ # text layer (born-digital). Set LOW on purpose (PARSE-2, doc3): a true image-only scan
197
+ # page extracts ~0 chars, but a SPARSE born-digital page -- a schedule, an exhibit
198
+ # divider, a signature page (doc3 pages 52-58 = 91-179 chars) -- extracts only tens.
199
+ # The old 200 threshold mislabeled those sparse-but-real pages as scans, so a single one
200
+ # flagged by the OCR gate triggered a WHOLE-DOCUMENT VLM escalation that, on a 63-page
201
+ # doc, blew the 600s parse deadline. A real text layer of any size is authoritative.
202
+
203
+
204
+ def _text_layer_pages(source: Path, *, min_chars: int = _MIN_TEXT_LAYER_CHARS) -> set[int]:
205
+ """PARSE-1: the 1-based page numbers of `source` that carry a usable NATIVE text layer (born-digital) -- read
206
+ DIRECTLY from the PDF (pypdfium2, no OCR). A page here is authoritative and must never be OCR-quality-assessed
207
+ or VLM-escalated. Best-effort: a non-PDF or any read error -> empty set (no override -> the tiered OCR path is
208
+ unchanged), so a scan / text / office source is never affected."""
209
+ try:
210
+ import pypdfium2 as pdfium
211
+
212
+ pdf = pdfium.PdfDocument(str(source))
213
+ pages: set[int] = set()
214
+ for i in range(len(pdf)):
215
+ text = pdf[i].get_textpage().get_text_range()
216
+ if len(text.strip()) >= min_chars:
217
+ pages.add(i + 1)
218
+ return pages
219
+ except Exception: # noqa: BLE001 - the text-layer probe is an optional authority signal; never fail the parse
220
+ return set()
221
+
222
+
223
+ def _render_gray_pages(source: Path, dpi: int = 200) -> dict:
224
+ """Render each PDF page to a grayscale image {page_no(1-based): ndarray} for the scan-quality gate's image
225
+ metrics. 200 DPI to match the validated Laplacian/dark_frac thresholds. Best-effort: a non-PDF or any render
226
+ error -> {} (the gate falls back to text-only), so a text/office source never breaks the parse."""
227
+ try:
228
+ import numpy as np
229
+ import pypdfium2 as pdfium
230
+
231
+ pdf = pdfium.PdfDocument(str(source))
232
+ return {i + 1: np.asarray(pdf[i].render(scale=dpi / 72.0).to_pil().convert("L")) for i in range(len(pdf))}
233
+ except Exception: # noqa: BLE001 - image metrics are an optional gate signal; never fail the parse over them
234
+ return {}
235
+
236
+
237
+ def _vlm_available() -> bool:
238
+ """The escalation VLM is usable only if an OpenRouter key is configured. Absent -> graceful degrade."""
239
+ import os
240
+
241
+ return bool(os.environ.get("OPENROUTER_API_KEY"))
242
+
243
+
244
+ def _log_unreadable(source: Path, pages: list[int], why: str) -> None:
245
+ """Surface unreadable/degraded pages (never silently ingest gibberish -- 0006-C / ENG-1 applied to OCR)."""
246
+ logging.getLogger(__name__).warning(
247
+ "[ocr] %s: pages %s could not be read (%s) -- flagged PARTIAL / needs-rescan", getattr(source, "name", source),
248
+ pages, why)
249
+
250
+
251
+ def _source_doc_id(source: Path) -> str:
252
+ """A delimiter-safe id from the file stem via the ONE canonical slug (HYG-1)."""
253
+ return canonical_source_doc_id(source.stem)
254
+
255
+
256
+ def _content_hash(source: Path) -> str:
257
+ return hashlib.sha256(source.read_bytes()).hexdigest()
258
+
259
+
260
+ def parse(source: Path, *, cache_dir: Path, parser: Parser) -> ParsedDocument:
261
+ """Parse `source` into the cached structured representation, parsed once (content-hash gated).
262
+
263
+ If a manifest for this content hash already exists, it is reused (no re-parse); otherwise the
264
+ source is converted and the `DoclingDocument` is cached as JSON.
265
+ """
266
+ source_doc_id = _source_doc_id(source)
267
+ content_hash = _content_hash(source)
268
+ cache_dir.mkdir(parents=True, exist_ok=True)
269
+ manifest_path = cache_dir / f"{source_doc_id}.{content_hash[:16]}.json"
270
+
271
+ if not manifest_path.exists(): # the content-hash gate: parse once
272
+ document = parser.convert(source)
273
+ document.save_as_json(manifest_path)
274
+
275
+ return ParsedDocument(
276
+ source_doc_id=source_doc_id,
277
+ content_hash=content_hash,
278
+ manifest_path=str(manifest_path),
279
+ )
280
+
281
+
282
+ def load_document(parsed: ParsedDocument) -> DoclingDocument:
283
+ """Load the full structured representation from the parse manifest."""
284
+ return DoclingDocument.load_from_json(parsed.manifest_path)
285
+
286
+
@@ -0,0 +1,125 @@
1
+ """SPAN-CLAUSE-RERANK (b) (FR-Q, ADR-0033): property-boosted typed retrieval over the CUAD-full KG.
2
+
3
+ The KG-5 V4 adopted shape, bound to CUAD-full via the operative-span `edge.span_id` join (SPAN-CLAUSE-RERANK):
4
+ BGE base pool (`store.span_hybrid_search`, function-filtered, bounded) -> join each span to its clause's
5
+ typed (dimension, value) props (`store.span_properties`, the edge.span_id join) -> `typed_constraint_match_rank`
6
+ (STABLE sort: constraint-match count primary, so the BGE pool order is the tiebreak) -> top-k cited spans.
7
+
8
+ The routed `functions` and typed `constraints` come from the query front-door (the MS1-6 A100 path: LegalBERT +
9
+ granite routing, granite constraint-extraction); they are passed in so this capability is a pure store+embedder
10
+ composition, hermetically testable with fakes. `store` / `embedder` are the seams (local or the A100 adapters).
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ from typing import Any, Iterable
16
+
17
+ from pydantic import BaseModel
18
+
19
+ from rag_wright.capabilities.retrieval_core import typed_constraint_match_rank
20
+
21
+
22
+ class RankedSpan(BaseModel):
23
+ """One property-boosted, cited retrieval result (FR-Q.6): the span citation + text, its function, the
24
+ constraint-match score, the query constraints it satisfied, and its 1-based rank. A PURE retrieval contract --
25
+ relevance is a separate judgement (issue 0023): the `typed_property_retrieval` subgraph composes a RankedSpan
26
+ with a `RelevanceVerdict` into a `JudgedSpan`, rather than growing a verdict field here."""
27
+
28
+ span_id: str
29
+ text: str
30
+ function: str
31
+ match_score: float
32
+ matched: list[tuple[str, str]]
33
+ rank: int
34
+
35
+
36
+ def _select_with_dense_floor(reranked_ids: list[str], dense_floor: list[str], k: int) -> list[str]:
37
+ """Take the first `k` reranked ids (fusion/constraint order), but RESERVE slots so every dense-floor span
38
+ survives into the returned `k` (issue 0041): a non-floor span is skipped when the remaining slots are needed
39
+ for floor spans not yet included, so a strong dense match the RRF fusion buried is never dropped. A floor span
40
+ keeps its reranked position where a slot is free; constraint-matching spans (high in the reranked order) are
41
+ reached before slots run low, so the floor only displaces the weak, non-matching tail."""
42
+ floor = list(dict.fromkeys(dense_floor)) # dedup, keep dense order
43
+ result: list[str] = []
44
+ for sid in reranked_ids:
45
+ if len(result) >= k:
46
+ break
47
+ if sid in result:
48
+ continue
49
+ pending = [g for g in floor if g not in result and g != sid]
50
+ if sid in floor or (k - len(result)) > len(pending):
51
+ result.append(sid)
52
+ # else: skip this non-floor span, reserving the slot for a still-pending floor span
53
+ for g in floor: # safety net: any floor span the reranked pass didn't reach (should not happen)
54
+ if len(result) >= k:
55
+ break
56
+ if g not in result:
57
+ result.append(g)
58
+ return result[:k]
59
+
60
+
61
+ def property_boosted_retrieval(
62
+ query: str,
63
+ *,
64
+ store: Any,
65
+ embedder: Any,
66
+ functions: Iterable[str],
67
+ constraints: Iterable[tuple[str, str]],
68
+ k: int = 8,
69
+ pool_k: int = 30,
70
+ documents: list[str] | None = None,
71
+ dense_floor_n: int = 3,
72
+ match_count_fn: Any,
73
+ ) -> list[RankedSpan]:
74
+ """Retrieve the top-`k` cited spans for `query`, property-boosted by the typed constraints. Pool =
75
+ `span_hybrid_search` (BGE RRF) over each routed function (deduped, BGE order preserved); rerank =
76
+ constraint-match primary + BGE tiebreak via the stable `typed_constraint_match_rank`.
77
+
78
+ Issue 0031: `documents` scopes the pool to a workspace's source documents IN THE STORE (`contract_id IN
79
+ [...]`), so out-of-scope spans are never pooled or reranked. `None` = whole index; `[]` = no results.
80
+
81
+ Issue 0041: DENSE FLOOR. RRF equal-weights the dense and sparse legs, so a short query on a ubiquitous token
82
+ ('...terms?') lets the sparse leg crowd the strong dense match out of the pool entirely -> the answer span is
83
+ never returned and the product abstains. The top-`dense_floor_n` PURE-DENSE spans are unioned into the pool
84
+ and GUARANTEED into the returned `k` (`_select_with_dense_floor`): fusion still decides order, dense guarantees
85
+ membership. Kept small (default 3) so it recovers the buried dense match without displacing the working
86
+ queries' RRF/constraint results (it only fills non-matching tail slots). `dense_floor_n=0` disables it."""
87
+ constraints = set(constraints)
88
+ dense, sparse = embedder.encode_dense(query), embedder.encode_sparse(query)
89
+ ordered: list[str] = []
90
+ function_of: dict[str, str] = {}
91
+ seen: set[str] = set()
92
+ for f in (list(functions) or [None]): # None -> no function filter (whole-index pool)
93
+ for h in store.span_hybrid_search(dense, sparse, k=pool_k, function=f, documents=documents):
94
+ sid = h["span_id"]
95
+ if sid not in seen:
96
+ seen.add(sid)
97
+ ordered.append(sid)
98
+ function_of[sid] = h.get("function", "") or (f or "")
99
+ dense_floor: list[str] = [] # issue 0041: the top-N pure-dense spans, guaranteed into the returned k
100
+ if dense_floor_n:
101
+ for h in store.span_dense_search(dense, k=dense_floor_n, documents=documents):
102
+ sid = h["span_id"]
103
+ dense_floor.append(sid)
104
+ if sid not in seen: # pool it (for props + rerank) if the RRF leg missed it
105
+ seen.add(sid)
106
+ ordered.append(sid)
107
+ function_of[sid] = h.get("function", "")
108
+ if not ordered:
109
+ return []
110
+ from rag_wright.capabilities.contract_kg_store import ContractKGStore # EP-REF-1a-ii: typed reads via the domain store
111
+ props = ContractKGStore(store).span_properties(ordered)
112
+ ranked = typed_constraint_match_rank(
113
+ constraints, [(sid, props[sid]) for sid in ordered], match_count_fn=match_count_fn).ranked
114
+ top_ids = _select_with_dense_floor([r.clause_id for r in ranked], dense_floor, k)
115
+ texts = store.span_texts(top_ids)
116
+ score_of = {r.clause_id: r.match_score for r in ranked}
117
+ return [
118
+ RankedSpan(
119
+ span_id=sid, text=texts.get(sid, ""), function=function_of.get(sid, ""),
120
+ match_score=score_of.get(sid, 0.0), matched=sorted(constraints & props.get(sid, set())), rank=i,
121
+ )
122
+ for i, sid in enumerate(top_ids, 1)
123
+ ]
124
+
125
+
@@ -0,0 +1,94 @@
1
+ """KG-5e lever (b): the taxonomy-constrained query->function classifier -- a SINGLE structured granite call.
2
+
3
+ KG-5c/5e proved that query-side function routing is the dominant retrieval lever and that the LegalBERT
4
+ classifier (trained on CLAUSE spans) is out-of-distribution on short query text. This routes instead with an
5
+ LLM that reads the query in-distribution, its output FORCED to the closed FUNCTION taxonomy (structured output,
6
+ then normalized at the boundary via `canonical_function`, dropping anything off-taxonomy -- the "strict
7
+ contracts, normalize at the boundary" rule). Unlike granite's free-text `clause_type` (KG-5b, unreliable), the
8
+ label space here is closed and ranked.
9
+
10
+ `route_query` is the KG-5e (b) front door: TWO separate granite calls -- one for the typed constraints, one
11
+ for the function -- on the hypothesis that granite does one task at a time better than both in one prompt (the
12
+ fallback, if this underperforms, is to unify into a single prompt). The structured factory is injectable so
13
+ the normalization is tested hermetically (no LLM, no network).
14
+ """
15
+
16
+ from __future__ import annotations
17
+
18
+ from pydantic import BaseModel
19
+
20
+ from rag_wright.contracts.function import FUNCTION_LABELS, canonical_function
21
+ from rag_wright.models.tag_structured import build_tag_structured # ADR-0045: LLM-agnostic client-side output
22
+
23
+ _PROMPT = (
24
+ "You match a legal question about a contract to clause types from a FIXED taxonomy. Return the EXACT "
25
+ "taxonomy labels the question is about, MOST RELEVANT FIRST -- usually 1, at most 3, and only more than "
26
+ "one when the question genuinely spans multiple types. Use only labels from the taxonomy; if none apply, "
27
+ "return an empty list.\n\nTaxonomy:\n{taxonomy}\n\nQuestion: {query}"
28
+ )
29
+
30
+
31
+ class _FunctionChoice(BaseModel):
32
+ """The LOOSE schema the LLM fills (ranked labels); normalized to canonical `FUNCTION_LABELS` at the boundary."""
33
+
34
+ clause_types: list[str] = []
35
+
36
+
37
+ def _taxonomy_block() -> str:
38
+ return "\n".join(f"- {label}" for label in FUNCTION_LABELS)
39
+
40
+
41
+ def classify_query_functions(
42
+ query: str, model_id: str, *, k: int = 3, structured_factory=build_tag_structured
43
+ ) -> list[str]:
44
+ """The top-`k` canonical FUNCTION_LABELS a query is about, ranked, via one structured call. Off-taxonomy or
45
+ unmappable labels are dropped; a failed structured emit (None or a persistent parse failure) -> ``[]``
46
+ (degrades to no routing, never a hard error). Order preserved, deduped, truncated to `k`."""
47
+ try:
48
+ raw = structured_factory(model_id, _FunctionChoice).invoke(
49
+ _PROMPT.format(taxonomy=_taxonomy_block(), query=query)
50
+ )
51
+ except Exception: # noqa: BLE001 - client-side tag parse gave up -> no routing (degrade, never a hard error)
52
+ return []
53
+ if raw is None:
54
+ return []
55
+ out: list[str] = []
56
+ for label in raw.clause_types:
57
+ mapped = canonical_function(label)
58
+ if mapped is not None and mapped not in out:
59
+ out.append(mapped)
60
+ return out[:k]
61
+
62
+
63
+ def route_query(
64
+ query: str, *, extract_model, function_model_id: str, k: int = 3
65
+ ) -> tuple[list[tuple[str, str]], list[str]]:
66
+ """The KG-5e (b) query front door: TWO separate granite calls -> (typed constraints, ranked functions).
67
+
68
+ Call 1 extracts the typed (dimension, value) constraints (docling-graph + `clause_template`); call 2
69
+ classifies the function (this module). Two tasks, two calls -- if this underperforms a unified prompt, fold
70
+ them. Kept as a single wrapper so production and the eval share one shape.
71
+ """
72
+ from rag_wright.capabilities.dg_extraction import extract_clause
73
+ from rag_wright.contracts.identifiers import ChunkId
74
+ from rag_wright.spans.clause_kg_extractor import clause_to_record
75
+
76
+ constraints: list[tuple[str, str]] = []
77
+ clause = extract_clause(query, extract_model)
78
+ if clause is not None:
79
+ rec = clause_to_record(clause, chunk_id=ChunkId.of("q", 0, query), function="Cap On Liability")
80
+ constraints = [(a.dimension.value, a.value) for a in rec.assertions]
81
+ functions = classify_query_functions(query, function_model_id, k=k)
82
+ return constraints, functions
83
+
84
+
85
+ def register_query_function_classification(registry) -> None:
86
+ """CAP-REG-2: register `query_function_classification` (agent_skill; taxonomy-constrained LLM classifier)."""
87
+ from rag_wright.contracts.function import FunctionClassification
88
+
89
+ registry.register(
90
+ "query_function_classification",
91
+ contract=FunctionClassification,
92
+ kind="agent_skill",
93
+ display_name="Query function classification",
94
+ )
@@ -0,0 +1,109 @@
1
+ """CU-C1: NL->type query understanding -- the front door of the CUAD highlight pipeline.
2
+
3
+ A natural-language question about a KNOWN contract is parsed in ONE structured LLM call into a `QueryIntent`:
4
+ which clause type(s) the user asks about (mapped to the FUNCTION taxonomy), what they want done (highlight /
5
+ extract a value / discriminate among same-type clauses), and whether the ask is in-taxonomy.
6
+
7
+ This is a user-required MVP front door (robust natural-language handling, NOT templates), layered ON TOP of the
8
+ registered capabilities. It is intentionally NOT registered under a canonical capability slug: NL->type is not
9
+ an FR-C/FR-Q capability in the (closed) spec catalog, and registering one would invent a requirement the spec
10
+ does not state. It is ordinary tested software the compiled query graph can call as glue.
11
+
12
+ The strict `QueryIntent` validator rejects non-taxonomy labels, so the LLM emits a LOOSE schema and this module
13
+ NORMALIZES at the boundary (canonicalize labels, drop unmappable ones, derive `in_taxonomy`) -- the "strict
14
+ contracts, normalize at the boundary" rule. Multi-type is allowed. Out-of-taxonomy (nothing maps) ->
15
+ `in_taxonomy=False`, `clause_types=[]` -> the serve stage does semantic fallback + a low-confidence flag.
16
+ `structured_factory` is injectable so the mapping logic is tested hermetically (no LLM, no network).
17
+ """
18
+
19
+ from __future__ import annotations
20
+
21
+ from pydantic import BaseModel
22
+
23
+ from rag_wright.contracts.function import FUNCTION_LABELS, canonical_function
24
+ from rag_wright.contracts.query_intent import QueryIntent
25
+ from rag_wright.models.profiles import ModelRole, model_for
26
+ from rag_wright.models.seam import build_model
27
+ from rag_wright.models.tag_structured import build_tag_structured # ADR-0045: LLM-agnostic client-side output
28
+
29
+ _INTENTS = ("highlight", "extract", "discriminate")
30
+
31
+ # Two-step (reason -> emit), the sanctioned pattern for a model that cannot combine reasoning with a forced
32
+ # structured call (CLAUDE.md standing rule; ADR-0006 Qwen precedent; ADR-0032). Step 1 reasons in free text
33
+ # (a GENERAL-model strength -- no forced tool, so no thinking-mode tool rejection); step 2 emits the schema
34
+ # from that reasoning (the profile disables thinking on this forced call for the models that need it, e.g.
35
+ # Gemma). This makes NL->type work on the cheap GENERAL model, not just DeepSeek Pro (benchmarked in CU-D2).
36
+ _REASON_PROMPT = (
37
+ "You match a user's natural-language question about a SINGLE known contract to clause types from a fixed "
38
+ "taxonomy. Reason briefly about which type(s) the question concerns and what the user wants, then END "
39
+ "with EXACTLY these three lines:\n"
40
+ "TYPES: <comma-separated EXACT taxonomy labels the question is about, or NONE if the concept is absent "
41
+ "from the taxonomy>\n"
42
+ "INTENT: <highlight to locate the clause | extract if the user asks for a specific value inside it | "
43
+ "discriminate if the user wants the one clause matching a condition among several of the same type>\n"
44
+ "VALUE: <the value to extract or the selecting condition, or NONE>\n\n"
45
+ "Taxonomy:\n{taxonomy}\n\nQuestion: {query}"
46
+ )
47
+ _EMIT_PROMPT = (
48
+ "Convert this analysis into the structured intent. clause_types = the EXACT labels listed after TYPES "
49
+ "(empty list if TYPES is NONE). intent = the word after INTENT. value_to_extract = the VALUE when "
50
+ "intent is extract, else null; value_condition = the VALUE when intent is discriminate, else null. "
51
+ "in_taxonomy = false iff TYPES is NONE.\n\nAnalysis:\n{reasoning}"
52
+ )
53
+
54
+
55
+ class _RawIntent(BaseModel):
56
+ """The LOOSE schema the LLM fills; normalized into the strict `QueryIntent` at the boundary."""
57
+
58
+ clause_types: list[str] = []
59
+ intent: str = "highlight"
60
+ value_to_extract: str | None = None
61
+ value_condition: str | None = None
62
+ in_taxonomy: bool = True
63
+ confidence: float = 1.0
64
+
65
+
66
+ def _taxonomy_block() -> str:
67
+ return "\n".join(f"- {label}" for label in FUNCTION_LABELS)
68
+
69
+
70
+ def understand_query(
71
+ query: str,
72
+ *,
73
+ reason_factory=build_model,
74
+ structured_factory=build_tag_structured,
75
+ model_id: str | None = None,
76
+ ) -> QueryIntent:
77
+ """Parse a natural-language question into a `QueryIntent` via the two-step reason->emit (see the prompts
78
+ above): step 1 reasons in free text, step 2 emits the schema from that reasoning. Then boundary
79
+ normalization: `in_taxonomy` is DERIVED from what actually maps (a mapping miss degrades gracefully to
80
+ semantic fallback, never a hard error); a failed emit (None) degrades to out-of-taxonomy low-confidence.
81
+ Defaults to the GENERAL model (Gemma): the two-step ties/beats DeepSeek Pro on NL->type at ~3x less
82
+ latency and cost, and is not throttled (benchmarked CU-D2 / ADR-0032). Factories are injectable for
83
+ hermetic tests."""
84
+ model_id = model_id or model_for(ModelRole.GENERAL)
85
+ reasoning = reason_factory(model_id).invoke(
86
+ _REASON_PROMPT.format(taxonomy=_taxonomy_block(), query=query)
87
+ ).content
88
+ try:
89
+ raw = structured_factory(model_id, _RawIntent).invoke(_EMIT_PROMPT.format(reasoning=reasoning))
90
+ except Exception: # noqa: BLE001 - a persistent client-side parse failure degrades like a None emit
91
+ raw = None
92
+ if raw is None: # the emit failed -> out-of-taxonomy, low confidence (never a hard error)
93
+ return QueryIntent(clause_types=[], intent="highlight", in_taxonomy=False, confidence=0.0)
94
+ canon: list[str] = []
95
+ for label in raw.clause_types:
96
+ mapped = canonical_function(label)
97
+ if mapped is not None and mapped not in canon:
98
+ canon.append(mapped)
99
+ in_taxonomy = bool(canon) # ground truth = did anything map; the LLM's own flag is advisory only
100
+ intent = raw.intent if raw.intent in _INTENTS else "highlight"
101
+ confidence = min(1.0, max(0.0, raw.confidence))
102
+ return QueryIntent(
103
+ clause_types=canon,
104
+ intent=intent,
105
+ value_to_extract=raw.value_to_extract if intent == "extract" else None,
106
+ value_condition=raw.value_condition if intent == "discriminate" else None,
107
+ in_taxonomy=in_taxonomy,
108
+ confidence=confidence,
109
+ )