rag-wright 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (184) hide show
  1. rag_wright/__init__.py +13 -0
  2. rag_wright/api/__init__.py +33 -0
  3. rag_wright/api/config.py +59 -0
  4. rag_wright/api/discover.py +70 -0
  5. rag_wright/api/documents.py +39 -0
  6. rag_wright/api/ids.py +31 -0
  7. rag_wright/api/invoke.py +99 -0
  8. rag_wright/api/kg.py +61 -0
  9. rag_wright/api/mcp.py +94 -0
  10. rag_wright/api/usage.py +30 -0
  11. rag_wright/api/workspace.py +85 -0
  12. rag_wright/capabilities/__init__.py +8 -0
  13. rag_wright/capabilities/answer_generator.py +427 -0
  14. rag_wright/capabilities/ard.py +286 -0
  15. rag_wright/capabilities/assertion_extraction.py +79 -0
  16. rag_wright/capabilities/chunk_read.py +58 -0
  17. rag_wright/capabilities/chunk_write.py +163 -0
  18. rag_wright/capabilities/claim_extraction.py +153 -0
  19. rag_wright/capabilities/clause_exception_linking.py +117 -0
  20. rag_wright/capabilities/compliance_judgment.py +322 -0
  21. rag_wright/capabilities/compliance_store.py +87 -0
  22. rag_wright/capabilities/contract_kg_serve.py +156 -0
  23. rag_wright/capabilities/contract_kg_store.py +251 -0
  24. rag_wright/capabilities/dg_extraction.py +585 -0
  25. rag_wright/capabilities/disambiguation.py +163 -0
  26. rag_wright/capabilities/document_parse.py +87 -0
  27. rag_wright/capabilities/document_scope.py +49 -0
  28. rag_wright/capabilities/embedding.py +164 -0
  29. rag_wright/capabilities/embedding_profiles.py +43 -0
  30. rag_wright/capabilities/entity_resolution.py +154 -0
  31. rag_wright/capabilities/fusion.py +64 -0
  32. rag_wright/capabilities/graph_extraction.py +243 -0
  33. rag_wright/capabilities/graph_query.py +73 -0
  34. rag_wright/capabilities/graph_storage.py +111 -0
  35. rag_wright/capabilities/highlight_serve.py +142 -0
  36. rag_wright/capabilities/hybrid_search.py +65 -0
  37. rag_wright/capabilities/invoke.py +31 -0
  38. rag_wright/capabilities/jev_decision.py +38 -0
  39. rag_wright/capabilities/manifests.py +872 -0
  40. rag_wright/capabilities/okf_navigate.py +456 -0
  41. rag_wright/capabilities/parsing.py +286 -0
  42. rag_wright/capabilities/property_boosted_retrieval.py +125 -0
  43. rag_wright/capabilities/query_function_classifier.py +94 -0
  44. rag_wright/capabilities/query_understanding.py +109 -0
  45. rag_wright/capabilities/registry.py +262 -0
  46. rag_wright/capabilities/remote_encoders.py +94 -0
  47. rag_wright/capabilities/requirement_extraction.py +247 -0
  48. rag_wright/capabilities/reranking.py +123 -0
  49. rag_wright/capabilities/retrieval_core.py +126 -0
  50. rag_wright/capabilities/rlm_chunking.py +808 -0
  51. rag_wright/capabilities/rlm_synthesis.py +316 -0
  52. rag_wright/capabilities/scan_quality.py +136 -0
  53. rag_wright/capabilities/span_relevance_judgment.py +191 -0
  54. rag_wright/capabilities/vision_to_text.py +85 -0
  55. rag_wright/capabilities/vlm_ocr.py +85 -0
  56. rag_wright/contracts/__init__.py +6 -0
  57. rag_wright/contracts/chunk.py +79 -0
  58. rag_wright/contracts/compliance.py +303 -0
  59. rag_wright/contracts/contract_meta.py +27 -0
  60. rag_wright/contracts/extraction.py +130 -0
  61. rag_wright/contracts/function.py +167 -0
  62. rag_wright/contracts/function_routing.py +91 -0
  63. rag_wright/contracts/highlight.py +74 -0
  64. rag_wright/contracts/identifiers.py +153 -0
  65. rag_wright/contracts/jurisdiction.py +96 -0
  66. rag_wright/contracts/ontology.py +142 -0
  67. rag_wright/contracts/property.py +201 -0
  68. rag_wright/contracts/provenance.py +78 -0
  69. rag_wright/contracts/query_intent.py +53 -0
  70. rag_wright/contracts/span.py +76 -0
  71. rag_wright/contracts/value_match.py +84 -0
  72. rag_wright/corpus/__init__.py +0 -0
  73. rag_wright/corpus/canonicalize.py +116 -0
  74. rag_wright/corpus/cuad.py +153 -0
  75. rag_wright/corpus/cuad_ingestion.py +72 -0
  76. rag_wright/corpus/document_parser.py +299 -0
  77. rag_wright/corpus/edgar.py +231 -0
  78. rag_wright/corpus/gcs_ingestion.py +120 -0
  79. rag_wright/corpus/http.py +110 -0
  80. rag_wright/corpus/selection.py +152 -0
  81. rag_wright/mcp/__init__.py +11 -0
  82. rag_wright/mcp/compliance_server.py +299 -0
  83. rag_wright/mcp/intra_document_qa_server.py +170 -0
  84. rag_wright/mcp/relational_qa_server.py +171 -0
  85. rag_wright/mcp/session_store.py +64 -0
  86. rag_wright/mcp/typed_property_retrieval_server.py +191 -0
  87. rag_wright/models/__init__.py +8 -0
  88. rag_wright/models/profiles.py +331 -0
  89. rag_wright/models/seam.py +497 -0
  90. rag_wright/models/tag_structured.py +285 -0
  91. rag_wright/models/tracing.py +179 -0
  92. rag_wright/models/usage.py +102 -0
  93. rag_wright/okf/__init__.py +11 -0
  94. rag_wright/okf/compile.py +292 -0
  95. rag_wright/okf/document.py +47 -0
  96. rag_wright/okf/enrich.py +176 -0
  97. rag_wright/okf/links.py +190 -0
  98. rag_wright/okf/lint.py +105 -0
  99. rag_wright/ontology/__init__.py +6 -0
  100. rag_wright/ontology/_generated_template_meta.py +60 -0
  101. rag_wright/ontology/_generated_vocab.py +52 -0
  102. rag_wright/ontology/clause_template.py +964 -0
  103. rag_wright/ontology/codegen.py +84 -0
  104. rag_wright/ontology/compliance_bridge.ttl +186 -0
  105. rag_wright/ontology/contract_bridge.ttl +2685 -0
  106. rag_wright/ontology/contract_taxonomy.py +24 -0
  107. rag_wright/ontology/derive.py +58 -0
  108. rag_wright/ontology/loader.py +435 -0
  109. rag_wright/ontology/packs/ftc_16cfr255.ttl +29 -0
  110. rag_wright/ontology/registry.py +87 -0
  111. rag_wright/ontology/template_introspect.py +100 -0
  112. rag_wright/py.typed +0 -0
  113. rag_wright/reference/__init__.py +2 -0
  114. rag_wright/reference/compliance.py +41 -0
  115. rag_wright/reference/contract_seam.py +123 -0
  116. rag_wright/skills/__init__.py +7 -0
  117. rag_wright/skills/claim_extraction/SKILL.md +47 -0
  118. rag_wright/skills/claim_extraction/__init__.py +1 -0
  119. rag_wright/skills/claim_extraction/template.py +50 -0
  120. rag_wright/skills/compliance_judgment/SKILL.md +59 -0
  121. rag_wright/skills/corpus_ingest/SKILL.md +106 -0
  122. rag_wright/skills/extraction_semantic_judge/SKILL.md +51 -0
  123. rag_wright/skills/extraction_semantic_judge/__init__.py +1 -0
  124. rag_wright/skills/generation/SKILL.md +64 -0
  125. rag_wright/skills/generation/__init__.py +1 -0
  126. rag_wright/skills/generic_compliance_judgment/SKILL.md +58 -0
  127. rag_wright/skills/okf_navigate/SKILL.md +137 -0
  128. rag_wright/skills/requirement_extraction/SKILL.md +47 -0
  129. rag_wright/skills/requirement_extraction/__init__.py +1 -0
  130. rag_wright/skills/requirement_extraction/template.py +50 -0
  131. rag_wright/skills/rlm/SKILL.md +186 -0
  132. rag_wright/skills/rlm/__init__.py +31 -0
  133. rag_wright/skills/rlm/agent.py +292 -0
  134. rag_wright/skills/span_relevance_judgment/SKILL.md +67 -0
  135. rag_wright/skills/vision_to_text/SKILL.md +36 -0
  136. rag_wright/skills/vision_to_text/__init__.py +1 -0
  137. rag_wright/spans/__init__.py +1 -0
  138. rag_wright/spans/boundary.py +78 -0
  139. rag_wright/spans/clause_function_classifier.py +490 -0
  140. rag_wright/spans/clause_kg_extractor.py +337 -0
  141. rag_wright/spans/cuad_labels.py +81 -0
  142. rag_wright/spans/dim_classifier.py +158 -0
  143. rag_wright/spans/dim_fleet.json +411 -0
  144. rag_wright/spans/function_classifier.py +77 -0
  145. rag_wright/spans/function_families.py +62 -0
  146. rag_wright/spans/hybrid_classifier.py +103 -0
  147. rag_wright/spans/legalbert_classifier.py +83 -0
  148. rag_wright/spans/model_capabilities.py +107 -0
  149. rag_wright/spans/new_function_labels.py +111 -0
  150. rag_wright/spans/page_map.py +68 -0
  151. rag_wright/spans/property_extractor.py +365 -0
  152. rag_wright/spans/property_grounding.py +182 -0
  153. rag_wright/spans/reclassify.py +77 -0
  154. rag_wright/spans/scarce_function_labels.py +105 -0
  155. rag_wright/spans/segment.py +341 -0
  156. rag_wright/spans/semantic_judge.py +197 -0
  157. rag_wright/spans/symbolic_validation.py +131 -0
  158. rag_wright/spans/tag_clause_extractor.py +182 -0
  159. rag_wright/store/__init__.py +6 -0
  160. rag_wright/store/arcadedb.py +1135 -0
  161. rag_wright/store/chunk_text.py +66 -0
  162. rag_wright/store/seam.py +213 -0
  163. rag_wright/subgraphs/__init__.py +0 -0
  164. rag_wright/subgraphs/async_ingestion.py +204 -0
  165. rag_wright/subgraphs/compliance_check.py +1042 -0
  166. rag_wright/subgraphs/compliance_ingestion.py +306 -0
  167. rag_wright/subgraphs/contract_ingestion_pipeline.py +999 -0
  168. rag_wright/subgraphs/graph_extraction.py +102 -0
  169. rag_wright/subgraphs/intra_document_qa.py +328 -0
  170. rag_wright/subgraphs/observability.py +140 -0
  171. rag_wright/subgraphs/query_constraint_extraction.py +73 -0
  172. rag_wright/subgraphs/relational_qa.py +165 -0
  173. rag_wright/subgraphs/requirement_extraction.py +137 -0
  174. rag_wright/subgraphs/scaffold.py +65 -0
  175. rag_wright/subgraphs/semantic_chunking.py +183 -0
  176. rag_wright/subgraphs/typed_clause_extraction.py +172 -0
  177. rag_wright/subgraphs/typed_property_retrieval.py +278 -0
  178. rag_wright/util/__init__.py +1 -0
  179. rag_wright/util/concurrent.py +153 -0
  180. rag_wright/util/spacy_model.py +45 -0
  181. rag_wright-0.1.0.dist-info/METADATA +168 -0
  182. rag_wright-0.1.0.dist-info/RECORD +184 -0
  183. rag_wright-0.1.0.dist-info/WHEEL +4 -0
  184. rag_wright-0.1.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,299 @@
1
+ """DOCPARSE-1 (ADR-0049 generic-customer lens): the generic raw-document parser -- the SHARED front-end that lets a
2
+ customer's own PDF / DOCX / HTML flow into BOTH ingestion sides.
3
+
4
+ The docling parse capability already exists (`capabilities/parsing.py`: `DocumentConverter` -> `DoclingDocument`);
5
+ this module adds (a) a BYTES entry point (customer docs arrive as bytes, e.g. from GCS, not a local path) and the
6
+ two PROJECTIONS the two sides need from a parsed document:
7
+ - `document_to_text` -> the contract side (`GcsCorpusAdapter.parse_bytes` seam): one text blob per document.
8
+ - `document_to_sections` -> the compliance side (`RegulationAdapter`'s `[{section, heading, text}]` shape): the
9
+ document split into sections at its headings, so a policy PDF ingests the same way an eCFR `sections.json` does.
10
+
11
+ The DoclingDocument API is grounded (framework graph + installed `inspect`): `export_to_markdown()` for text;
12
+ `iterate_items() -> (item, level)` with heading labels SECTION_HEADER / TITLE / FIELD_HEADING (PAGE_HEADER is
13
+ page furniture, not a section boundary)."""
14
+ from __future__ import annotations
15
+
16
+ import re
17
+ import tempfile
18
+ from dataclasses import dataclass
19
+ from pathlib import Path
20
+ from typing import Any, Optional
21
+
22
+ from docling_core.types.doc.labels import DocItemLabel
23
+
24
+ # Labels that START a new section (a real heading), vs PAGE_HEADER which is running page furniture (ignored).
25
+ _HEADING_LABELS = frozenset({DocItemLabel.SECTION_HEADER, DocItemLabel.TITLE, DocItemLabel.FIELD_HEADING})
26
+ _TEXT_EXTS = frozenset({"txt", "md", "text"})
27
+
28
+
29
+ # 0009-WIRE2: a generous per-document OCR ceiling. A degraded multi-page doc escalated whole to the VLM is a
30
+ # batch op (~30s/page), so it needs more than the per-model-call deadline; per-page escalation would let us
31
+ # tighten this. Overridable via `deadline_s`.
32
+ _OCR_PARSE_DEADLINE_S = 600.0
33
+
34
+
35
+ def _default_document_parser() -> Any:
36
+ """0009-WIRE2: the default parser for raw-document bytes is the TIERED OCR parser -- fast OCR, then a
37
+ scan-quality gate escalates only degraded pages to the VLM (default Gemma-4 via OpenRouter), and flags what
38
+ even the VLM cannot read. Graceful degrade when no VLM is configured. So BOTH ingestion pipelines and the MCP
39
+ document tool get degraded-scan handling through this one chokepoint."""
40
+ from rag_wright.capabilities.parsing import TieredOCRParser
41
+
42
+ return TieredOCRParser()
43
+
44
+
45
+ def parse_document_bytes(name: str, data: bytes, *, parser: Any = None) -> Any:
46
+ """Parse raw document BYTES into a `DoclingDocument`. `.txt`/`.md` bytes are wrapped directly; binary docs
47
+ (PDF/DOCX/HTML) go through docling. `parser` defaults to the tiered OCR parser (0009-WIRE2), injectable for
48
+ tests. `name` supplies the file extension docling needs to pick a backend."""
49
+ ext = name.rsplit(".", 1)[-1].lower() if "." in name else ""
50
+ parser = parser or _default_document_parser()
51
+ # docling reads a file path, not raw bytes -> write to a temp file preserving the extension for backend choice
52
+ suffix = f".{ext}" if ext else ".txt"
53
+ tmp = Path(tempfile.mkdtemp(prefix="docparse_")) / f"doc{suffix}"
54
+ tmp.write_bytes(data)
55
+ return parser.convert(tmp)
56
+
57
+
58
+ async def aparse_document_bytes(name: str, data: bytes, *, parser: Any = None,
59
+ deadline_s: float = _OCR_PARSE_DEADLINE_S) -> Any:
60
+ """ASYNC-bounded document parse (ADR-0057): run the sync `parse_document_bytes` (incl. the tiered VLM
61
+ escalation -- the slowest call in the pipeline) OFF the event loop via `to_thread`, under a wall-clock
62
+ `asyncio.timeout` so a hung/slow OCR never stalls the async ingestion. NOTE: `to_thread` cannot cancel the
63
+ worker thread, so the deadline unblocks the CALLER (raises TimeoutError); the docling parse thread finishes
64
+ in the background. True cancellation would require routing the vision call through the async model seam."""
65
+ import asyncio
66
+
67
+ async with asyncio.timeout(deadline_s):
68
+ return await asyncio.to_thread(parse_document_bytes, name, data, parser=parser)
69
+
70
+
71
+ def document_to_text(doc: Any) -> str:
72
+ """A parsed document -> one text blob (docling markdown export) -- the contract side's `parse_bytes` output."""
73
+ return doc.export_to_markdown()
74
+
75
+
76
+ @dataclass(frozen=True)
77
+ class ContentItem:
78
+ """issue 0014: one chunkable unit of a parsed document's READING-ORDER body. Duck-typed to a docling text
79
+ item (`.label`, `.level`, `.text`) so the chunker/segmenter consume it unchanged. The reading-order body is a
80
+ SUPERSET of `document.texts`: a docling TABLE lives in `document.tables` and a figure in `document.pictures`,
81
+ NEVER in `.texts`, so a `document.texts`-only chunker silently drops them (never chunked, never indexed, never
82
+ retrievable, and no failure recorded -- the exact 0014 loss). This projection is the single authority for
83
+ 'the document's chunkable content, in reading order'.
84
+
85
+ issue 0032: each item also carries its parse-time PAGE provenance (`page`, 1-based, from `prov[0].page_no`)
86
+ and, when the parser produced one, a `bbox` (l, t, r, b on `page`). This is the provenance the CU-B5 page
87
+ map threads to a span's citation (a scanned-PDF click-through lands on the right page). `bbox` is best-effort
88
+ -- omitted where `prov` has none, and dropped when lines are merged into a paragraph (ambiguous then)."""
89
+
90
+ label: Any
91
+ level: Optional[int]
92
+ text: str
93
+ page: Optional[int] = None # issue 0032: 1-based source page (prov[0].page_no); None when prov is absent
94
+ bbox: Optional[tuple[float, float, float, float]] = None # (l, t, r, b) on `page`; only when prov has one
95
+
96
+
97
+ def _table_content_text(item: Any, doc: Any) -> str:
98
+ """A TABLE item -> its atomic markdown (issue 0014), caption prefixed when docling captured one. The markdown
99
+ keeps the header row with the data rows, so the fee/payment schedule retrieves as a unit."""
100
+ caption = (item.caption_text(doc) or "").strip()
101
+ body = (item.export_to_markdown(doc) or "").strip()
102
+ return f"{caption}\n\n{body}".strip() if caption else body
103
+
104
+
105
+ def _picture_content_text(item: Any, doc: Any) -> str:
106
+ """A PICTURE item -> its extractable text (issue 0014): the caption plus any description annotation
107
+ (a VLM/description the tiered OCR attached). Empty when the figure carries no text -- nothing to index."""
108
+ parts: list[str] = []
109
+ caption = (item.caption_text(doc) or "").strip()
110
+ if caption:
111
+ parts.append(caption)
112
+ for annotation in getattr(item, "annotations", None) or []:
113
+ text = (getattr(annotation, "text", "") or "").strip() # DescriptionAnnotation (figure description)
114
+ if text:
115
+ parts.append(text)
116
+ return "\n\n".join(parts)
117
+
118
+
119
+ _ENDS_SENTENCE = re.compile(r"""[.:;?!]["')\]]*$""") # a line that completes a sentence (terminal punct + closers)
120
+
121
+
122
+ def _ends_sentence(text: str) -> bool:
123
+ return bool(_ENDS_SENTENCE.search(text.rstrip()))
124
+
125
+
126
+ def _starts_new_sentence(text: str) -> bool:
127
+ """A line that BEGINS a new provision: its first non-space char is a capital, a digit, or an opener
128
+ (`(`, `[`, quote, `§`, bullet). A lowercase start is a wrapped continuation ('and (ii) ...')."""
129
+ stripped = text.lstrip()
130
+ if not stripped:
131
+ return False
132
+ c = stripped[0]
133
+ return c.isupper() or c.isdigit() or c in "([{\"'§•-"
134
+
135
+
136
+ def _join_wrapped(prev: str, nxt: str) -> str:
137
+ """Join a wrapped continuation to its paragraph: de-hyphenate a soft line-break (`Distribu-` + `tor` ->
138
+ `Distributor`), otherwise a single space."""
139
+ if prev.endswith("-") and len(prev) >= 2 and prev[-2].isalpha():
140
+ return prev[:-1] + nxt
141
+ return f"{prev} {nxt}"
142
+
143
+
144
+ def _merge_wrapped_lines(items: list[ContentItem]) -> list[ContentItem]:
145
+ """DEFRAG-1: reconstruct paragraphs from docling's per-LINE text items. docling emits each PDF text line as its
146
+ own item; joined with `\\n\\n` and split by `segment_clause`, one clause shatters into per-line fragments that
147
+ then fail extraction (NEONSYSTEMS: 219 segments / 121 degenerate -> 87 / 9 after this pass). Consecutive
148
+ `TEXT` items are merged into one paragraph, breaking ONLY when the previous line ends a sentence AND the next
149
+ starts one (two-sided, so an abbreviation like 'Inc.' followed by a lowercase 'and' does not false-split, and a
150
+ clean paragraph-per-item document is left untouched). Any NON-text item (heading, table, figure, list item,
151
+ page furniture) is a hard boundary -- never merged across."""
152
+ out: list[ContentItem] = []
153
+ buf: str = ""
154
+ buf_level: Optional[int] = None
155
+ buf_page: Optional[int] = None # issue 0032: the merged paragraph keeps its FIRST line's page (bbox dropped)
156
+ for item in items:
157
+ if item.label != DocItemLabel.TEXT: # heading / table / picture / list-item / furniture -> hard boundary
158
+ if buf.strip():
159
+ out.append(ContentItem(label=DocItemLabel.TEXT, level=buf_level, text=buf, page=buf_page))
160
+ buf, buf_level, buf_page = "", None, None
161
+ out.append(item)
162
+ continue
163
+ line = (item.text or "").strip()
164
+ if not line:
165
+ continue
166
+ if not buf:
167
+ buf, buf_level, buf_page = line, item.level, item.page
168
+ elif _ends_sentence(buf) and _starts_new_sentence(line): # a real paragraph break
169
+ out.append(ContentItem(label=DocItemLabel.TEXT, level=buf_level, text=buf, page=buf_page))
170
+ buf, buf_level, buf_page = line, item.level, item.page
171
+ else: # a wrapped continuation of the same clause
172
+ buf = _join_wrapped(buf, line)
173
+ if buf.strip():
174
+ out.append(ContentItem(label=DocItemLabel.TEXT, level=buf_level, text=buf, page=buf_page))
175
+ return out
176
+
177
+
178
+ def _prov_page_bbox(node: Any) -> tuple[Optional[int], Optional[tuple[float, float, float, float]]]:
179
+ """issue 0032: a docling item's page (1-based) and best-effort bbox from `prov[0]` (the same provenance
180
+ `scan_quality._page_texts` reads). Returns `(None, None)` when the item has no provenance (a test stub or a
181
+ born-item with none). The bbox is `(l, t, r, b)` when the parser produced one, else `None`."""
182
+ prov = getattr(node, "prov", None) or []
183
+ if not prov:
184
+ return None, None
185
+ page = getattr(prov[0], "page_no", None)
186
+ box = getattr(prov[0], "bbox", None)
187
+ bbox = None
188
+ if box is not None:
189
+ try:
190
+ bbox = (float(box.l), float(box.t), float(box.r), float(box.b))
191
+ except (AttributeError, TypeError, ValueError):
192
+ bbox = None
193
+ return (int(page) if page is not None else None), bbox
194
+
195
+
196
+ def _node_text(node: Any) -> str:
197
+ """issue 0039: a docling ENUMERATED list item (a numbered contract provision) carries its number in `marker`
198
+ and STRIPS it from `.text` ('1.1.' + text -> text). The chunker/segmenter/provision-detector read the text,
199
+ so the section number vanishes and `starts_new_provision` never fires -> every provision falls back to the
200
+ chunk boundary (issue 0038 grouping degrades to chunk-level). Reconstruct the original by prepending the
201
+ marker (docling's own `orig`), so the number is present exactly where the detector looks. Only for an
202
+ `enumerated` node with a marker; a bullet/letter marker is prepended too (it restores the original and does
203
+ NOT trip the numeric section detector, so those list items still fold into their provision)."""
204
+ text = getattr(node, "text", "") or ""
205
+ marker = (getattr(node, "marker", "") or "").strip()
206
+ if marker and getattr(node, "enumerated", False) and not text.lstrip().startswith(marker):
207
+ return f"{marker} {text}".strip()
208
+ return text
209
+
210
+
211
+ def content_items(doc: Any) -> list[ContentItem]:
212
+ """A parsed document -> its READING-ORDER chunkable content items (issue 0014). Walks `iterate_items` over the
213
+ BODY and FURNITURE layers (so everything in `document.texts`, incl. page furniture, is covered), mapping each
214
+ item to a `ContentItem`: a TABLE -> its atomic markdown, a PICTURE -> caption + description text, any other
215
+ item -> its `.text`. A table/picture with no extractable text yields an empty-text item (the chunker strips it
216
+ exactly as it strips an empty text item today) -- present in reading order, never silently missing.
217
+
218
+ NO-SILENT-LOSS GUARANTEE (0014, the issue's Q2): any `document.texts` item the reading-order walk did not
219
+ visit is appended, so the projection is a strict SUPERSET of `.texts` -- a content item can never vanish
220
+ upstream of the failure accounting the way a table did before this fix."""
221
+ from docling_core.types.doc.document import ContentLayer
222
+
223
+ def _text_item(node: Any) -> ContentItem:
224
+ page, bbox = _prov_page_bbox(node)
225
+ return ContentItem(label=getattr(node, "label", None), level=getattr(node, "level", None),
226
+ text=_node_text(node), page=page, bbox=bbox) # issue 0039: keep the enumerated marker
227
+
228
+ if not hasattr(doc, "iterate_items"): # a plain `.texts`-bearing view (a `_SubDocument` slice / test stub):
229
+ return [_text_item(t) for t in getattr(doc, "texts", None) or []] # no reading-order body, no tables
230
+
231
+ layers = {ContentLayer.BODY, ContentLayer.FURNITURE}
232
+ items: list[ContentItem] = []
233
+ seen: set[int] = set()
234
+ for node, _level in doc.iterate_items(included_content_layers=layers):
235
+ seen.add(id(node))
236
+ label = getattr(node, "label", None)
237
+ if label == DocItemLabel.TABLE and hasattr(node, "export_to_markdown"):
238
+ text = _table_content_text(node, doc)
239
+ elif label == DocItemLabel.PICTURE:
240
+ text = _picture_content_text(node, doc)
241
+ else:
242
+ text = _node_text(node) # issue 0039: an enumerated list item keeps its section-number marker
243
+ page, bbox = _prov_page_bbox(node)
244
+ items.append(ContentItem(label=label, level=getattr(node, "level", None), text=text,
245
+ page=page, bbox=bbox))
246
+ items = _merge_wrapped_lines(items) # DEFRAG-1: rejoin per-line items into whole-clause paragraphs
247
+ for text_item in getattr(doc, "texts", None) or []: # coverage backstop: never drop a `.texts` item
248
+ if id(text_item) not in seen:
249
+ items.append(_text_item(text_item))
250
+ return items
251
+
252
+
253
+ def _section_number(heading: str, index: int) -> str:
254
+ """A short section id: the leading numeric token of the heading (e.g. '1' from '1. Confidentiality'), else the
255
+ 1-based position -- so the compliance citation is stable and human-meaningful."""
256
+ token = heading.strip().split()[0].rstrip(".").rstrip(")") if heading.strip() else ""
257
+ return token if token and any(c.isdigit() for c in token) else str(index)
258
+
259
+
260
+ def document_to_sections(doc: Any) -> list[dict]:
261
+ """A parsed document -> `[{section, heading, text, pages, bbox}]` split at its headings -- the compliance side's
262
+ shape (`RegulationAdapter` ingests exactly this). Body before the first heading is kept as a leading section
263
+ (heading ""), so nothing is dropped. PAGE_HEADER and whitespace-only items are skipped.
264
+
265
+ issue 0043: each section carries `pages` (the source page(s) its items span, from the parse's per-item
266
+ provenance -- present even on a scan) and a best-effort `bbox`: a single-item section reports that item's box,
267
+ a multi-item section reports None (a section is not one rectangle, and a fabricated box is worse than none)."""
268
+ sections: list[dict] = []
269
+ heading = ""
270
+ body: list[str] = []
271
+ pages: set[int] = set()
272
+ boxes: list[Any] = [] # the (page, bbox) of each item, to pick a single-item section's box
273
+
274
+ def _flush() -> None:
275
+ text = "\n".join(body).strip()
276
+ if heading or text: # keep a section if it has a heading OR any body (never emit a fully empty one)
277
+ bbox = boxes[0] if len(boxes) == 1 else None # best-effort: only a single-item section has one box
278
+ sections.append({"section": _section_number(heading, len(sections) + 1), "heading": heading,
279
+ "text": text, "pages": sorted(pages), "bbox": bbox})
280
+
281
+ for item, _level in doc.iterate_items():
282
+ label = getattr(item, "label", None)
283
+ text = (getattr(item, "text", "") or "").strip()
284
+ page, box = _prov_page_bbox(item)
285
+ if label in _HEADING_LABELS:
286
+ _flush() # close the previous section
287
+ heading, body, pages, boxes = text, [], set(), []
288
+ if page is not None:
289
+ pages.add(page) # the heading's page belongs to its section
290
+ elif label == DocItemLabel.PAGE_HEADER:
291
+ continue # running page furniture -> neither a boundary nor body
292
+ elif text:
293
+ body.append(text)
294
+ if page is not None:
295
+ pages.add(page)
296
+ if box is not None:
297
+ boxes.append(box)
298
+ _flush() # the final section
299
+ return sections
@@ -0,0 +1,231 @@
1
+ """EDGAR entity acquisition logic (T7, docs/archive/plans/Corpus_Acquisition.md).
2
+
3
+ Pure logic; the throttled/cached fetch and the CLI live in `scripts/acquire_edgar.py`.
4
+
5
+ `normalize_cik` is the **single** canonical CIK->EntityId normalization point: a CIK is 10-digit
6
+ zero-padded, which is exactly the canonical `EntityId` (T1), so normalization is zero-pad-to-10 then
7
+ validate against the strict contract. T8's registry loader **reuses this exact function** so the two
8
+ cannot drift from the T1 contract.
9
+
10
+ Name->CIK proposals are mechanical and structurally **UNVERIFIED**: linking a contract party to a
11
+ CIK is itself the entity-resolution problem (FR-C.7), so a fuzzy/mechanical proposal is never ground
12
+ truth. Each proposal carries an explicit `status` (T7 only ever emits UNVERIFIED; VERIFIED is set by
13
+ human verification at T10), and the CLI writes proposals to a separate `proposed/` location. The
14
+ `MatchCoverage` records which parties resolved and which did not, so T10 verification starts from a
15
+ known map.
16
+ """
17
+
18
+ from __future__ import annotations
19
+
20
+ import re
21
+ from enum import Enum
22
+
23
+ from pydantic import BaseModel
24
+
25
+ from rag_wright.contracts.identifiers import EntityId
26
+
27
+ _CIK_PREFIX = re.compile(r"(?i)^cik[-:_ ]*")
28
+ _NON_ALNUM = re.compile(r"[^a-z0-9]+")
29
+
30
+ # Read this before treating the unresolved set as "not entities" (a note for the T10 verifier).
31
+ # Conservative normalized-conformed-name matching systematically leaves two classes unresolved, by
32
+ # design, not as a bug: (1) private companies and individuals, who are not in EDGAR at all (only
33
+ # public filers are), so a public-company-to-private-counterparty contract resolves one side only;
34
+ # and (2) name variants not close to the conformed name (subsidiaries filing under a parent, former
35
+ # names, DBAs), some of which the submissions former-names data can later help resolve. Both are
36
+ # exactly the cases human verification (T10) exists for. Consequence for the eval: graph entity
37
+ # coverage will be public-filer-centric (FR-C.7), which matters when reading multi-hop results.
38
+ UNRESOLVED_NOTE = (
39
+ "Unresolved does NOT mean 'not an entity'. Conservative conformed-name matching intentionally "
40
+ "misses private companies / individuals (absent from EDGAR) and name variants (subsidiaries, "
41
+ "former names, DBAs). These are the cases human verification (T10) exists for; graph entity "
42
+ "coverage is therefore public-filer-centric (FR-C.7)."
43
+ )
44
+
45
+
46
+ def normalize_cik(raw: int | str) -> EntityId:
47
+ """Normalize a raw EDGAR CIK (int, unpadded, or ``CIK``-prefixed) to a canonical `EntityId`.
48
+
49
+ This is the one place messy EDGAR CIK forms become the canonical identifier; T8 reuses it.
50
+ """
51
+ if isinstance(raw, bool): # bool is an int subclass; reject explicitly
52
+ raise ValueError("CIK must be an int or digit string, not bool")
53
+ if isinstance(raw, int):
54
+ digits = str(raw)
55
+ elif isinstance(raw, str):
56
+ digits = _CIK_PREFIX.sub("", raw.strip())
57
+ else:
58
+ raise ValueError(f"CIK must be an int or str, got {type(raw).__name__}")
59
+ if not digits.isdigit():
60
+ raise ValueError(f"CIK must be numeric, got {raw!r}")
61
+ if len(digits) > 10:
62
+ raise ValueError(f"CIK must be at most 10 digits, got {raw!r}")
63
+ return EntityId.of(digits.zfill(10))
64
+
65
+
66
+ def normalize_name(name: str) -> str:
67
+ """Fold a company/party name to a comparison key (lowercase, alphanumeric runs collapsed)."""
68
+ return _NON_ALNUM.sub(" ", name.lower()).strip()
69
+
70
+
71
+ class MatchStatus(str, Enum):
72
+ """The verification state of a name->CIK match. T7 emits only UNVERIFIED."""
73
+
74
+ UNVERIFIED = "UNVERIFIED" # mechanical proposal; NOT ground truth until human-verified (T10)
75
+ VERIFIED = "VERIFIED" # set only by human verification at T10
76
+
77
+
78
+ class NameToCikProposal(BaseModel):
79
+ """A mechanical, UNVERIFIED name->CIK proposal (never ground truth until verified at T10)."""
80
+
81
+ party_name: str
82
+ proposed_cik: str # canonical 10-digit EntityId value
83
+ proposed_conformed_name: str # the EDGAR conformed name matched
84
+ match_method: str # how the proposal was made, e.g. "normalized_conformed_name"
85
+ status: MatchStatus = MatchStatus.UNVERIFIED
86
+
87
+
88
+ class MatchCoverage(BaseModel):
89
+ """Which parties got a proposed CIK and which did not, so T10 starts from a known map."""
90
+
91
+ resolved: list[NameToCikProposal]
92
+ unresolved: list[str]
93
+
94
+
95
+ class LooseCandidate(BaseModel):
96
+ """A looser token-overlap CIK candidate, shown WITH its evidence for human confirmation.
97
+
98
+ A proposal to eyeball, never an auto-commit: the registry name and the tokens it matched on are
99
+ surfaced so a common-token collision ("Federated", "Premier", "Excite") is caught on the
100
+ evidence, not confirmed on a bare CIK. Always UNVERIFIED until a human approves it.
101
+ """
102
+
103
+ proposed_cik: str
104
+ registry_name: str # the EDGAR conformed name matched (the evidence)
105
+ matched_tokens: list[str]
106
+ score: float # Jaccard token overlap
107
+ status: MatchStatus = MatchStatus.UNVERIFIED
108
+
109
+
110
+ _CIK_TAG = re.compile(r"<cik>(\d+)</cik>", re.IGNORECASE)
111
+
112
+
113
+ def parse_browse_edgar_ciks(atom_xml: str) -> list[str]:
114
+ """CIKs from an EDGAR `browse-edgar ...&output=atom` company-search response (delisted filers
115
+ included). Returns canonical 10-digit CIKs, deduped in order; multiple means an ambiguous name."""
116
+ seen: list[str] = []
117
+ for match in _CIK_TAG.finditer(atom_xml):
118
+ try:
119
+ value = normalize_cik(match.group(1)).value
120
+ except ValueError:
121
+ continue
122
+ if value not in seen:
123
+ seen.append(value)
124
+ return seen
125
+
126
+
127
+ class EdgarEvidence(BaseModel):
128
+ """Grounded EDGAR evidence for a CIK, to confirm on (former_names resolve dot-com name changes)."""
129
+
130
+ proposed_cik: str
131
+ registry_name: str
132
+ former_names: list[str] = []
133
+ tickers: list[str] = []
134
+ source: str = "edgar_submissions"
135
+ status: MatchStatus = MatchStatus.UNVERIFIED
136
+
137
+
138
+ def parse_submissions_evidence(submissions: dict) -> EdgarEvidence:
139
+ """The conformed name, former names, and tickers from an EDGAR submissions record."""
140
+ return EdgarEvidence(
141
+ proposed_cik=normalize_cik(submissions["cik"]).value,
142
+ registry_name=submissions.get("name", ""),
143
+ former_names=[f["name"] for f in submissions.get("formerNames", []) if f.get("name")],
144
+ tickers=submissions.get("tickers", []),
145
+ )
146
+
147
+
148
+ def former_names(submissions: dict) -> list[str]:
149
+ """Former company names from an EDGAR submissions record (alias handling; reused by T24)."""
150
+ return [f["name"] for f in submissions.get("formerNames", []) if f.get("name")]
151
+
152
+
153
+ def loose_cik_candidates(
154
+ name: str, company_tickers: list[dict], *, top_n: int = 3, min_overlap: float = 0.34
155
+ ) -> list[LooseCandidate]:
156
+ """Token-overlap CIK candidates for a mention, ranked, each carrying its match evidence."""
157
+ query = set(normalize_name(name).split())
158
+ if not query:
159
+ return []
160
+ scored: list[tuple[float, set[str], dict]] = []
161
+ for row in company_tickers:
162
+ title_tokens = set(normalize_name(row["title"]).split())
163
+ overlap = query & title_tokens
164
+ if not overlap:
165
+ continue
166
+ jaccard = len(overlap) / len(query | title_tokens)
167
+ if jaccard >= min_overlap:
168
+ scored.append((jaccard, overlap, row))
169
+ scored.sort(key=lambda s: (-s[0], s[2]["title"]))
170
+ return [
171
+ LooseCandidate(
172
+ proposed_cik=normalize_cik(row["cik_str"]).value,
173
+ registry_name=row["title"],
174
+ matched_tokens=sorted(overlap),
175
+ score=round(jaccard, 3),
176
+ )
177
+ for jaccard, overlap, row in scored[:top_n]
178
+ ]
179
+
180
+
181
+ def propose_matches(
182
+ party_names: list[str], company_tickers: list[dict]
183
+ ) -> MatchCoverage:
184
+ """Propose name->CIK matches mechanically against the EDGAR company registry seed.
185
+
186
+ A conservative normalized-conformed-name match: it proposes only where a party's folded name
187
+ equals an EDGAR conformed name. Everything else is left unresolved for human verification (T10),
188
+ rather than fuzzy-guessed into a circular answer key.
189
+ """
190
+ by_name: dict[str, dict] = {}
191
+ for row in company_tickers:
192
+ by_name.setdefault(normalize_name(row["title"]), row)
193
+
194
+ resolved: list[NameToCikProposal] = []
195
+ unresolved: list[str] = []
196
+ for name in party_names:
197
+ row = by_name.get(normalize_name(name))
198
+ if row is None:
199
+ unresolved.append(name)
200
+ continue
201
+ resolved.append(
202
+ NameToCikProposal(
203
+ party_name=name,
204
+ proposed_cik=normalize_cik(row["cik_str"]).value,
205
+ proposed_conformed_name=row["title"],
206
+ match_method="normalized_conformed_name",
207
+ )
208
+ )
209
+ return MatchCoverage(resolved=resolved, unresolved=unresolved)
210
+
211
+
212
+ def build_edgar_registry(rows, *, aliases_by_cik=None):
213
+ """ADR-0067: the SEC/EDGAR registry BUILDER (moved off the generic EntityRegistry). Build an EntityRegistry
214
+ from `company_tickers.json` rows, keyed by a normalized CIK `EntityId`, with `normalize_name` as the surface
215
+ normalizer. A row whose CIK is invalid is skipped (recorded in `skipped_ids`), never fabricated. The engine's
216
+ EntityRegistry stays domain-neutral; this SEC builder + normalize_cik live in the SEC layer (the plug-in)."""
217
+ from rag_wright.ontology.registry import EntityRegistry, RegistryRecord
218
+
219
+ aliases_by_cik = aliases_by_cik or {}
220
+ registry = EntityRegistry(normalize=normalize_name)
221
+ for row in rows:
222
+ raw_cik = row["cik_str"]
223
+ try:
224
+ entity_id = normalize_cik(raw_cik)
225
+ except ValueError:
226
+ registry.skipped_ids.append(str(raw_cik))
227
+ continue
228
+ registry.add(RegistryRecord(
229
+ entity_id=entity_id, canonical_name=row["title"], ticker=row.get("ticker"),
230
+ aliases=aliases_by_cik.get(entity_id.value, [])))
231
+ return registry
@@ -0,0 +1,120 @@
1
+ """PROD-1 (ADR-0049 generic-customer lens): a GCS `CorpusAdapter` — the first REAL source integration.
2
+
3
+ Where `CuadAdapter` reads a bundled dataset file, this reads a customer's documents straight from a Google Cloud
4
+ Storage prefix (`gs://<bucket>/<prefix>/`), the shape a real onboarding takes: the customer uploads their corpus
5
+ to a bucket, we ingest it. Everything downstream (`run_corpus_ingestion` + the per-document graph) is unchanged
6
+ and corpus-agnostic; only this adapter is GCS-specific.
7
+
8
+ Parse-by-extension: `.txt`/`.md`/`.text` are read directly (the PROD-1 corpus -- MAUD + ContractNLI -- is text);
9
+ binary customer docs (`.pdf`/`.docx`/`.html`) route through an injected `parse_bytes` seam (docling in production),
10
+ kept injectable so this adapter stays hermetically testable and so the doc-parse dependency is explicit, not
11
+ hidden. The google-cloud-storage client is injectable for the same reason (tests pass a fake; production builds a
12
+ real `storage.Client`).
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ from typing import Any, Callable, Iterable, Optional
18
+
19
+ from rag_wright.subgraphs.contract_ingestion_pipeline import SourceDocument
20
+
21
+ _TEXT_EXTS = frozenset({"txt", "md", "text"})
22
+
23
+
24
+ class GcsCorpusAdapter:
25
+ """`CorpusAdapter` over `gs://<bucket>/<prefix>`: list blobs under the prefix, read/parse each, yield one
26
+ `SourceDocument` per document. `client` (a `google.cloud.storage.Client`) and `parse_bytes` (bytes->text for
27
+ non-text blobs) are injected -- production builds them, tests fake them."""
28
+
29
+ def __init__(
30
+ self,
31
+ bucket: str,
32
+ prefix: str,
33
+ *,
34
+ limit: int = 0,
35
+ include: Optional[frozenset] = None,
36
+ client: Any = None,
37
+ parse_bytes: Optional[Callable[[str, bytes], str]] = None,
38
+ parse_doc: Optional[Callable[[str, str, bytes], SourceDocument]] = None,
39
+ ) -> None:
40
+ self._bucket = bucket
41
+ self._prefix = prefix
42
+ self._limit = limit
43
+ self._include = include # if set, only blobs whose BASENAME is in this set (a curated subset ingest)
44
+ self._client = client
45
+ self._parse_bytes = parse_bytes
46
+ # CHUNK-7 (ADR-0058): structure-preserving seam for a non-text blob -- (source_doc_id, name, bytes) ->
47
+ # a SourceDocument carrying `.parsed` (the real docling parse), so the chunker's structural pass fires.
48
+ # Preferred over `parse_bytes` (text-only) when both are set; production injects it via the factory.
49
+ self._parse_doc = parse_doc
50
+
51
+ def _get_client(self) -> Any:
52
+ if self._client is not None:
53
+ return self._client
54
+ from google.cloud import storage # lazy: keep the module import-light + hermetic
55
+
56
+ return storage.Client()
57
+
58
+ def _to_source(self, blob: Any, source_doc_id: str, meta: dict) -> Any: # SourceDocument | PendingDocument | None
59
+ """One blob -> a SourceDocument (or None to skip an empty object). A text blob is read as text; a non-text
60
+ blob prefers the structure-preserving `parse_doc` seam (carries `.parsed`), else the `parse_bytes` text
61
+ seam, else raises (no way to ingest a binary document without a parser)."""
62
+ name = blob.name
63
+ ext = name.rsplit(".", 1)[-1].lower() if "." in name else ""
64
+ if ext in _TEXT_EXTS:
65
+ text = blob.download_as_text()
66
+ return SourceDocument(source_doc_id=source_doc_id, text=text, metadata=meta) if text.strip() else None
67
+ if self._parse_doc is not None: # 0009-ASYNC-INGEST: DEFER download+parse (incl. OCR/VLM escalation) so the
68
+ from rag_wright.subgraphs.contract_ingestion_pipeline import PendingDocument # ingest parses it
69
+ _parse = self._parse_doc # concurrently + bounded
70
+ return PendingDocument(source_doc_id=source_doc_id,
71
+ parse=lambda: _parse(source_doc_id, name, blob.download_as_bytes()), metadata=meta)
72
+ if self._parse_bytes is not None: # legacy text-only seam (no structure)
73
+ text = self._parse_bytes(name, blob.download_as_bytes())
74
+ return SourceDocument(source_doc_id=source_doc_id, text=text, metadata=meta) if text.strip() else None
75
+ raise NotImplementedError(
76
+ f"non-text blob {name!r}: inject `parse_doc` (structure-preserving, the production default) or "
77
+ f"`parse_bytes` (text) to ingest PDF/DOCX/HTML customer documents. The PROD-1 corpus is text (.txt).")
78
+
79
+ def documents(self) -> Iterable[Any]: # SourceDocument (text) or PendingDocument (binary, deferred parse)
80
+ from rag_wright.contracts.identifiers import canonical_source_doc_id
81
+
82
+ client = self._get_client()
83
+ blobs = [b for b in client.list_blobs(self._bucket, prefix=self._prefix) if not b.name.endswith("/")]
84
+ if self._include is not None: # curated subset: keep only the named blobs (by basename)
85
+ blobs = [b for b in blobs if b.name.rsplit("/", 1)[-1] in self._include]
86
+ if self._limit:
87
+ blobs = blobs[: self._limit]
88
+ for blob in blobs:
89
+ basename = blob.name.rsplit("/", 1)[-1]
90
+ meta = {"source": "gcs", "bucket": self._bucket, "blob": blob.name}
91
+ sd = self._to_source(blob, canonical_source_doc_id(basename), meta)
92
+ if sd is not None: # None -> empty/whitespace object skipped (a bad upload never dead-letters the run)
93
+ yield sd
94
+
95
+
96
+ def production_gcs_adapter(bucket: str, prefix: str, *, limit: int = 0,
97
+ include: Optional[frozenset] = None, cache_dir: str = "data/cache/gcs_parse",
98
+ parse_bytes: Optional[Callable[[str, bytes], str]] = None) -> GcsCorpusAdapter:
99
+ """Build the adapter with a real `google.cloud.storage.Client` (application-default credentials, the same auth
100
+ gsutil uses). Lazy import so importing this module needs no GCS client. CHUNK-7 (ADR-0058): a non-text
101
+ customer document (PDF/DOCX/HTML) is docling-parsed with its STRUCTURE PRESERVED (`.parsed`) so the chunker's
102
+ structural pass fires -- no longer flattened to text. `parse_bytes` is a legacy text-only override."""
103
+ from google.auth.exceptions import DefaultCredentialsError
104
+ from google.cloud import storage
105
+
106
+ try:
107
+ client = storage.Client()
108
+ except DefaultCredentialsError as e: # actionable message: the python client needs ADC (gsutil uses gcloud auth)
109
+ raise RuntimeError(
110
+ "GCS python client needs Application Default Credentials. Run once: "
111
+ "`gcloud auth application-default login` (or set GOOGLE_APPLICATION_CREDENTIALS to a service-account "
112
+ "key in production). Note: gsutil/gcloud being authed is NOT sufficient for the python client.") from e
113
+ parse_doc = None
114
+ if parse_bytes is None: # DOCPARSE-1 + CHUNK-7: structure-preserving docling parse for customer documents
115
+ from rag_wright.subgraphs.contract_ingestion_pipeline import parsed_source_document
116
+
117
+ parse_doc = lambda sid, name, data: parsed_source_document( # noqa: E731
118
+ sid, name, data, cache_dir=cache_dir)
119
+ return GcsCorpusAdapter(bucket, prefix, limit=limit, include=include, client=client,
120
+ parse_bytes=parse_bytes, parse_doc=parse_doc)