rag-wright 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (184) hide show
  1. rag_wright/__init__.py +13 -0
  2. rag_wright/api/__init__.py +33 -0
  3. rag_wright/api/config.py +59 -0
  4. rag_wright/api/discover.py +70 -0
  5. rag_wright/api/documents.py +39 -0
  6. rag_wright/api/ids.py +31 -0
  7. rag_wright/api/invoke.py +99 -0
  8. rag_wright/api/kg.py +61 -0
  9. rag_wright/api/mcp.py +94 -0
  10. rag_wright/api/usage.py +30 -0
  11. rag_wright/api/workspace.py +85 -0
  12. rag_wright/capabilities/__init__.py +8 -0
  13. rag_wright/capabilities/answer_generator.py +427 -0
  14. rag_wright/capabilities/ard.py +286 -0
  15. rag_wright/capabilities/assertion_extraction.py +79 -0
  16. rag_wright/capabilities/chunk_read.py +58 -0
  17. rag_wright/capabilities/chunk_write.py +163 -0
  18. rag_wright/capabilities/claim_extraction.py +153 -0
  19. rag_wright/capabilities/clause_exception_linking.py +117 -0
  20. rag_wright/capabilities/compliance_judgment.py +322 -0
  21. rag_wright/capabilities/compliance_store.py +87 -0
  22. rag_wright/capabilities/contract_kg_serve.py +156 -0
  23. rag_wright/capabilities/contract_kg_store.py +251 -0
  24. rag_wright/capabilities/dg_extraction.py +585 -0
  25. rag_wright/capabilities/disambiguation.py +163 -0
  26. rag_wright/capabilities/document_parse.py +87 -0
  27. rag_wright/capabilities/document_scope.py +49 -0
  28. rag_wright/capabilities/embedding.py +164 -0
  29. rag_wright/capabilities/embedding_profiles.py +43 -0
  30. rag_wright/capabilities/entity_resolution.py +154 -0
  31. rag_wright/capabilities/fusion.py +64 -0
  32. rag_wright/capabilities/graph_extraction.py +243 -0
  33. rag_wright/capabilities/graph_query.py +73 -0
  34. rag_wright/capabilities/graph_storage.py +111 -0
  35. rag_wright/capabilities/highlight_serve.py +142 -0
  36. rag_wright/capabilities/hybrid_search.py +65 -0
  37. rag_wright/capabilities/invoke.py +31 -0
  38. rag_wright/capabilities/jev_decision.py +38 -0
  39. rag_wright/capabilities/manifests.py +872 -0
  40. rag_wright/capabilities/okf_navigate.py +456 -0
  41. rag_wright/capabilities/parsing.py +286 -0
  42. rag_wright/capabilities/property_boosted_retrieval.py +125 -0
  43. rag_wright/capabilities/query_function_classifier.py +94 -0
  44. rag_wright/capabilities/query_understanding.py +109 -0
  45. rag_wright/capabilities/registry.py +262 -0
  46. rag_wright/capabilities/remote_encoders.py +94 -0
  47. rag_wright/capabilities/requirement_extraction.py +247 -0
  48. rag_wright/capabilities/reranking.py +123 -0
  49. rag_wright/capabilities/retrieval_core.py +126 -0
  50. rag_wright/capabilities/rlm_chunking.py +808 -0
  51. rag_wright/capabilities/rlm_synthesis.py +316 -0
  52. rag_wright/capabilities/scan_quality.py +136 -0
  53. rag_wright/capabilities/span_relevance_judgment.py +191 -0
  54. rag_wright/capabilities/vision_to_text.py +85 -0
  55. rag_wright/capabilities/vlm_ocr.py +85 -0
  56. rag_wright/contracts/__init__.py +6 -0
  57. rag_wright/contracts/chunk.py +79 -0
  58. rag_wright/contracts/compliance.py +303 -0
  59. rag_wright/contracts/contract_meta.py +27 -0
  60. rag_wright/contracts/extraction.py +130 -0
  61. rag_wright/contracts/function.py +167 -0
  62. rag_wright/contracts/function_routing.py +91 -0
  63. rag_wright/contracts/highlight.py +74 -0
  64. rag_wright/contracts/identifiers.py +153 -0
  65. rag_wright/contracts/jurisdiction.py +96 -0
  66. rag_wright/contracts/ontology.py +142 -0
  67. rag_wright/contracts/property.py +201 -0
  68. rag_wright/contracts/provenance.py +78 -0
  69. rag_wright/contracts/query_intent.py +53 -0
  70. rag_wright/contracts/span.py +76 -0
  71. rag_wright/contracts/value_match.py +84 -0
  72. rag_wright/corpus/__init__.py +0 -0
  73. rag_wright/corpus/canonicalize.py +116 -0
  74. rag_wright/corpus/cuad.py +153 -0
  75. rag_wright/corpus/cuad_ingestion.py +72 -0
  76. rag_wright/corpus/document_parser.py +299 -0
  77. rag_wright/corpus/edgar.py +231 -0
  78. rag_wright/corpus/gcs_ingestion.py +120 -0
  79. rag_wright/corpus/http.py +110 -0
  80. rag_wright/corpus/selection.py +152 -0
  81. rag_wright/mcp/__init__.py +11 -0
  82. rag_wright/mcp/compliance_server.py +299 -0
  83. rag_wright/mcp/intra_document_qa_server.py +170 -0
  84. rag_wright/mcp/relational_qa_server.py +171 -0
  85. rag_wright/mcp/session_store.py +64 -0
  86. rag_wright/mcp/typed_property_retrieval_server.py +191 -0
  87. rag_wright/models/__init__.py +8 -0
  88. rag_wright/models/profiles.py +331 -0
  89. rag_wright/models/seam.py +497 -0
  90. rag_wright/models/tag_structured.py +285 -0
  91. rag_wright/models/tracing.py +179 -0
  92. rag_wright/models/usage.py +102 -0
  93. rag_wright/okf/__init__.py +11 -0
  94. rag_wright/okf/compile.py +292 -0
  95. rag_wright/okf/document.py +47 -0
  96. rag_wright/okf/enrich.py +176 -0
  97. rag_wright/okf/links.py +190 -0
  98. rag_wright/okf/lint.py +105 -0
  99. rag_wright/ontology/__init__.py +6 -0
  100. rag_wright/ontology/_generated_template_meta.py +60 -0
  101. rag_wright/ontology/_generated_vocab.py +52 -0
  102. rag_wright/ontology/clause_template.py +964 -0
  103. rag_wright/ontology/codegen.py +84 -0
  104. rag_wright/ontology/compliance_bridge.ttl +186 -0
  105. rag_wright/ontology/contract_bridge.ttl +2685 -0
  106. rag_wright/ontology/contract_taxonomy.py +24 -0
  107. rag_wright/ontology/derive.py +58 -0
  108. rag_wright/ontology/loader.py +435 -0
  109. rag_wright/ontology/packs/ftc_16cfr255.ttl +29 -0
  110. rag_wright/ontology/registry.py +87 -0
  111. rag_wright/ontology/template_introspect.py +100 -0
  112. rag_wright/py.typed +0 -0
  113. rag_wright/reference/__init__.py +2 -0
  114. rag_wright/reference/compliance.py +41 -0
  115. rag_wright/reference/contract_seam.py +123 -0
  116. rag_wright/skills/__init__.py +7 -0
  117. rag_wright/skills/claim_extraction/SKILL.md +47 -0
  118. rag_wright/skills/claim_extraction/__init__.py +1 -0
  119. rag_wright/skills/claim_extraction/template.py +50 -0
  120. rag_wright/skills/compliance_judgment/SKILL.md +59 -0
  121. rag_wright/skills/corpus_ingest/SKILL.md +106 -0
  122. rag_wright/skills/extraction_semantic_judge/SKILL.md +51 -0
  123. rag_wright/skills/extraction_semantic_judge/__init__.py +1 -0
  124. rag_wright/skills/generation/SKILL.md +64 -0
  125. rag_wright/skills/generation/__init__.py +1 -0
  126. rag_wright/skills/generic_compliance_judgment/SKILL.md +58 -0
  127. rag_wright/skills/okf_navigate/SKILL.md +137 -0
  128. rag_wright/skills/requirement_extraction/SKILL.md +47 -0
  129. rag_wright/skills/requirement_extraction/__init__.py +1 -0
  130. rag_wright/skills/requirement_extraction/template.py +50 -0
  131. rag_wright/skills/rlm/SKILL.md +186 -0
  132. rag_wright/skills/rlm/__init__.py +31 -0
  133. rag_wright/skills/rlm/agent.py +292 -0
  134. rag_wright/skills/span_relevance_judgment/SKILL.md +67 -0
  135. rag_wright/skills/vision_to_text/SKILL.md +36 -0
  136. rag_wright/skills/vision_to_text/__init__.py +1 -0
  137. rag_wright/spans/__init__.py +1 -0
  138. rag_wright/spans/boundary.py +78 -0
  139. rag_wright/spans/clause_function_classifier.py +490 -0
  140. rag_wright/spans/clause_kg_extractor.py +337 -0
  141. rag_wright/spans/cuad_labels.py +81 -0
  142. rag_wright/spans/dim_classifier.py +158 -0
  143. rag_wright/spans/dim_fleet.json +411 -0
  144. rag_wright/spans/function_classifier.py +77 -0
  145. rag_wright/spans/function_families.py +62 -0
  146. rag_wright/spans/hybrid_classifier.py +103 -0
  147. rag_wright/spans/legalbert_classifier.py +83 -0
  148. rag_wright/spans/model_capabilities.py +107 -0
  149. rag_wright/spans/new_function_labels.py +111 -0
  150. rag_wright/spans/page_map.py +68 -0
  151. rag_wright/spans/property_extractor.py +365 -0
  152. rag_wright/spans/property_grounding.py +182 -0
  153. rag_wright/spans/reclassify.py +77 -0
  154. rag_wright/spans/scarce_function_labels.py +105 -0
  155. rag_wright/spans/segment.py +341 -0
  156. rag_wright/spans/semantic_judge.py +197 -0
  157. rag_wright/spans/symbolic_validation.py +131 -0
  158. rag_wright/spans/tag_clause_extractor.py +182 -0
  159. rag_wright/store/__init__.py +6 -0
  160. rag_wright/store/arcadedb.py +1135 -0
  161. rag_wright/store/chunk_text.py +66 -0
  162. rag_wright/store/seam.py +213 -0
  163. rag_wright/subgraphs/__init__.py +0 -0
  164. rag_wright/subgraphs/async_ingestion.py +204 -0
  165. rag_wright/subgraphs/compliance_check.py +1042 -0
  166. rag_wright/subgraphs/compliance_ingestion.py +306 -0
  167. rag_wright/subgraphs/contract_ingestion_pipeline.py +999 -0
  168. rag_wright/subgraphs/graph_extraction.py +102 -0
  169. rag_wright/subgraphs/intra_document_qa.py +328 -0
  170. rag_wright/subgraphs/observability.py +140 -0
  171. rag_wright/subgraphs/query_constraint_extraction.py +73 -0
  172. rag_wright/subgraphs/relational_qa.py +165 -0
  173. rag_wright/subgraphs/requirement_extraction.py +137 -0
  174. rag_wright/subgraphs/scaffold.py +65 -0
  175. rag_wright/subgraphs/semantic_chunking.py +183 -0
  176. rag_wright/subgraphs/typed_clause_extraction.py +172 -0
  177. rag_wright/subgraphs/typed_property_retrieval.py +278 -0
  178. rag_wright/util/__init__.py +1 -0
  179. rag_wright/util/concurrent.py +153 -0
  180. rag_wright/util/spacy_model.py +45 -0
  181. rag_wright-0.1.0.dist-info/METADATA +168 -0
  182. rag_wright-0.1.0.dist-info/RECORD +184 -0
  183. rag_wright-0.1.0.dist-info/WHEEL +4 -0
  184. rag_wright-0.1.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,66 @@
1
+ """The chunk-text sidecar (T40, FR-I.3): the per-chunk full text the retrieval index omits.
2
+
3
+ The index is dense-over-summary by design: `ChunkRecord` holds the summary and the vectors, never the
4
+ raw chunk text (the text is used once to compute the `chunk_id` content hash, then discarded). But
5
+ synthesis (FR-Q.5) extracts over full chunk text, so the text must be persisted somewhere keyed by
6
+ `chunk_id` and rehydrated at query time by `chunk_read` (T38). This is that store: one JSON file per
7
+ source document, mapping the canonical `chunk_id` string to its full text.
8
+
9
+ It is deliberately NOT part of the swappable `Store` seam (ArcadeDB / LanceDB): it holds no vectors and
10
+ answers no query, it only round-trips text by id, so it does not belong behind the retrieval seam. Its
11
+ lifecycle is coupled to the chunk lifecycle: written at chunk write under the same content-hash gate as
12
+ the index upsert (so the two never diverge), and deleted per `source_doc_id` when a document is
13
+ re-chunked (`delete_document`, the T34 seam), so stale text is not orphaned relative to the index.
14
+ """
15
+
16
+ from __future__ import annotations
17
+
18
+ import hashlib
19
+ import json
20
+ from pathlib import Path
21
+ from typing import Optional
22
+
23
+ from rag_wright.contracts.identifiers import ChunkId
24
+
25
+
26
+ class ChunkTextStore:
27
+ """A per-source-document JSON sidecar mapping `chunk_id` -> full chunk text (T40, FR-I.3)."""
28
+
29
+ def __init__(self, root: Path) -> None:
30
+ self._root = Path(root)
31
+ self._root.mkdir(parents=True, exist_ok=True)
32
+
33
+ def put(self, chunk_id: ChunkId, text: str) -> None:
34
+ """Persist `text` for `chunk_id`, upserting by id under the document's sidecar file.
35
+
36
+ Integrity check: the `chunk_id` already carries the content hash of its text (the same hash
37
+ `ChunkId.of` computed at chunking, over these exact bytes), so `text` MUST hash to it. This is
38
+ the sidecar's core promise made provable — a mismatch is a silent-wrong-text bug (a loop index
39
+ error, a mismatched map) that would otherwise surface only as synthesis citing a valid-looking
40
+ but wrong `chunk_id`; it is rejected here at the boundary, before it can be persisted.
41
+ """
42
+ digest = hashlib.sha256(text.encode("utf-8")).hexdigest()
43
+ if digest != chunk_id.content_hash:
44
+ raise ValueError(
45
+ f"chunk-text integrity: text for {chunk_id.value!r} hashes to {digest!r}, "
46
+ f"not the chunk_id's content_hash {chunk_id.content_hash!r}"
47
+ )
48
+ path = self._path(chunk_id.source_doc_id)
49
+ mapping = json.loads(path.read_text(encoding="utf-8")) if path.exists() else {}
50
+ mapping[chunk_id.value] = text
51
+ path.write_text(json.dumps(mapping, ensure_ascii=False), encoding="utf-8")
52
+
53
+ def get(self, chunk_id: str) -> Optional[str]:
54
+ """The persisted text for `chunk_id` (canonical string form), or None if absent."""
55
+ source_doc_id = chunk_id.rsplit(":", 2)[0] # <source_doc_id>:<chunk_index>:<content_hash>
56
+ path = self._path(source_doc_id)
57
+ if not path.exists():
58
+ return None
59
+ return json.loads(path.read_text(encoding="utf-8")).get(chunk_id)
60
+
61
+ def delete_document(self, source_doc_id: str) -> None:
62
+ """Remove all persisted text for a document (the T34 delete-and-re-chunk seam). Idempotent."""
63
+ self._path(source_doc_id).unlink(missing_ok=True)
64
+
65
+ def _path(self, source_doc_id: str) -> Path:
66
+ return self._root / f"{source_doc_id}.json"
@@ -0,0 +1,213 @@
1
+ """The store seam: the swappable interface every store implementation binds (T13, FR-S.5).
2
+
3
+ The store is reached only through this interface so it can be swapped without touching capability
4
+ code: ArcadeDB is the default (one multi-model store for both the hybrid index and the graph,
5
+ FR-S.1), and a separate hybrid vector store (LanceDB) is the eval-gated Phase 1 fallback for the
6
+ retrieval leg (GATE-2). The seam is intentionally semantic, not SQL: it exposes schema readiness and
7
+ introspection, never a query string, so a second implementation can bind it without inheriting
8
+ ArcadeDB's SQL dialect. Write-side and query-side methods are added by the tasks that need them
9
+ (T20, T21, T26); T13 defines only the schema-management surface the foundation needs.
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ from dataclasses import dataclass
15
+ from typing import Optional, Protocol, runtime_checkable
16
+
17
+ from rag_wright.contracts.chunk import ChunkRecord, MetadataValue
18
+
19
+
20
+ class _NotNull:
21
+ """Sentinel for a `kg_edges` where-value meaning `<field> IS NOT NULL` (vs an equality/membership match)."""
22
+
23
+ def __repr__(self) -> str: # pragma: no cover - debug aid
24
+ return "NOT_NULL"
25
+
26
+
27
+ NOT_NULL = _NotNull() # kg_edges where-value: presence test, e.g. edge_where={"dimension": NOT_NULL}
28
+
29
+
30
+ @dataclass(frozen=True)
31
+ class GraphNode:
32
+ """A graph entity node to write (T25). `node_key` is the vertex identity (the resolver's canonical id when
33
+ linked, an `UNLINKED:<key>` surrogate otherwise); `entity_id` is the canonical id, or empty when unlinked."""
34
+
35
+ node_key: str
36
+ entity_id: str
37
+ name: str
38
+ entity_type: str
39
+ confidence: str
40
+ chunk_id: str
41
+
42
+
43
+ @dataclass(frozen=True)
44
+ class GraphEdge:
45
+ """A relationship edge between two entity nodes (by `node_key`), carrying provenance + confidence."""
46
+
47
+ source_key: str
48
+ target_key: str
49
+ relationship_type: str
50
+ confidence: str
51
+ chunk_id: str
52
+
53
+
54
+ @dataclass(frozen=True)
55
+ class KgNode:
56
+ """A typed KG node to upsert (DD-1b, ADR-0117): `type` is the vertex type, `key_field` the identity field to
57
+ upsert on, `props` the fields (including `key_field`) as DOMAIN-NATIVE values. The store encodes each prop by its
58
+ pack-declared storage type -- the caller never serializes to the store's wire format."""
59
+
60
+ type: str
61
+ key_field: str
62
+ props: dict[str, object]
63
+
64
+
65
+ @dataclass(frozen=True)
66
+ class KgEdge:
67
+ """A typed KG edge to create between two nodes identified by (type, key_field, key). `props` are native values
68
+ (edge properties are type-driven: edges declare no storage schema)."""
69
+
70
+ type: str
71
+ from_type: str
72
+ from_key_field: str
73
+ from_key: object
74
+ to_type: str
75
+ to_key_field: str
76
+ to_key: object
77
+ props: dict[str, object]
78
+
79
+
80
+ @runtime_checkable
81
+ class Store(Protocol):
82
+ """A swappable store. The ArcadeDB implementation is the default; a stub proves swappability."""
83
+
84
+ def ensure_schema(self) -> None:
85
+ """Create the chunk-record and graph-node types and their indexes, idempotently."""
86
+
87
+ def type_names(self) -> set[str]:
88
+ """The names of the types (tables/classes) present in the store."""
89
+
90
+ def property_names(self, type_name: str) -> set[str]:
91
+ """The property names declared on `type_name` (empty if the type is absent)."""
92
+
93
+ def index_names(self) -> set[str]:
94
+ """The names of the indexes present in the store."""
95
+
96
+ def ping(self) -> bool:
97
+ """True if the store is reachable."""
98
+
99
+ def close(self) -> None:
100
+ """Release any resources held by the implementation."""
101
+
102
+ # --- write-side (T20): the chunk-record write leg. Semantic, not SQL: the seam takes the T3
103
+ # ChunkRecord and each store maps it to its own representation (ArcadeDB decomposes the sparse
104
+ # vector into two parallel arrays; a LanceDB fallback would store it its own way).
105
+
106
+ def upsert_chunk(self, record: ChunkRecord) -> None:
107
+ """Write a chunk record, upserting by `chunk_id` (re-write of the same id updates in place)."""
108
+
109
+ def get_chunk(self, chunk_id: str) -> Optional[dict]:
110
+ """The stored row for `chunk_id` (store-native fields), or None if absent."""
111
+
112
+ def chunk_count(self) -> int:
113
+ """The number of chunk records in the store."""
114
+
115
+ # --- query-side (T21): the hybrid retrieval leg. Semantic, not SQL: the seam takes the two
116
+ # query vectors and each store fuses them its own way (ArcadeDB by server-side RRF over its
117
+ # dense/sparse indexes; a LanceDB fallback by its own hybrid query), so FR-C.3 is swappable.
118
+
119
+ def hybrid_search(
120
+ self,
121
+ dense_query: list[float],
122
+ sparse_query: dict[int, float],
123
+ *,
124
+ k: int,
125
+ filters: Optional[dict[str, MetadataValue]] = None,
126
+ ) -> list[dict]:
127
+ """Fuse the dense and sparse legs into one ranked candidate list (Reciprocal Rank Fusion),
128
+ honoring equality metadata filters, returning up to `k` rows (each with at least `chunk_id`
129
+ and `source_doc_id`) in ranked order, best first."""
130
+
131
+ # --- graph-write (T25): the knowledge-graph leg. Semantic, not SQL: the seam takes entity nodes
132
+ # and relationship edges and each store writes them its own way (ArcadeDB as vertices/edges in
133
+ # one transaction; a fallback store however it models a graph). Nodes/edges carry chunk_id (FR-I.4).
134
+
135
+ def write_graph(self, nodes: list[GraphNode], edges: list[GraphEdge]) -> None:
136
+ """Write entity nodes (upsert by `node_key`) and relationship edges between them in ONE
137
+ transaction (FR-S.1: a chunk and its entities land together), connecting each entity to its
138
+ source chunk. Nodes and edges carry `chunk_id` and confidence (FR-I.4)."""
139
+
140
+ def graph_counts(self) -> dict[str, int]:
141
+ """Counts for introspection/tests: `{'entities': n, 'relationships': m}`."""
142
+
143
+ def entities_by_name(self, name: str) -> list[dict]:
144
+ """Resolve a party NAME to its graph entities (issue 0030): the first step before
145
+ `graph_neighbors`/`graph_query`, which take a `start_entity_id` (an exact node key) and cannot be
146
+ reached from a name otherwise. Returns `[{entity_id, name, entity_type}]` for every stored entity
147
+ whose name normalizes to the same clustering key as `name`, via the SAME `normalize_entity_name`
148
+ the ingestion side uses to merge 'Acme Corp' / 'Acme Corporation' / 'ACME, Inc.' into one entity.
149
+ Normalization is the engine's rule and is applied HERE, so a caller never re-implements it (a raw
150
+ or an already-normalized name both work; the normalization is idempotent). `entity_id` is exactly
151
+ the node key `graph_neighbors`/`graph_query` take as `start_entity_id`. Empty list on no match."""
152
+
153
+ # --- graph-query (T26): relationship traversal. Semantic, not SQL: returns store-agnostic path
154
+ # rows (target + the entity_ids/chunk_ids/confidences along the path) so the capability can shape
155
+ # the cited evidence. Confidence is SURFACED on every path, not filtered on (FR-C.5/FR-Q.3).
156
+
157
+ def graph_neighbors(
158
+ self, entity_id: str, *, relationship_type: str, max_hops: int, documents: Optional[list[str]] = None
159
+ ) -> list[dict]:
160
+ """Traverse `relationship_type` edges from the start entity up to `max_hops`, returning one row
161
+ per reached entity+path: `target_id`, `target_name`, `path_entity_ids`, `path_chunk_ids`,
162
+ `path_confidences`, `hops`. Every edge is surfaced regardless of confidence (T26 does not gate).
163
+ `documents` (issue 0031): scope the traversal to those source documents -- EVERY edge on a path must
164
+ belong to one of them; `None` = the whole graph, `[]` = scope-to-nothing (no rows)."""
165
+
166
+ # --- generic typed-node read (DD-1a, ADR-0117): the backend-agnostic primitive a domain store extension
167
+ # delegates to, so a domain pack never emits store-native SQL. Semantic, not SQL.
168
+
169
+ def kg_read(
170
+ self,
171
+ node_type: str,
172
+ *,
173
+ where: Optional[dict[str, object]] = None,
174
+ fields: Optional[list[str]] = None,
175
+ distinct: Optional[str] = None,
176
+ order_by: Optional[str] = None,
177
+ limit: Optional[int] = None,
178
+ ) -> list[dict]:
179
+ """Read typed nodes of `node_type`. `where` maps a field to a scalar (equality) or a list (membership);
180
+ a list value that is EMPTY means scope-to-nothing and returns `[]` without a query. `distinct` returns the
181
+ distinct values of one field; `fields=None` returns all fields. Equality/membership clauses are AND-ed."""
182
+
183
+ def kg_edges(
184
+ self,
185
+ from_type: Optional[str] = None,
186
+ *,
187
+ where: Optional[dict[str, object]] = None,
188
+ key_range: Optional[tuple[str, object, object]] = None,
189
+ direction: str = "out",
190
+ edge_type: Optional[str] = None,
191
+ edge_where: Optional[dict[str, object]] = None,
192
+ target_where: Optional[dict[str, object]] = None,
193
+ select: dict[str, str],
194
+ ) -> list[dict]:
195
+ """Generic, backend-agnostic edge TRAVERSAL -- the primitive a domain store extension delegates to so it
196
+ never emits store-native traversal SQL (EP-REF-1a, ADR-0117). Three idioms behind one surface:
197
+
198
+ - **node-start traversal** (give a start selector): from the `from_type` nodes matched by `where`
199
+ (scalar=equality, list=membership, `NOT_NULL`=presence; a field in `key_range=(field, lo, hi)` adds
200
+ `field >= lo AND field < hi`, the contract-scope range), follow `direction="out"` (outgoing edges to
201
+ the TARGET vertex) or `"in"` (incoming edges to the SOURCE vertex). `edge_type=None` means every edge
202
+ in that direction. `edge_where` filters the edge, `target_where` the reached vertex.
203
+ - **edge scan** (give NEITHER `where` NOR `key_range`): scan the `edge_type` table directly, filtered by
204
+ `edge_where` -- for edge properties that are not reachable from a node key (e.g. a span id on the edge).
205
+
206
+ `select` maps each output alias to an expression: `c.<f>` (start node), `e.<f>` / `e.@type` (edge),
207
+ `v.<f>` (far vertex) for a traversal; or a bare edge field / `inV().<f>` / `outV().<f>` for an edge scan.
208
+ A membership value that is an EMPTY list means scope-to-nothing -> `[]` with no query. Returns the rows."""
209
+
210
+ def kg_write(self, nodes: list["KgNode"], edges: "Iterable[KgEdge]" = ()) -> None:
211
+ """Upsert typed `nodes` (by each node's `key_field`) then create typed `edges` (FROM/TO by node key), ALL in
212
+ ONE transaction, nodes first so endpoints exist. The caller passes DOMAIN-NATIVE values; the store owns all
213
+ wire encoding, driven by each node type's pack-declared property storage type. Empty input is a no-op."""
File without changes
@@ -0,0 +1,204 @@
1
+ """PROD-3 (ADR-0050): async, job-based ingestion -- submit returns a job_id immediately, the corpus ingests in
2
+ the background with bounded document parallelism, and progress is POLLED via a status read (never a blocking run
3
+ with a spinner).
4
+
5
+ This module holds the durable job model + store (2a). The async runner (2b) and the submit/status MCP tools (2c)
6
+ build on it. The per-document pipeline stays the existing LangGraph subgraph (with its RetryPolicy +
7
+ Increment-1 dead-letter/partial); this is the thin async envelope around `run_corpus_ingestion`'s work.
8
+
9
+ The JobStore is file-based (one <job_id>.json per job): process-independent, so a status reader in another
10
+ process (an MCP call) sees live progress; durable, so a crashed/restarted runner's job record survives; and it
11
+ needs no KG schema change for the MVP. The full version can move the store to ArcadeDB/Postgres (ADR-0050).
12
+ """
13
+ from __future__ import annotations
14
+
15
+ import asyncio
16
+ import os
17
+ import threading
18
+ from datetime import datetime, timezone
19
+ from enum import Enum
20
+ from pathlib import Path
21
+ from typing import Any, Callable
22
+
23
+ from pydantic import BaseModel, Field
24
+
25
+ from rag_wright.subgraphs.contract_ingestion_pipeline import ( # ENG-1 shape + 0009 deferred parse
26
+ PendingDocument,
27
+ aparse_pending,
28
+ build_partial_entry,
29
+ )
30
+
31
+
32
+ def _now() -> str:
33
+ return datetime.now(timezone.utc).isoformat()
34
+
35
+
36
+ class JobStatus(str, Enum):
37
+ QUEUED = "queued" # submitted, not yet started
38
+ RUNNING = "running" # the background runner is processing documents
39
+ SUCCEEDED = "succeeded" # all documents accounted for (ingested / partial / dead-lettered), runner finished
40
+ FAILED = "failed" # the runner itself errored irrecoverably (NOT a per-document failure -> that's dead_lettered)
41
+
42
+
43
+ class IngestionJob(BaseModel):
44
+ """A durable ingestion job. `dead_lettered` / `partial` carry the PROD-3 lossless outcomes (a failure is
45
+ KNOWN here at completion, never grep-only). Progress = `documents_done` / `documents_total`."""
46
+
47
+ job_id: str
48
+ corpus_ref: dict # e.g. {"kind": "gcs", "bucket": ..., "prefix": ..., "include": [...]}
49
+ db: str # the target KG database
50
+ status: JobStatus = JobStatus.QUEUED
51
+ documents_total: int = 0
52
+ documents_done: int = 0 # ingested (incl. partial) + dead-lettered + resume-skipped
53
+ ingested: int = 0
54
+ dead_lettered: list[dict] = Field(default_factory=list)
55
+ partial: list[dict] = Field(default_factory=list)
56
+ party_links: int = 0
57
+ created_at: str = Field(default_factory=_now)
58
+ updated_at: str = Field(default_factory=_now)
59
+ error: str | None = None # a runner-level (not per-document) failure
60
+
61
+ @property
62
+ def done(self) -> bool:
63
+ return self.status in (JobStatus.SUCCEEDED, JobStatus.FAILED)
64
+
65
+
66
+ def _atomic_write(path: Path, text: str) -> None:
67
+ """Write `text` to `path` atomically: write a temp file, then `os.replace` (an atomic rename on POSIX and
68
+ Windows). A concurrent reader therefore sees either the old complete file or the new complete one -- never a
69
+ truncated/empty file mid-write (the JobStore read-mid-write race, ADR-0057 B2e/B4)."""
70
+ tmp = path.with_name(f"{path.name}.tmp.{os.getpid()}")
71
+ tmp.write_text(text, encoding="utf-8")
72
+ os.replace(tmp, path)
73
+
74
+
75
+ class JobStore:
76
+ """File-based job store: one `<job_id>.json` per job under `jobs_dir`. Read-modify-write `update` is safe for
77
+ a SINGLE writer per job (the runner's orchestrator coroutine updates the record; parallel document workers
78
+ report back to it, they do not write the file) -- so no cross-writer race in the MVP."""
79
+
80
+ def __init__(self, jobs_dir: Any) -> None:
81
+ self._dir = Path(jobs_dir)
82
+ self._dir.mkdir(parents=True, exist_ok=True)
83
+
84
+ def _path(self, job_id: str) -> Path:
85
+ return self._dir / f"{job_id}.json"
86
+
87
+ def create(self, job: IngestionJob) -> IngestionJob:
88
+ _atomic_write(self._path(job.job_id), job.model_dump_json(indent=2))
89
+ return job
90
+
91
+ def get(self, job_id: str) -> IngestionJob | None:
92
+ path = self._path(job_id)
93
+ if not path.exists():
94
+ return None
95
+ text = path.read_text(encoding="utf-8")
96
+ # atomic writes (create) mean a reader never sees a partial file; tolerate an empty read defensively
97
+ # (e.g. a truncated legacy write) as "not ready yet" rather than raising.
98
+ return IngestionJob.model_validate_json(text) if text.strip() else None
99
+
100
+ def update(self, job_id: str, **fields: Any) -> IngestionJob:
101
+ """Read-modify-write the job's fields, stamping `updated_at`. Raises KeyError if the job is unknown."""
102
+ job = self.get(job_id)
103
+ if job is None:
104
+ raise KeyError(job_id)
105
+ updated = job.model_copy(update={**fields, "updated_at": _now()})
106
+ return self.create(updated)
107
+
108
+ def list_jobs(self) -> list[IngestionJob]:
109
+ return [IngestionJob.model_validate_json(p.read_text(encoding="utf-8")) for p in sorted(self._dir.glob("*.json"))]
110
+
111
+
112
+ # --- 2b: the async runner (submit returns immediately; a background thread ingests in parallel) ----------------
113
+
114
+
115
+ async def run_job(
116
+ job_id: str,
117
+ documents: list,
118
+ ingest_graph: Any,
119
+ store: JobStore,
120
+ *,
121
+ link_fn: Callable[[], int] = lambda: 0,
122
+ is_done: Callable[[Any], bool] = lambda _doc: False,
123
+ max_concurrency: int = 8,
124
+ ) -> IngestionJob:
125
+ """Ingest a materialized document list in the background with bounded parallelism, updating the job record
126
+ as each document completes (so status polling sees live progress). Per-document dead-letter / partial (the
127
+ Increment-1 lossless outcomes) are accumulated onto the job. A per-document failure NEVER fails the job -- it
128
+ is dead_lettered; only a runner-level error sets status=FAILED. Updates come from THIS single orchestrator
129
+ coroutine (asyncio is single-threaded; `store.update` has no await), so there is no cross-writer file race."""
130
+ try:
131
+ store.update(job_id, status=JobStatus.RUNNING, documents_total=len(documents))
132
+ sem = asyncio.Semaphore(max_concurrency)
133
+ dead_lettered: list[dict] = []
134
+ partial: list[dict] = []
135
+ done = 0
136
+ ingested = 0
137
+
138
+ async def _one(item: Any) -> tuple[str, Any, Any]:
139
+ async with sem: # bound concurrent LLM/DB + PARSE work (rate limits)
140
+ if is_done(item): # RESUME: a prior run already wrote this document (skip before parsing)
141
+ return ("skip", item, None)
142
+ doc = item
143
+ try:
144
+ # 0009-ASYNC-INGEST: parse a deferred doc HERE -- concurrently + deadline-bounded, off the loop
145
+ # (its tiered OCR/VLM escalation is the slowest call) -- so it never blocks the others.
146
+ if isinstance(item, PendingDocument):
147
+ doc = await aparse_pending(item)
148
+ # ASYNC-B2e (ADR-0057): ainvoke runs an async-node graph on the loop (true deadline) and a
149
+ # sync-node graph in LangGraph's threadpool -- so it is safe on any compiled graph.
150
+ out = await ingest_graph.ainvoke({"document": doc}) # per-doc LangGraph graph
151
+ except Exception as exc: # noqa: BLE001 - a per-doc CRASH/parse-timeout must dead-letter THAT doc,
152
+ out = {"dead_letter": { # never fail the whole job
153
+ "source_doc_id": item.source_doc_id, "stage": "invoke",
154
+ "reason": "ingest_crashed", "error": str(exc)[:200]}}
155
+ return ("out", doc, out)
156
+
157
+ for coro in asyncio.as_completed([_one(doc) for doc in documents]):
158
+ kind, doc, out = await coro
159
+ done += 1
160
+ if kind == "out" and out.get("dead_letter"):
161
+ dead_lettered.append(out["dead_letter"])
162
+ else:
163
+ ingested += 1
164
+ ocr_failures = [{"page": pg, "reason": "unreadable scan (OCR + VLM failed)"} # 0009-WIRE2
165
+ for pg in (getattr(doc, "ocr_unreadable_pages", None) or [])]
166
+ entry = build_partial_entry( # ENG-1: same shape as the blocking driver; also surfaces span/ocr losses
167
+ doc.source_doc_id, (out or {}).get("clause_failures"), (out or {}).get("span_failures"),
168
+ ocr_failures)
169
+ if entry is not None:
170
+ partial.append(entry)
171
+ store.update(job_id, documents_done=done, ingested=ingested,
172
+ dead_lettered=dead_lettered, partial=partial)
173
+
174
+ links = link_fn() # KG-7 party linking ONCE, after all documents
175
+ return store.update(job_id, status=JobStatus.SUCCEEDED, party_links=links)
176
+ except Exception as exc: # noqa: BLE001 - a RUNNER-level failure (not a per-document one) -> job FAILED, visible
177
+ return store.update(job_id, status=JobStatus.FAILED, error=str(exc)[:500])
178
+
179
+
180
+ def submit_ingestion(
181
+ adapter: Any,
182
+ ingest_graph: Any,
183
+ store: JobStore,
184
+ *,
185
+ job_id: str,
186
+ db: str,
187
+ corpus_ref: dict,
188
+ link_fn: Callable[[], int] = lambda: 0,
189
+ is_done: Callable[[Any], bool] = lambda _doc: False,
190
+ max_concurrency: int = 8,
191
+ ) -> str:
192
+ """Create a QUEUED job, start the background runner, and return the `job_id` IMMEDIATELY (non-blocking). The
193
+ runner ingests in a daemon thread with its own event loop; poll `store.get(job_id)` (or the status MCP tool)
194
+ for progress. NOTE (MVP): the daemon thread lives with the submitting process -- a persistent service/worker
195
+ (or LangGraph Platform / Pub-Sub, the 'full' version) is what survives process exit; documented in ADR-0050."""
196
+ store.create(IngestionJob(job_id=job_id, corpus_ref=corpus_ref, db=db))
197
+
198
+ def _worker() -> None:
199
+ documents = list(adapter.documents()) # materialize inside the worker (may hit the network)
200
+ asyncio.run(run_job(job_id, documents, ingest_graph, store,
201
+ link_fn=link_fn, is_done=is_done, max_concurrency=max_concurrency))
202
+
203
+ threading.Thread(target=_worker, daemon=True, name=f"ingest-{job_id}").start()
204
+ return job_id