rag-wright 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- rag_wright/__init__.py +13 -0
- rag_wright/api/__init__.py +33 -0
- rag_wright/api/config.py +59 -0
- rag_wright/api/discover.py +70 -0
- rag_wright/api/documents.py +39 -0
- rag_wright/api/ids.py +31 -0
- rag_wright/api/invoke.py +99 -0
- rag_wright/api/kg.py +61 -0
- rag_wright/api/mcp.py +94 -0
- rag_wright/api/usage.py +30 -0
- rag_wright/api/workspace.py +85 -0
- rag_wright/capabilities/__init__.py +8 -0
- rag_wright/capabilities/answer_generator.py +427 -0
- rag_wright/capabilities/ard.py +286 -0
- rag_wright/capabilities/assertion_extraction.py +79 -0
- rag_wright/capabilities/chunk_read.py +58 -0
- rag_wright/capabilities/chunk_write.py +163 -0
- rag_wright/capabilities/claim_extraction.py +153 -0
- rag_wright/capabilities/clause_exception_linking.py +117 -0
- rag_wright/capabilities/compliance_judgment.py +322 -0
- rag_wright/capabilities/compliance_store.py +87 -0
- rag_wright/capabilities/contract_kg_serve.py +156 -0
- rag_wright/capabilities/contract_kg_store.py +251 -0
- rag_wright/capabilities/dg_extraction.py +585 -0
- rag_wright/capabilities/disambiguation.py +163 -0
- rag_wright/capabilities/document_parse.py +87 -0
- rag_wright/capabilities/document_scope.py +49 -0
- rag_wright/capabilities/embedding.py +164 -0
- rag_wright/capabilities/embedding_profiles.py +43 -0
- rag_wright/capabilities/entity_resolution.py +154 -0
- rag_wright/capabilities/fusion.py +64 -0
- rag_wright/capabilities/graph_extraction.py +243 -0
- rag_wright/capabilities/graph_query.py +73 -0
- rag_wright/capabilities/graph_storage.py +111 -0
- rag_wright/capabilities/highlight_serve.py +142 -0
- rag_wright/capabilities/hybrid_search.py +65 -0
- rag_wright/capabilities/invoke.py +31 -0
- rag_wright/capabilities/jev_decision.py +38 -0
- rag_wright/capabilities/manifests.py +872 -0
- rag_wright/capabilities/okf_navigate.py +456 -0
- rag_wright/capabilities/parsing.py +286 -0
- rag_wright/capabilities/property_boosted_retrieval.py +125 -0
- rag_wright/capabilities/query_function_classifier.py +94 -0
- rag_wright/capabilities/query_understanding.py +109 -0
- rag_wright/capabilities/registry.py +262 -0
- rag_wright/capabilities/remote_encoders.py +94 -0
- rag_wright/capabilities/requirement_extraction.py +247 -0
- rag_wright/capabilities/reranking.py +123 -0
- rag_wright/capabilities/retrieval_core.py +126 -0
- rag_wright/capabilities/rlm_chunking.py +808 -0
- rag_wright/capabilities/rlm_synthesis.py +316 -0
- rag_wright/capabilities/scan_quality.py +136 -0
- rag_wright/capabilities/span_relevance_judgment.py +191 -0
- rag_wright/capabilities/vision_to_text.py +85 -0
- rag_wright/capabilities/vlm_ocr.py +85 -0
- rag_wright/contracts/__init__.py +6 -0
- rag_wright/contracts/chunk.py +79 -0
- rag_wright/contracts/compliance.py +303 -0
- rag_wright/contracts/contract_meta.py +27 -0
- rag_wright/contracts/extraction.py +130 -0
- rag_wright/contracts/function.py +167 -0
- rag_wright/contracts/function_routing.py +91 -0
- rag_wright/contracts/highlight.py +74 -0
- rag_wright/contracts/identifiers.py +153 -0
- rag_wright/contracts/jurisdiction.py +96 -0
- rag_wright/contracts/ontology.py +142 -0
- rag_wright/contracts/property.py +201 -0
- rag_wright/contracts/provenance.py +78 -0
- rag_wright/contracts/query_intent.py +53 -0
- rag_wright/contracts/span.py +76 -0
- rag_wright/contracts/value_match.py +84 -0
- rag_wright/corpus/__init__.py +0 -0
- rag_wright/corpus/canonicalize.py +116 -0
- rag_wright/corpus/cuad.py +153 -0
- rag_wright/corpus/cuad_ingestion.py +72 -0
- rag_wright/corpus/document_parser.py +299 -0
- rag_wright/corpus/edgar.py +231 -0
- rag_wright/corpus/gcs_ingestion.py +120 -0
- rag_wright/corpus/http.py +110 -0
- rag_wright/corpus/selection.py +152 -0
- rag_wright/mcp/__init__.py +11 -0
- rag_wright/mcp/compliance_server.py +299 -0
- rag_wright/mcp/intra_document_qa_server.py +170 -0
- rag_wright/mcp/relational_qa_server.py +171 -0
- rag_wright/mcp/session_store.py +64 -0
- rag_wright/mcp/typed_property_retrieval_server.py +191 -0
- rag_wright/models/__init__.py +8 -0
- rag_wright/models/profiles.py +331 -0
- rag_wright/models/seam.py +497 -0
- rag_wright/models/tag_structured.py +285 -0
- rag_wright/models/tracing.py +179 -0
- rag_wright/models/usage.py +102 -0
- rag_wright/okf/__init__.py +11 -0
- rag_wright/okf/compile.py +292 -0
- rag_wright/okf/document.py +47 -0
- rag_wright/okf/enrich.py +176 -0
- rag_wright/okf/links.py +190 -0
- rag_wright/okf/lint.py +105 -0
- rag_wright/ontology/__init__.py +6 -0
- rag_wright/ontology/_generated_template_meta.py +60 -0
- rag_wright/ontology/_generated_vocab.py +52 -0
- rag_wright/ontology/clause_template.py +964 -0
- rag_wright/ontology/codegen.py +84 -0
- rag_wright/ontology/compliance_bridge.ttl +186 -0
- rag_wright/ontology/contract_bridge.ttl +2685 -0
- rag_wright/ontology/contract_taxonomy.py +24 -0
- rag_wright/ontology/derive.py +58 -0
- rag_wright/ontology/loader.py +435 -0
- rag_wright/ontology/packs/ftc_16cfr255.ttl +29 -0
- rag_wright/ontology/registry.py +87 -0
- rag_wright/ontology/template_introspect.py +100 -0
- rag_wright/py.typed +0 -0
- rag_wright/reference/__init__.py +2 -0
- rag_wright/reference/compliance.py +41 -0
- rag_wright/reference/contract_seam.py +123 -0
- rag_wright/skills/__init__.py +7 -0
- rag_wright/skills/claim_extraction/SKILL.md +47 -0
- rag_wright/skills/claim_extraction/__init__.py +1 -0
- rag_wright/skills/claim_extraction/template.py +50 -0
- rag_wright/skills/compliance_judgment/SKILL.md +59 -0
- rag_wright/skills/corpus_ingest/SKILL.md +106 -0
- rag_wright/skills/extraction_semantic_judge/SKILL.md +51 -0
- rag_wright/skills/extraction_semantic_judge/__init__.py +1 -0
- rag_wright/skills/generation/SKILL.md +64 -0
- rag_wright/skills/generation/__init__.py +1 -0
- rag_wright/skills/generic_compliance_judgment/SKILL.md +58 -0
- rag_wright/skills/okf_navigate/SKILL.md +137 -0
- rag_wright/skills/requirement_extraction/SKILL.md +47 -0
- rag_wright/skills/requirement_extraction/__init__.py +1 -0
- rag_wright/skills/requirement_extraction/template.py +50 -0
- rag_wright/skills/rlm/SKILL.md +186 -0
- rag_wright/skills/rlm/__init__.py +31 -0
- rag_wright/skills/rlm/agent.py +292 -0
- rag_wright/skills/span_relevance_judgment/SKILL.md +67 -0
- rag_wright/skills/vision_to_text/SKILL.md +36 -0
- rag_wright/skills/vision_to_text/__init__.py +1 -0
- rag_wright/spans/__init__.py +1 -0
- rag_wright/spans/boundary.py +78 -0
- rag_wright/spans/clause_function_classifier.py +490 -0
- rag_wright/spans/clause_kg_extractor.py +337 -0
- rag_wright/spans/cuad_labels.py +81 -0
- rag_wright/spans/dim_classifier.py +158 -0
- rag_wright/spans/dim_fleet.json +411 -0
- rag_wright/spans/function_classifier.py +77 -0
- rag_wright/spans/function_families.py +62 -0
- rag_wright/spans/hybrid_classifier.py +103 -0
- rag_wright/spans/legalbert_classifier.py +83 -0
- rag_wright/spans/model_capabilities.py +107 -0
- rag_wright/spans/new_function_labels.py +111 -0
- rag_wright/spans/page_map.py +68 -0
- rag_wright/spans/property_extractor.py +365 -0
- rag_wright/spans/property_grounding.py +182 -0
- rag_wright/spans/reclassify.py +77 -0
- rag_wright/spans/scarce_function_labels.py +105 -0
- rag_wright/spans/segment.py +341 -0
- rag_wright/spans/semantic_judge.py +197 -0
- rag_wright/spans/symbolic_validation.py +131 -0
- rag_wright/spans/tag_clause_extractor.py +182 -0
- rag_wright/store/__init__.py +6 -0
- rag_wright/store/arcadedb.py +1135 -0
- rag_wright/store/chunk_text.py +66 -0
- rag_wright/store/seam.py +213 -0
- rag_wright/subgraphs/__init__.py +0 -0
- rag_wright/subgraphs/async_ingestion.py +204 -0
- rag_wright/subgraphs/compliance_check.py +1042 -0
- rag_wright/subgraphs/compliance_ingestion.py +306 -0
- rag_wright/subgraphs/contract_ingestion_pipeline.py +999 -0
- rag_wright/subgraphs/graph_extraction.py +102 -0
- rag_wright/subgraphs/intra_document_qa.py +328 -0
- rag_wright/subgraphs/observability.py +140 -0
- rag_wright/subgraphs/query_constraint_extraction.py +73 -0
- rag_wright/subgraphs/relational_qa.py +165 -0
- rag_wright/subgraphs/requirement_extraction.py +137 -0
- rag_wright/subgraphs/scaffold.py +65 -0
- rag_wright/subgraphs/semantic_chunking.py +183 -0
- rag_wright/subgraphs/typed_clause_extraction.py +172 -0
- rag_wright/subgraphs/typed_property_retrieval.py +278 -0
- rag_wright/util/__init__.py +1 -0
- rag_wright/util/concurrent.py +153 -0
- rag_wright/util/spacy_model.py +45 -0
- rag_wright-0.1.0.dist-info/METADATA +168 -0
- rag_wright-0.1.0.dist-info/RECORD +184 -0
- rag_wright-0.1.0.dist-info/WHEEL +4 -0
- rag_wright-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
"""The chunk-text sidecar (T40, FR-I.3): the per-chunk full text the retrieval index omits.
|
|
2
|
+
|
|
3
|
+
The index is dense-over-summary by design: `ChunkRecord` holds the summary and the vectors, never the
|
|
4
|
+
raw chunk text (the text is used once to compute the `chunk_id` content hash, then discarded). But
|
|
5
|
+
synthesis (FR-Q.5) extracts over full chunk text, so the text must be persisted somewhere keyed by
|
|
6
|
+
`chunk_id` and rehydrated at query time by `chunk_read` (T38). This is that store: one JSON file per
|
|
7
|
+
source document, mapping the canonical `chunk_id` string to its full text.
|
|
8
|
+
|
|
9
|
+
It is deliberately NOT part of the swappable `Store` seam (ArcadeDB / LanceDB): it holds no vectors and
|
|
10
|
+
answers no query, it only round-trips text by id, so it does not belong behind the retrieval seam. Its
|
|
11
|
+
lifecycle is coupled to the chunk lifecycle: written at chunk write under the same content-hash gate as
|
|
12
|
+
the index upsert (so the two never diverge), and deleted per `source_doc_id` when a document is
|
|
13
|
+
re-chunked (`delete_document`, the T34 seam), so stale text is not orphaned relative to the index.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import hashlib
|
|
19
|
+
import json
|
|
20
|
+
from pathlib import Path
|
|
21
|
+
from typing import Optional
|
|
22
|
+
|
|
23
|
+
from rag_wright.contracts.identifiers import ChunkId
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
class ChunkTextStore:
|
|
27
|
+
"""A per-source-document JSON sidecar mapping `chunk_id` -> full chunk text (T40, FR-I.3)."""
|
|
28
|
+
|
|
29
|
+
def __init__(self, root: Path) -> None:
|
|
30
|
+
self._root = Path(root)
|
|
31
|
+
self._root.mkdir(parents=True, exist_ok=True)
|
|
32
|
+
|
|
33
|
+
def put(self, chunk_id: ChunkId, text: str) -> None:
|
|
34
|
+
"""Persist `text` for `chunk_id`, upserting by id under the document's sidecar file.
|
|
35
|
+
|
|
36
|
+
Integrity check: the `chunk_id` already carries the content hash of its text (the same hash
|
|
37
|
+
`ChunkId.of` computed at chunking, over these exact bytes), so `text` MUST hash to it. This is
|
|
38
|
+
the sidecar's core promise made provable — a mismatch is a silent-wrong-text bug (a loop index
|
|
39
|
+
error, a mismatched map) that would otherwise surface only as synthesis citing a valid-looking
|
|
40
|
+
but wrong `chunk_id`; it is rejected here at the boundary, before it can be persisted.
|
|
41
|
+
"""
|
|
42
|
+
digest = hashlib.sha256(text.encode("utf-8")).hexdigest()
|
|
43
|
+
if digest != chunk_id.content_hash:
|
|
44
|
+
raise ValueError(
|
|
45
|
+
f"chunk-text integrity: text for {chunk_id.value!r} hashes to {digest!r}, "
|
|
46
|
+
f"not the chunk_id's content_hash {chunk_id.content_hash!r}"
|
|
47
|
+
)
|
|
48
|
+
path = self._path(chunk_id.source_doc_id)
|
|
49
|
+
mapping = json.loads(path.read_text(encoding="utf-8")) if path.exists() else {}
|
|
50
|
+
mapping[chunk_id.value] = text
|
|
51
|
+
path.write_text(json.dumps(mapping, ensure_ascii=False), encoding="utf-8")
|
|
52
|
+
|
|
53
|
+
def get(self, chunk_id: str) -> Optional[str]:
|
|
54
|
+
"""The persisted text for `chunk_id` (canonical string form), or None if absent."""
|
|
55
|
+
source_doc_id = chunk_id.rsplit(":", 2)[0] # <source_doc_id>:<chunk_index>:<content_hash>
|
|
56
|
+
path = self._path(source_doc_id)
|
|
57
|
+
if not path.exists():
|
|
58
|
+
return None
|
|
59
|
+
return json.loads(path.read_text(encoding="utf-8")).get(chunk_id)
|
|
60
|
+
|
|
61
|
+
def delete_document(self, source_doc_id: str) -> None:
|
|
62
|
+
"""Remove all persisted text for a document (the T34 delete-and-re-chunk seam). Idempotent."""
|
|
63
|
+
self._path(source_doc_id).unlink(missing_ok=True)
|
|
64
|
+
|
|
65
|
+
def _path(self, source_doc_id: str) -> Path:
|
|
66
|
+
return self._root / f"{source_doc_id}.json"
|
rag_wright/store/seam.py
ADDED
|
@@ -0,0 +1,213 @@
|
|
|
1
|
+
"""The store seam: the swappable interface every store implementation binds (T13, FR-S.5).
|
|
2
|
+
|
|
3
|
+
The store is reached only through this interface so it can be swapped without touching capability
|
|
4
|
+
code: ArcadeDB is the default (one multi-model store for both the hybrid index and the graph,
|
|
5
|
+
FR-S.1), and a separate hybrid vector store (LanceDB) is the eval-gated Phase 1 fallback for the
|
|
6
|
+
retrieval leg (GATE-2). The seam is intentionally semantic, not SQL: it exposes schema readiness and
|
|
7
|
+
introspection, never a query string, so a second implementation can bind it without inheriting
|
|
8
|
+
ArcadeDB's SQL dialect. Write-side and query-side methods are added by the tasks that need them
|
|
9
|
+
(T20, T21, T26); T13 defines only the schema-management surface the foundation needs.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
from dataclasses import dataclass
|
|
15
|
+
from typing import Optional, Protocol, runtime_checkable
|
|
16
|
+
|
|
17
|
+
from rag_wright.contracts.chunk import ChunkRecord, MetadataValue
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class _NotNull:
|
|
21
|
+
"""Sentinel for a `kg_edges` where-value meaning `<field> IS NOT NULL` (vs an equality/membership match)."""
|
|
22
|
+
|
|
23
|
+
def __repr__(self) -> str: # pragma: no cover - debug aid
|
|
24
|
+
return "NOT_NULL"
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
NOT_NULL = _NotNull() # kg_edges where-value: presence test, e.g. edge_where={"dimension": NOT_NULL}
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
@dataclass(frozen=True)
|
|
31
|
+
class GraphNode:
|
|
32
|
+
"""A graph entity node to write (T25). `node_key` is the vertex identity (the resolver's canonical id when
|
|
33
|
+
linked, an `UNLINKED:<key>` surrogate otherwise); `entity_id` is the canonical id, or empty when unlinked."""
|
|
34
|
+
|
|
35
|
+
node_key: str
|
|
36
|
+
entity_id: str
|
|
37
|
+
name: str
|
|
38
|
+
entity_type: str
|
|
39
|
+
confidence: str
|
|
40
|
+
chunk_id: str
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
@dataclass(frozen=True)
|
|
44
|
+
class GraphEdge:
|
|
45
|
+
"""A relationship edge between two entity nodes (by `node_key`), carrying provenance + confidence."""
|
|
46
|
+
|
|
47
|
+
source_key: str
|
|
48
|
+
target_key: str
|
|
49
|
+
relationship_type: str
|
|
50
|
+
confidence: str
|
|
51
|
+
chunk_id: str
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
@dataclass(frozen=True)
|
|
55
|
+
class KgNode:
|
|
56
|
+
"""A typed KG node to upsert (DD-1b, ADR-0117): `type` is the vertex type, `key_field` the identity field to
|
|
57
|
+
upsert on, `props` the fields (including `key_field`) as DOMAIN-NATIVE values. The store encodes each prop by its
|
|
58
|
+
pack-declared storage type -- the caller never serializes to the store's wire format."""
|
|
59
|
+
|
|
60
|
+
type: str
|
|
61
|
+
key_field: str
|
|
62
|
+
props: dict[str, object]
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
@dataclass(frozen=True)
|
|
66
|
+
class KgEdge:
|
|
67
|
+
"""A typed KG edge to create between two nodes identified by (type, key_field, key). `props` are native values
|
|
68
|
+
(edge properties are type-driven: edges declare no storage schema)."""
|
|
69
|
+
|
|
70
|
+
type: str
|
|
71
|
+
from_type: str
|
|
72
|
+
from_key_field: str
|
|
73
|
+
from_key: object
|
|
74
|
+
to_type: str
|
|
75
|
+
to_key_field: str
|
|
76
|
+
to_key: object
|
|
77
|
+
props: dict[str, object]
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
@runtime_checkable
|
|
81
|
+
class Store(Protocol):
|
|
82
|
+
"""A swappable store. The ArcadeDB implementation is the default; a stub proves swappability."""
|
|
83
|
+
|
|
84
|
+
def ensure_schema(self) -> None:
|
|
85
|
+
"""Create the chunk-record and graph-node types and their indexes, idempotently."""
|
|
86
|
+
|
|
87
|
+
def type_names(self) -> set[str]:
|
|
88
|
+
"""The names of the types (tables/classes) present in the store."""
|
|
89
|
+
|
|
90
|
+
def property_names(self, type_name: str) -> set[str]:
|
|
91
|
+
"""The property names declared on `type_name` (empty if the type is absent)."""
|
|
92
|
+
|
|
93
|
+
def index_names(self) -> set[str]:
|
|
94
|
+
"""The names of the indexes present in the store."""
|
|
95
|
+
|
|
96
|
+
def ping(self) -> bool:
|
|
97
|
+
"""True if the store is reachable."""
|
|
98
|
+
|
|
99
|
+
def close(self) -> None:
|
|
100
|
+
"""Release any resources held by the implementation."""
|
|
101
|
+
|
|
102
|
+
# --- write-side (T20): the chunk-record write leg. Semantic, not SQL: the seam takes the T3
|
|
103
|
+
# ChunkRecord and each store maps it to its own representation (ArcadeDB decomposes the sparse
|
|
104
|
+
# vector into two parallel arrays; a LanceDB fallback would store it its own way).
|
|
105
|
+
|
|
106
|
+
def upsert_chunk(self, record: ChunkRecord) -> None:
|
|
107
|
+
"""Write a chunk record, upserting by `chunk_id` (re-write of the same id updates in place)."""
|
|
108
|
+
|
|
109
|
+
def get_chunk(self, chunk_id: str) -> Optional[dict]:
|
|
110
|
+
"""The stored row for `chunk_id` (store-native fields), or None if absent."""
|
|
111
|
+
|
|
112
|
+
def chunk_count(self) -> int:
|
|
113
|
+
"""The number of chunk records in the store."""
|
|
114
|
+
|
|
115
|
+
# --- query-side (T21): the hybrid retrieval leg. Semantic, not SQL: the seam takes the two
|
|
116
|
+
# query vectors and each store fuses them its own way (ArcadeDB by server-side RRF over its
|
|
117
|
+
# dense/sparse indexes; a LanceDB fallback by its own hybrid query), so FR-C.3 is swappable.
|
|
118
|
+
|
|
119
|
+
def hybrid_search(
|
|
120
|
+
self,
|
|
121
|
+
dense_query: list[float],
|
|
122
|
+
sparse_query: dict[int, float],
|
|
123
|
+
*,
|
|
124
|
+
k: int,
|
|
125
|
+
filters: Optional[dict[str, MetadataValue]] = None,
|
|
126
|
+
) -> list[dict]:
|
|
127
|
+
"""Fuse the dense and sparse legs into one ranked candidate list (Reciprocal Rank Fusion),
|
|
128
|
+
honoring equality metadata filters, returning up to `k` rows (each with at least `chunk_id`
|
|
129
|
+
and `source_doc_id`) in ranked order, best first."""
|
|
130
|
+
|
|
131
|
+
# --- graph-write (T25): the knowledge-graph leg. Semantic, not SQL: the seam takes entity nodes
|
|
132
|
+
# and relationship edges and each store writes them its own way (ArcadeDB as vertices/edges in
|
|
133
|
+
# one transaction; a fallback store however it models a graph). Nodes/edges carry chunk_id (FR-I.4).
|
|
134
|
+
|
|
135
|
+
def write_graph(self, nodes: list[GraphNode], edges: list[GraphEdge]) -> None:
|
|
136
|
+
"""Write entity nodes (upsert by `node_key`) and relationship edges between them in ONE
|
|
137
|
+
transaction (FR-S.1: a chunk and its entities land together), connecting each entity to its
|
|
138
|
+
source chunk. Nodes and edges carry `chunk_id` and confidence (FR-I.4)."""
|
|
139
|
+
|
|
140
|
+
def graph_counts(self) -> dict[str, int]:
|
|
141
|
+
"""Counts for introspection/tests: `{'entities': n, 'relationships': m}`."""
|
|
142
|
+
|
|
143
|
+
def entities_by_name(self, name: str) -> list[dict]:
|
|
144
|
+
"""Resolve a party NAME to its graph entities (issue 0030): the first step before
|
|
145
|
+
`graph_neighbors`/`graph_query`, which take a `start_entity_id` (an exact node key) and cannot be
|
|
146
|
+
reached from a name otherwise. Returns `[{entity_id, name, entity_type}]` for every stored entity
|
|
147
|
+
whose name normalizes to the same clustering key as `name`, via the SAME `normalize_entity_name`
|
|
148
|
+
the ingestion side uses to merge 'Acme Corp' / 'Acme Corporation' / 'ACME, Inc.' into one entity.
|
|
149
|
+
Normalization is the engine's rule and is applied HERE, so a caller never re-implements it (a raw
|
|
150
|
+
or an already-normalized name both work; the normalization is idempotent). `entity_id` is exactly
|
|
151
|
+
the node key `graph_neighbors`/`graph_query` take as `start_entity_id`. Empty list on no match."""
|
|
152
|
+
|
|
153
|
+
# --- graph-query (T26): relationship traversal. Semantic, not SQL: returns store-agnostic path
|
|
154
|
+
# rows (target + the entity_ids/chunk_ids/confidences along the path) so the capability can shape
|
|
155
|
+
# the cited evidence. Confidence is SURFACED on every path, not filtered on (FR-C.5/FR-Q.3).
|
|
156
|
+
|
|
157
|
+
def graph_neighbors(
|
|
158
|
+
self, entity_id: str, *, relationship_type: str, max_hops: int, documents: Optional[list[str]] = None
|
|
159
|
+
) -> list[dict]:
|
|
160
|
+
"""Traverse `relationship_type` edges from the start entity up to `max_hops`, returning one row
|
|
161
|
+
per reached entity+path: `target_id`, `target_name`, `path_entity_ids`, `path_chunk_ids`,
|
|
162
|
+
`path_confidences`, `hops`. Every edge is surfaced regardless of confidence (T26 does not gate).
|
|
163
|
+
`documents` (issue 0031): scope the traversal to those source documents -- EVERY edge on a path must
|
|
164
|
+
belong to one of them; `None` = the whole graph, `[]` = scope-to-nothing (no rows)."""
|
|
165
|
+
|
|
166
|
+
# --- generic typed-node read (DD-1a, ADR-0117): the backend-agnostic primitive a domain store extension
|
|
167
|
+
# delegates to, so a domain pack never emits store-native SQL. Semantic, not SQL.
|
|
168
|
+
|
|
169
|
+
def kg_read(
|
|
170
|
+
self,
|
|
171
|
+
node_type: str,
|
|
172
|
+
*,
|
|
173
|
+
where: Optional[dict[str, object]] = None,
|
|
174
|
+
fields: Optional[list[str]] = None,
|
|
175
|
+
distinct: Optional[str] = None,
|
|
176
|
+
order_by: Optional[str] = None,
|
|
177
|
+
limit: Optional[int] = None,
|
|
178
|
+
) -> list[dict]:
|
|
179
|
+
"""Read typed nodes of `node_type`. `where` maps a field to a scalar (equality) or a list (membership);
|
|
180
|
+
a list value that is EMPTY means scope-to-nothing and returns `[]` without a query. `distinct` returns the
|
|
181
|
+
distinct values of one field; `fields=None` returns all fields. Equality/membership clauses are AND-ed."""
|
|
182
|
+
|
|
183
|
+
def kg_edges(
|
|
184
|
+
self,
|
|
185
|
+
from_type: Optional[str] = None,
|
|
186
|
+
*,
|
|
187
|
+
where: Optional[dict[str, object]] = None,
|
|
188
|
+
key_range: Optional[tuple[str, object, object]] = None,
|
|
189
|
+
direction: str = "out",
|
|
190
|
+
edge_type: Optional[str] = None,
|
|
191
|
+
edge_where: Optional[dict[str, object]] = None,
|
|
192
|
+
target_where: Optional[dict[str, object]] = None,
|
|
193
|
+
select: dict[str, str],
|
|
194
|
+
) -> list[dict]:
|
|
195
|
+
"""Generic, backend-agnostic edge TRAVERSAL -- the primitive a domain store extension delegates to so it
|
|
196
|
+
never emits store-native traversal SQL (EP-REF-1a, ADR-0117). Three idioms behind one surface:
|
|
197
|
+
|
|
198
|
+
- **node-start traversal** (give a start selector): from the `from_type` nodes matched by `where`
|
|
199
|
+
(scalar=equality, list=membership, `NOT_NULL`=presence; a field in `key_range=(field, lo, hi)` adds
|
|
200
|
+
`field >= lo AND field < hi`, the contract-scope range), follow `direction="out"` (outgoing edges to
|
|
201
|
+
the TARGET vertex) or `"in"` (incoming edges to the SOURCE vertex). `edge_type=None` means every edge
|
|
202
|
+
in that direction. `edge_where` filters the edge, `target_where` the reached vertex.
|
|
203
|
+
- **edge scan** (give NEITHER `where` NOR `key_range`): scan the `edge_type` table directly, filtered by
|
|
204
|
+
`edge_where` -- for edge properties that are not reachable from a node key (e.g. a span id on the edge).
|
|
205
|
+
|
|
206
|
+
`select` maps each output alias to an expression: `c.<f>` (start node), `e.<f>` / `e.@type` (edge),
|
|
207
|
+
`v.<f>` (far vertex) for a traversal; or a bare edge field / `inV().<f>` / `outV().<f>` for an edge scan.
|
|
208
|
+
A membership value that is an EMPTY list means scope-to-nothing -> `[]` with no query. Returns the rows."""
|
|
209
|
+
|
|
210
|
+
def kg_write(self, nodes: list["KgNode"], edges: "Iterable[KgEdge]" = ()) -> None:
|
|
211
|
+
"""Upsert typed `nodes` (by each node's `key_field`) then create typed `edges` (FROM/TO by node key), ALL in
|
|
212
|
+
ONE transaction, nodes first so endpoints exist. The caller passes DOMAIN-NATIVE values; the store owns all
|
|
213
|
+
wire encoding, driven by each node type's pack-declared property storage type. Empty input is a no-op."""
|
|
File without changes
|
|
@@ -0,0 +1,204 @@
|
|
|
1
|
+
"""PROD-3 (ADR-0050): async, job-based ingestion -- submit returns a job_id immediately, the corpus ingests in
|
|
2
|
+
the background with bounded document parallelism, and progress is POLLED via a status read (never a blocking run
|
|
3
|
+
with a spinner).
|
|
4
|
+
|
|
5
|
+
This module holds the durable job model + store (2a). The async runner (2b) and the submit/status MCP tools (2c)
|
|
6
|
+
build on it. The per-document pipeline stays the existing LangGraph subgraph (with its RetryPolicy +
|
|
7
|
+
Increment-1 dead-letter/partial); this is the thin async envelope around `run_corpus_ingestion`'s work.
|
|
8
|
+
|
|
9
|
+
The JobStore is file-based (one <job_id>.json per job): process-independent, so a status reader in another
|
|
10
|
+
process (an MCP call) sees live progress; durable, so a crashed/restarted runner's job record survives; and it
|
|
11
|
+
needs no KG schema change for the MVP. The full version can move the store to ArcadeDB/Postgres (ADR-0050).
|
|
12
|
+
"""
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import asyncio
|
|
16
|
+
import os
|
|
17
|
+
import threading
|
|
18
|
+
from datetime import datetime, timezone
|
|
19
|
+
from enum import Enum
|
|
20
|
+
from pathlib import Path
|
|
21
|
+
from typing import Any, Callable
|
|
22
|
+
|
|
23
|
+
from pydantic import BaseModel, Field
|
|
24
|
+
|
|
25
|
+
from rag_wright.subgraphs.contract_ingestion_pipeline import ( # ENG-1 shape + 0009 deferred parse
|
|
26
|
+
PendingDocument,
|
|
27
|
+
aparse_pending,
|
|
28
|
+
build_partial_entry,
|
|
29
|
+
)
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def _now() -> str:
|
|
33
|
+
return datetime.now(timezone.utc).isoformat()
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
class JobStatus(str, Enum):
|
|
37
|
+
QUEUED = "queued" # submitted, not yet started
|
|
38
|
+
RUNNING = "running" # the background runner is processing documents
|
|
39
|
+
SUCCEEDED = "succeeded" # all documents accounted for (ingested / partial / dead-lettered), runner finished
|
|
40
|
+
FAILED = "failed" # the runner itself errored irrecoverably (NOT a per-document failure -> that's dead_lettered)
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
class IngestionJob(BaseModel):
|
|
44
|
+
"""A durable ingestion job. `dead_lettered` / `partial` carry the PROD-3 lossless outcomes (a failure is
|
|
45
|
+
KNOWN here at completion, never grep-only). Progress = `documents_done` / `documents_total`."""
|
|
46
|
+
|
|
47
|
+
job_id: str
|
|
48
|
+
corpus_ref: dict # e.g. {"kind": "gcs", "bucket": ..., "prefix": ..., "include": [...]}
|
|
49
|
+
db: str # the target KG database
|
|
50
|
+
status: JobStatus = JobStatus.QUEUED
|
|
51
|
+
documents_total: int = 0
|
|
52
|
+
documents_done: int = 0 # ingested (incl. partial) + dead-lettered + resume-skipped
|
|
53
|
+
ingested: int = 0
|
|
54
|
+
dead_lettered: list[dict] = Field(default_factory=list)
|
|
55
|
+
partial: list[dict] = Field(default_factory=list)
|
|
56
|
+
party_links: int = 0
|
|
57
|
+
created_at: str = Field(default_factory=_now)
|
|
58
|
+
updated_at: str = Field(default_factory=_now)
|
|
59
|
+
error: str | None = None # a runner-level (not per-document) failure
|
|
60
|
+
|
|
61
|
+
@property
|
|
62
|
+
def done(self) -> bool:
|
|
63
|
+
return self.status in (JobStatus.SUCCEEDED, JobStatus.FAILED)
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def _atomic_write(path: Path, text: str) -> None:
|
|
67
|
+
"""Write `text` to `path` atomically: write a temp file, then `os.replace` (an atomic rename on POSIX and
|
|
68
|
+
Windows). A concurrent reader therefore sees either the old complete file or the new complete one -- never a
|
|
69
|
+
truncated/empty file mid-write (the JobStore read-mid-write race, ADR-0057 B2e/B4)."""
|
|
70
|
+
tmp = path.with_name(f"{path.name}.tmp.{os.getpid()}")
|
|
71
|
+
tmp.write_text(text, encoding="utf-8")
|
|
72
|
+
os.replace(tmp, path)
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
class JobStore:
|
|
76
|
+
"""File-based job store: one `<job_id>.json` per job under `jobs_dir`. Read-modify-write `update` is safe for
|
|
77
|
+
a SINGLE writer per job (the runner's orchestrator coroutine updates the record; parallel document workers
|
|
78
|
+
report back to it, they do not write the file) -- so no cross-writer race in the MVP."""
|
|
79
|
+
|
|
80
|
+
def __init__(self, jobs_dir: Any) -> None:
|
|
81
|
+
self._dir = Path(jobs_dir)
|
|
82
|
+
self._dir.mkdir(parents=True, exist_ok=True)
|
|
83
|
+
|
|
84
|
+
def _path(self, job_id: str) -> Path:
|
|
85
|
+
return self._dir / f"{job_id}.json"
|
|
86
|
+
|
|
87
|
+
def create(self, job: IngestionJob) -> IngestionJob:
|
|
88
|
+
_atomic_write(self._path(job.job_id), job.model_dump_json(indent=2))
|
|
89
|
+
return job
|
|
90
|
+
|
|
91
|
+
def get(self, job_id: str) -> IngestionJob | None:
|
|
92
|
+
path = self._path(job_id)
|
|
93
|
+
if not path.exists():
|
|
94
|
+
return None
|
|
95
|
+
text = path.read_text(encoding="utf-8")
|
|
96
|
+
# atomic writes (create) mean a reader never sees a partial file; tolerate an empty read defensively
|
|
97
|
+
# (e.g. a truncated legacy write) as "not ready yet" rather than raising.
|
|
98
|
+
return IngestionJob.model_validate_json(text) if text.strip() else None
|
|
99
|
+
|
|
100
|
+
def update(self, job_id: str, **fields: Any) -> IngestionJob:
|
|
101
|
+
"""Read-modify-write the job's fields, stamping `updated_at`. Raises KeyError if the job is unknown."""
|
|
102
|
+
job = self.get(job_id)
|
|
103
|
+
if job is None:
|
|
104
|
+
raise KeyError(job_id)
|
|
105
|
+
updated = job.model_copy(update={**fields, "updated_at": _now()})
|
|
106
|
+
return self.create(updated)
|
|
107
|
+
|
|
108
|
+
def list_jobs(self) -> list[IngestionJob]:
|
|
109
|
+
return [IngestionJob.model_validate_json(p.read_text(encoding="utf-8")) for p in sorted(self._dir.glob("*.json"))]
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
# --- 2b: the async runner (submit returns immediately; a background thread ingests in parallel) ----------------
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
async def run_job(
|
|
116
|
+
job_id: str,
|
|
117
|
+
documents: list,
|
|
118
|
+
ingest_graph: Any,
|
|
119
|
+
store: JobStore,
|
|
120
|
+
*,
|
|
121
|
+
link_fn: Callable[[], int] = lambda: 0,
|
|
122
|
+
is_done: Callable[[Any], bool] = lambda _doc: False,
|
|
123
|
+
max_concurrency: int = 8,
|
|
124
|
+
) -> IngestionJob:
|
|
125
|
+
"""Ingest a materialized document list in the background with bounded parallelism, updating the job record
|
|
126
|
+
as each document completes (so status polling sees live progress). Per-document dead-letter / partial (the
|
|
127
|
+
Increment-1 lossless outcomes) are accumulated onto the job. A per-document failure NEVER fails the job -- it
|
|
128
|
+
is dead_lettered; only a runner-level error sets status=FAILED. Updates come from THIS single orchestrator
|
|
129
|
+
coroutine (asyncio is single-threaded; `store.update` has no await), so there is no cross-writer file race."""
|
|
130
|
+
try:
|
|
131
|
+
store.update(job_id, status=JobStatus.RUNNING, documents_total=len(documents))
|
|
132
|
+
sem = asyncio.Semaphore(max_concurrency)
|
|
133
|
+
dead_lettered: list[dict] = []
|
|
134
|
+
partial: list[dict] = []
|
|
135
|
+
done = 0
|
|
136
|
+
ingested = 0
|
|
137
|
+
|
|
138
|
+
async def _one(item: Any) -> tuple[str, Any, Any]:
|
|
139
|
+
async with sem: # bound concurrent LLM/DB + PARSE work (rate limits)
|
|
140
|
+
if is_done(item): # RESUME: a prior run already wrote this document (skip before parsing)
|
|
141
|
+
return ("skip", item, None)
|
|
142
|
+
doc = item
|
|
143
|
+
try:
|
|
144
|
+
# 0009-ASYNC-INGEST: parse a deferred doc HERE -- concurrently + deadline-bounded, off the loop
|
|
145
|
+
# (its tiered OCR/VLM escalation is the slowest call) -- so it never blocks the others.
|
|
146
|
+
if isinstance(item, PendingDocument):
|
|
147
|
+
doc = await aparse_pending(item)
|
|
148
|
+
# ASYNC-B2e (ADR-0057): ainvoke runs an async-node graph on the loop (true deadline) and a
|
|
149
|
+
# sync-node graph in LangGraph's threadpool -- so it is safe on any compiled graph.
|
|
150
|
+
out = await ingest_graph.ainvoke({"document": doc}) # per-doc LangGraph graph
|
|
151
|
+
except Exception as exc: # noqa: BLE001 - a per-doc CRASH/parse-timeout must dead-letter THAT doc,
|
|
152
|
+
out = {"dead_letter": { # never fail the whole job
|
|
153
|
+
"source_doc_id": item.source_doc_id, "stage": "invoke",
|
|
154
|
+
"reason": "ingest_crashed", "error": str(exc)[:200]}}
|
|
155
|
+
return ("out", doc, out)
|
|
156
|
+
|
|
157
|
+
for coro in asyncio.as_completed([_one(doc) for doc in documents]):
|
|
158
|
+
kind, doc, out = await coro
|
|
159
|
+
done += 1
|
|
160
|
+
if kind == "out" and out.get("dead_letter"):
|
|
161
|
+
dead_lettered.append(out["dead_letter"])
|
|
162
|
+
else:
|
|
163
|
+
ingested += 1
|
|
164
|
+
ocr_failures = [{"page": pg, "reason": "unreadable scan (OCR + VLM failed)"} # 0009-WIRE2
|
|
165
|
+
for pg in (getattr(doc, "ocr_unreadable_pages", None) or [])]
|
|
166
|
+
entry = build_partial_entry( # ENG-1: same shape as the blocking driver; also surfaces span/ocr losses
|
|
167
|
+
doc.source_doc_id, (out or {}).get("clause_failures"), (out or {}).get("span_failures"),
|
|
168
|
+
ocr_failures)
|
|
169
|
+
if entry is not None:
|
|
170
|
+
partial.append(entry)
|
|
171
|
+
store.update(job_id, documents_done=done, ingested=ingested,
|
|
172
|
+
dead_lettered=dead_lettered, partial=partial)
|
|
173
|
+
|
|
174
|
+
links = link_fn() # KG-7 party linking ONCE, after all documents
|
|
175
|
+
return store.update(job_id, status=JobStatus.SUCCEEDED, party_links=links)
|
|
176
|
+
except Exception as exc: # noqa: BLE001 - a RUNNER-level failure (not a per-document one) -> job FAILED, visible
|
|
177
|
+
return store.update(job_id, status=JobStatus.FAILED, error=str(exc)[:500])
|
|
178
|
+
|
|
179
|
+
|
|
180
|
+
def submit_ingestion(
|
|
181
|
+
adapter: Any,
|
|
182
|
+
ingest_graph: Any,
|
|
183
|
+
store: JobStore,
|
|
184
|
+
*,
|
|
185
|
+
job_id: str,
|
|
186
|
+
db: str,
|
|
187
|
+
corpus_ref: dict,
|
|
188
|
+
link_fn: Callable[[], int] = lambda: 0,
|
|
189
|
+
is_done: Callable[[Any], bool] = lambda _doc: False,
|
|
190
|
+
max_concurrency: int = 8,
|
|
191
|
+
) -> str:
|
|
192
|
+
"""Create a QUEUED job, start the background runner, and return the `job_id` IMMEDIATELY (non-blocking). The
|
|
193
|
+
runner ingests in a daemon thread with its own event loop; poll `store.get(job_id)` (or the status MCP tool)
|
|
194
|
+
for progress. NOTE (MVP): the daemon thread lives with the submitting process -- a persistent service/worker
|
|
195
|
+
(or LangGraph Platform / Pub-Sub, the 'full' version) is what survives process exit; documented in ADR-0050."""
|
|
196
|
+
store.create(IngestionJob(job_id=job_id, corpus_ref=corpus_ref, db=db))
|
|
197
|
+
|
|
198
|
+
def _worker() -> None:
|
|
199
|
+
documents = list(adapter.documents()) # materialize inside the worker (may hit the network)
|
|
200
|
+
asyncio.run(run_job(job_id, documents, ingest_graph, store,
|
|
201
|
+
link_fn=link_fn, is_done=is_done, max_concurrency=max_concurrency))
|
|
202
|
+
|
|
203
|
+
threading.Thread(target=_worker, daemon=True, name=f"ingest-{job_id}").start()
|
|
204
|
+
return job_id
|