rag-wright 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- rag_wright/__init__.py +13 -0
- rag_wright/api/__init__.py +33 -0
- rag_wright/api/config.py +59 -0
- rag_wright/api/discover.py +70 -0
- rag_wright/api/documents.py +39 -0
- rag_wright/api/ids.py +31 -0
- rag_wright/api/invoke.py +99 -0
- rag_wright/api/kg.py +61 -0
- rag_wright/api/mcp.py +94 -0
- rag_wright/api/usage.py +30 -0
- rag_wright/api/workspace.py +85 -0
- rag_wright/capabilities/__init__.py +8 -0
- rag_wright/capabilities/answer_generator.py +427 -0
- rag_wright/capabilities/ard.py +286 -0
- rag_wright/capabilities/assertion_extraction.py +79 -0
- rag_wright/capabilities/chunk_read.py +58 -0
- rag_wright/capabilities/chunk_write.py +163 -0
- rag_wright/capabilities/claim_extraction.py +153 -0
- rag_wright/capabilities/clause_exception_linking.py +117 -0
- rag_wright/capabilities/compliance_judgment.py +322 -0
- rag_wright/capabilities/compliance_store.py +87 -0
- rag_wright/capabilities/contract_kg_serve.py +156 -0
- rag_wright/capabilities/contract_kg_store.py +251 -0
- rag_wright/capabilities/dg_extraction.py +585 -0
- rag_wright/capabilities/disambiguation.py +163 -0
- rag_wright/capabilities/document_parse.py +87 -0
- rag_wright/capabilities/document_scope.py +49 -0
- rag_wright/capabilities/embedding.py +164 -0
- rag_wright/capabilities/embedding_profiles.py +43 -0
- rag_wright/capabilities/entity_resolution.py +154 -0
- rag_wright/capabilities/fusion.py +64 -0
- rag_wright/capabilities/graph_extraction.py +243 -0
- rag_wright/capabilities/graph_query.py +73 -0
- rag_wright/capabilities/graph_storage.py +111 -0
- rag_wright/capabilities/highlight_serve.py +142 -0
- rag_wright/capabilities/hybrid_search.py +65 -0
- rag_wright/capabilities/invoke.py +31 -0
- rag_wright/capabilities/jev_decision.py +38 -0
- rag_wright/capabilities/manifests.py +872 -0
- rag_wright/capabilities/okf_navigate.py +456 -0
- rag_wright/capabilities/parsing.py +286 -0
- rag_wright/capabilities/property_boosted_retrieval.py +125 -0
- rag_wright/capabilities/query_function_classifier.py +94 -0
- rag_wright/capabilities/query_understanding.py +109 -0
- rag_wright/capabilities/registry.py +262 -0
- rag_wright/capabilities/remote_encoders.py +94 -0
- rag_wright/capabilities/requirement_extraction.py +247 -0
- rag_wright/capabilities/reranking.py +123 -0
- rag_wright/capabilities/retrieval_core.py +126 -0
- rag_wright/capabilities/rlm_chunking.py +808 -0
- rag_wright/capabilities/rlm_synthesis.py +316 -0
- rag_wright/capabilities/scan_quality.py +136 -0
- rag_wright/capabilities/span_relevance_judgment.py +191 -0
- rag_wright/capabilities/vision_to_text.py +85 -0
- rag_wright/capabilities/vlm_ocr.py +85 -0
- rag_wright/contracts/__init__.py +6 -0
- rag_wright/contracts/chunk.py +79 -0
- rag_wright/contracts/compliance.py +303 -0
- rag_wright/contracts/contract_meta.py +27 -0
- rag_wright/contracts/extraction.py +130 -0
- rag_wright/contracts/function.py +167 -0
- rag_wright/contracts/function_routing.py +91 -0
- rag_wright/contracts/highlight.py +74 -0
- rag_wright/contracts/identifiers.py +153 -0
- rag_wright/contracts/jurisdiction.py +96 -0
- rag_wright/contracts/ontology.py +142 -0
- rag_wright/contracts/property.py +201 -0
- rag_wright/contracts/provenance.py +78 -0
- rag_wright/contracts/query_intent.py +53 -0
- rag_wright/contracts/span.py +76 -0
- rag_wright/contracts/value_match.py +84 -0
- rag_wright/corpus/__init__.py +0 -0
- rag_wright/corpus/canonicalize.py +116 -0
- rag_wright/corpus/cuad.py +153 -0
- rag_wright/corpus/cuad_ingestion.py +72 -0
- rag_wright/corpus/document_parser.py +299 -0
- rag_wright/corpus/edgar.py +231 -0
- rag_wright/corpus/gcs_ingestion.py +120 -0
- rag_wright/corpus/http.py +110 -0
- rag_wright/corpus/selection.py +152 -0
- rag_wright/mcp/__init__.py +11 -0
- rag_wright/mcp/compliance_server.py +299 -0
- rag_wright/mcp/intra_document_qa_server.py +170 -0
- rag_wright/mcp/relational_qa_server.py +171 -0
- rag_wright/mcp/session_store.py +64 -0
- rag_wright/mcp/typed_property_retrieval_server.py +191 -0
- rag_wright/models/__init__.py +8 -0
- rag_wright/models/profiles.py +331 -0
- rag_wright/models/seam.py +497 -0
- rag_wright/models/tag_structured.py +285 -0
- rag_wright/models/tracing.py +179 -0
- rag_wright/models/usage.py +102 -0
- rag_wright/okf/__init__.py +11 -0
- rag_wright/okf/compile.py +292 -0
- rag_wright/okf/document.py +47 -0
- rag_wright/okf/enrich.py +176 -0
- rag_wright/okf/links.py +190 -0
- rag_wright/okf/lint.py +105 -0
- rag_wright/ontology/__init__.py +6 -0
- rag_wright/ontology/_generated_template_meta.py +60 -0
- rag_wright/ontology/_generated_vocab.py +52 -0
- rag_wright/ontology/clause_template.py +964 -0
- rag_wright/ontology/codegen.py +84 -0
- rag_wright/ontology/compliance_bridge.ttl +186 -0
- rag_wright/ontology/contract_bridge.ttl +2685 -0
- rag_wright/ontology/contract_taxonomy.py +24 -0
- rag_wright/ontology/derive.py +58 -0
- rag_wright/ontology/loader.py +435 -0
- rag_wright/ontology/packs/ftc_16cfr255.ttl +29 -0
- rag_wright/ontology/registry.py +87 -0
- rag_wright/ontology/template_introspect.py +100 -0
- rag_wright/py.typed +0 -0
- rag_wright/reference/__init__.py +2 -0
- rag_wright/reference/compliance.py +41 -0
- rag_wright/reference/contract_seam.py +123 -0
- rag_wright/skills/__init__.py +7 -0
- rag_wright/skills/claim_extraction/SKILL.md +47 -0
- rag_wright/skills/claim_extraction/__init__.py +1 -0
- rag_wright/skills/claim_extraction/template.py +50 -0
- rag_wright/skills/compliance_judgment/SKILL.md +59 -0
- rag_wright/skills/corpus_ingest/SKILL.md +106 -0
- rag_wright/skills/extraction_semantic_judge/SKILL.md +51 -0
- rag_wright/skills/extraction_semantic_judge/__init__.py +1 -0
- rag_wright/skills/generation/SKILL.md +64 -0
- rag_wright/skills/generation/__init__.py +1 -0
- rag_wright/skills/generic_compliance_judgment/SKILL.md +58 -0
- rag_wright/skills/okf_navigate/SKILL.md +137 -0
- rag_wright/skills/requirement_extraction/SKILL.md +47 -0
- rag_wright/skills/requirement_extraction/__init__.py +1 -0
- rag_wright/skills/requirement_extraction/template.py +50 -0
- rag_wright/skills/rlm/SKILL.md +186 -0
- rag_wright/skills/rlm/__init__.py +31 -0
- rag_wright/skills/rlm/agent.py +292 -0
- rag_wright/skills/span_relevance_judgment/SKILL.md +67 -0
- rag_wright/skills/vision_to_text/SKILL.md +36 -0
- rag_wright/skills/vision_to_text/__init__.py +1 -0
- rag_wright/spans/__init__.py +1 -0
- rag_wright/spans/boundary.py +78 -0
- rag_wright/spans/clause_function_classifier.py +490 -0
- rag_wright/spans/clause_kg_extractor.py +337 -0
- rag_wright/spans/cuad_labels.py +81 -0
- rag_wright/spans/dim_classifier.py +158 -0
- rag_wright/spans/dim_fleet.json +411 -0
- rag_wright/spans/function_classifier.py +77 -0
- rag_wright/spans/function_families.py +62 -0
- rag_wright/spans/hybrid_classifier.py +103 -0
- rag_wright/spans/legalbert_classifier.py +83 -0
- rag_wright/spans/model_capabilities.py +107 -0
- rag_wright/spans/new_function_labels.py +111 -0
- rag_wright/spans/page_map.py +68 -0
- rag_wright/spans/property_extractor.py +365 -0
- rag_wright/spans/property_grounding.py +182 -0
- rag_wright/spans/reclassify.py +77 -0
- rag_wright/spans/scarce_function_labels.py +105 -0
- rag_wright/spans/segment.py +341 -0
- rag_wright/spans/semantic_judge.py +197 -0
- rag_wright/spans/symbolic_validation.py +131 -0
- rag_wright/spans/tag_clause_extractor.py +182 -0
- rag_wright/store/__init__.py +6 -0
- rag_wright/store/arcadedb.py +1135 -0
- rag_wright/store/chunk_text.py +66 -0
- rag_wright/store/seam.py +213 -0
- rag_wright/subgraphs/__init__.py +0 -0
- rag_wright/subgraphs/async_ingestion.py +204 -0
- rag_wright/subgraphs/compliance_check.py +1042 -0
- rag_wright/subgraphs/compliance_ingestion.py +306 -0
- rag_wright/subgraphs/contract_ingestion_pipeline.py +999 -0
- rag_wright/subgraphs/graph_extraction.py +102 -0
- rag_wright/subgraphs/intra_document_qa.py +328 -0
- rag_wright/subgraphs/observability.py +140 -0
- rag_wright/subgraphs/query_constraint_extraction.py +73 -0
- rag_wright/subgraphs/relational_qa.py +165 -0
- rag_wright/subgraphs/requirement_extraction.py +137 -0
- rag_wright/subgraphs/scaffold.py +65 -0
- rag_wright/subgraphs/semantic_chunking.py +183 -0
- rag_wright/subgraphs/typed_clause_extraction.py +172 -0
- rag_wright/subgraphs/typed_property_retrieval.py +278 -0
- rag_wright/util/__init__.py +1 -0
- rag_wright/util/concurrent.py +153 -0
- rag_wright/util/spacy_model.py +45 -0
- rag_wright-0.1.0.dist-info/METADATA +168 -0
- rag_wright-0.1.0.dist-info/RECORD +184 -0
- rag_wright-0.1.0.dist-info/WHEEL +4 -0
- rag_wright-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,286 @@
|
|
|
1
|
+
"""RAG-side mirror of GraphWright's ARD `RegistryEntry` schema (GraphWright ADR-0005).
|
|
2
|
+
|
|
3
|
+
This is a **shared wire-format contract**, mirrored here by **deliberate duplication** (see
|
|
4
|
+
docs/adr/0003). RAG_Wright authors Agentic Resource Discovery (ARD) manifests as JSON that
|
|
5
|
+
GraphWright's `RegistryStore` consumes; the capability half does not import the compiler half, so
|
|
6
|
+
its schema is duplicated rather than imported. A change to GraphWright's ADR-0005 schema is a
|
|
7
|
+
**cross-repo coordination point**: this mirror and GraphWright's `src/graphwright/registry/entry.py`
|
|
8
|
+
must be updated together, or a manifest that validates on one side fails on the other. The
|
|
9
|
+
conformance tests here catch a drift in RAG_Wright's own suite instead of only at GraphWright load.
|
|
10
|
+
|
|
11
|
+
Field names, shapes, and validators mirror GraphWright's `entry.py` verbatim (ARD v0.9, camelCase on
|
|
12
|
+
the wire via `to_camel`, `extra="forbid"`).
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import os
|
|
18
|
+
from pathlib import Path
|
|
19
|
+
from typing import Literal, Optional
|
|
20
|
+
|
|
21
|
+
from pydantic import BaseModel, ConfigDict, Field, model_validator
|
|
22
|
+
from pydantic.alias_generators import to_camel
|
|
23
|
+
|
|
24
|
+
# The ARD identifier scheme (ARD v0.9 section 4.2.1, https://github.com/ards-project/ard-spec):
|
|
25
|
+
# urn:air:<publisher>:<namespace>:<name>. Our vertical's publisher is the FQDN dreamai.io and the
|
|
26
|
+
# namespace is rag_wright, so every capability we author carries `RAG_URN_PREFIX + <slug>`. The
|
|
27
|
+
# generic schema mirror (ArdEnvelope, below) validates only the `urn:air:` prefix, matching
|
|
28
|
+
# GraphWright's schema which accepts any publisher; the exact-publisher check is applied where we
|
|
29
|
+
# author/write OUR manifests (write_manifest, and the emitter in registry.py).
|
|
30
|
+
URN_PUBLISHER = "dreamai.io"
|
|
31
|
+
URN_NAMESPACE = "rag_wright"
|
|
32
|
+
RAG_URN_PREFIX = f"urn:air:{URN_PUBLISHER}:{URN_NAMESPACE}:"
|
|
33
|
+
|
|
34
|
+
# The six governed kinds (GraphWright FR-5.1). `agent_skill` is loaded by an agent node; the others
|
|
35
|
+
# are callables that declare response bounds (FR-5.2).
|
|
36
|
+
EntryKind = Literal["agent_skill", "mcp_tool", "function", "model", "subgraph", "dagster_asset"]
|
|
37
|
+
CALLABLE_KINDS: frozenset[str] = frozenset(
|
|
38
|
+
{"mcp_tool", "function", "model", "subgraph", "dagster_asset"}
|
|
39
|
+
)
|
|
40
|
+
|
|
41
|
+
# kind -> ARD `type` (an IANA media type). Standard types where one exists, vendor types otherwise.
|
|
42
|
+
MEDIA_TYPE_BY_KIND: dict[str, str] = {
|
|
43
|
+
"agent_skill": "application/ai-skill+md",
|
|
44
|
+
"mcp_tool": "application/mcp-server-card+json",
|
|
45
|
+
"function": "application/vnd.dreamai.graphwright.function+json",
|
|
46
|
+
"model": "application/vnd.dreamai.graphwright.model+json",
|
|
47
|
+
"subgraph": "application/vnd.dreamai.graphwright.subgraph+json",
|
|
48
|
+
"dagster_asset": "application/vnd.dreamai.graphwright.dagster-asset+json",
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
class _ArdModel(BaseModel):
|
|
53
|
+
"""Base for the ARD-shaped models: camelCase on the wire, snake_case in Python."""
|
|
54
|
+
|
|
55
|
+
model_config = ConfigDict(alias_generator=to_camel, populate_by_name=True, extra="forbid")
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
class Attestation(_ArdModel):
|
|
59
|
+
"""An ARD trust attestation: a category, where the document lives, and an optional digest."""
|
|
60
|
+
|
|
61
|
+
type: str
|
|
62
|
+
uri: str
|
|
63
|
+
digest: Optional[str] = None
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
class TrustManifest(_ArdModel):
|
|
67
|
+
"""ARD trust manifest: a verifiable identity plus attestations. Attestations may be empty for a
|
|
68
|
+
new entry (GraphWright's control-level trust filter rejects an under-attested entry later)."""
|
|
69
|
+
|
|
70
|
+
identity: str
|
|
71
|
+
identity_type: str
|
|
72
|
+
attestations: list[Attestation] = Field(default_factory=list)
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
class ArdEnvelope(_ArdModel):
|
|
76
|
+
"""The ARD v0.9 publishable face of a capability. Only ARD fields; no internal governance."""
|
|
77
|
+
|
|
78
|
+
identifier: str # domain-anchored URN: urn:air:<publisher>:<namespace>:<name>
|
|
79
|
+
display_name: str
|
|
80
|
+
type: str # IANA media type; must match MEDIA_TYPE_BY_KIND[kind] (checked on the record)
|
|
81
|
+
representative_queries: list[str] = Field(min_length=2, max_length=5)
|
|
82
|
+
trust_manifest: TrustManifest
|
|
83
|
+
description: Optional[str] = None
|
|
84
|
+
tags: list[str] = Field(default_factory=list)
|
|
85
|
+
|
|
86
|
+
@model_validator(mode="after")
|
|
87
|
+
def _check_identifier_is_urn(self) -> ArdEnvelope:
|
|
88
|
+
if not self.identifier.startswith("urn:air:"):
|
|
89
|
+
raise ValueError(
|
|
90
|
+
"identifier must be a domain-anchored ARD URN "
|
|
91
|
+
f"(urn:air:<publisher>:<namespace>:<name>), got {self.identifier!r}"
|
|
92
|
+
)
|
|
93
|
+
return self
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
class ResponseBounds(_ArdModel):
|
|
97
|
+
"""Internal-only response bounds a callable entry declares (caps tool responses ~25,000 tokens)."""
|
|
98
|
+
|
|
99
|
+
max_tokens: int = 25_000
|
|
100
|
+
supports_pagination: bool = False
|
|
101
|
+
supports_filtering: bool = False
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
class GovernanceBlock(_ArdModel):
|
|
105
|
+
"""Internal-only governance the ARD envelope does not carry."""
|
|
106
|
+
|
|
107
|
+
owner: str
|
|
108
|
+
control_level_min: Literal["high", "moderate", "low"] = "moderate"
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
class SkillRuntime(_ArdModel):
|
|
112
|
+
"""An agent skill's intrinsic runtime requirements, so the compiler can hydrate an interpreter node
|
|
113
|
+
without fabricating anything (GraphWright RegistryEntry; mirrored per ADR-0003). Internal-only,
|
|
114
|
+
never the ARD envelope; `agent_skill` only (validated like `requires`); absent when a skill has no
|
|
115
|
+
special runtime needs. Intrinsic requirements only — the model is deployment config and stays out."""
|
|
116
|
+
|
|
117
|
+
needs_interpreter: bool = False # the skill runs code in the interpreter
|
|
118
|
+
rlm: bool = False # the auditable RLM-pattern marker (FR-1.4, FR-4.10)
|
|
119
|
+
granted_subagents: list[str] = Field(default_factory=list) # sub-agent names it may dispatch to
|
|
120
|
+
# This skill's execution requires the interpreter's code-driven fan-out (task() dispatch) to be
|
|
121
|
+
# triggered. A TYPED flag, NOT the trigger phrasing: the exact word (langchain-quickjs's "workflow")
|
|
122
|
+
# is owned by GraphWright's runtime, which translates this flag into whatever the installed
|
|
123
|
+
# interpreter version expects — so the magic word never enters the wire contract (ADR-0017). Implies
|
|
124
|
+
# `needs_interpreter` (dynamic dispatch is exposed by the interpreter), but kept a distinct field:
|
|
125
|
+
# a future interpreter-using skill might not need dynamic-dispatch triggering.
|
|
126
|
+
requires_dynamic_dispatch: bool = False
|
|
127
|
+
|
|
128
|
+
@model_validator(mode="after")
|
|
129
|
+
def _dispatch_implies_interpreter(self) -> SkillRuntime:
|
|
130
|
+
if self.requires_dynamic_dispatch and not self.needs_interpreter:
|
|
131
|
+
raise ValueError(
|
|
132
|
+
"requires_dynamic_dispatch implies needs_interpreter (task() fan-out is exposed by the "
|
|
133
|
+
"interpreter); set needs_interpreter=True"
|
|
134
|
+
)
|
|
135
|
+
return self
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
# The nominal type vocabulary GraphWright's lowering checker keys on (GraphWright ADR-0030 section 3). A
|
|
139
|
+
# MIRROR of a shared cross-repo contract, like the RegistryEntry schema above and the canonical-slug set in
|
|
140
|
+
# registry.py: the checker compares a port's type as an OPAQUE NAME string (two ports chain iff their type
|
|
141
|
+
# names are equal, no field-level reasoning), so the producer and the consumer of a chain-compatible shape
|
|
142
|
+
# MUST use the same name, and a typo silently breaks a chain check on GraphWright's side. We validate every
|
|
143
|
+
# declared type name against this set so the drift is caught in our own suite (the same discipline as the ARD
|
|
144
|
+
# schema mirror). Changing this set is a cross-repo coordination point with GraphWright's checker vocabulary.
|
|
145
|
+
NOMINAL_TYPE_VOCABULARY: frozenset[str] = frozenset(
|
|
146
|
+
{
|
|
147
|
+
# --- retrieval -> answer (T43) ---
|
|
148
|
+
"text", # a natural-language string (a query, an answer)
|
|
149
|
+
"chunk_id", # a chunk reference WITHOUT its text (id, plus provenance like source_doc_id)
|
|
150
|
+
"chunk_with_text", # a chunk reference WITH its text attached (only chunk_read produces it)
|
|
151
|
+
"scored_chunk", # a chunk reference carrying a relevance score (reranking's output)
|
|
152
|
+
"graph_answer", # the graph leg's cited answer ({answer?, evidence:[{entity_id, chunk_ids[]}]})
|
|
153
|
+
"cited_extract", # a citation: a chunk reference paired with the cited extract text ({chunk_id, extract})
|
|
154
|
+
# --- ingestion -> graph (T44); the ingestion data shapes the retrieval names don't cover ---
|
|
155
|
+
"document", # a raw source document (a file/scan to parse) — parsing's input
|
|
156
|
+
"parsed_doc", # a handle to the cached structured parse (DoclingDocument) — parsing out -> chunking in
|
|
157
|
+
"chunk", # the ingestion chunk: id + full text + SUMMARY + index (distinct from chunk_with_text,
|
|
158
|
+
# which is the query-side rehydrated id+text; embedding needs the summary this carries)
|
|
159
|
+
"embedding", # a chunk's dense+sparse vector record — embedding's output
|
|
160
|
+
"extraction", # chunk-anchored extracted facts (entity mentions + relationship facts) — graph_extraction out
|
|
161
|
+
"entity_cluster", # canonical mention clusters (human-verifiable proposals) — disambiguation's output
|
|
162
|
+
"resolved_entity", # entities + relationships linked to a canonical id (the knowledge graph) — resolution out
|
|
163
|
+
"image", # a raw image/scan (bytes) — vision_to_text's input
|
|
164
|
+
}
|
|
165
|
+
)
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
class CapabilityInterface(BaseModel):
|
|
169
|
+
"""The governed typed I/O interface GraphWright's lowering checker verifies a realization against
|
|
170
|
+
(GraphWright ADR-0030; our T43). A GraphWright VENDOR EXTENSION, not part of the ARD envelope: it rides
|
|
171
|
+
on the `RegistryEntry` beside the internal governance blocks, so ARD-standard consumers ignore it. It is
|
|
172
|
+
the DATA that flows between orchestration steps as named channels (the query, chunk references, the
|
|
173
|
+
answer) — NOT the callable's config (model, keys, thresholds, top-k), which is deployment config.
|
|
174
|
+
|
|
175
|
+
Nominal typing: the checker compares the SET of input/output type NAMES (opaque strings from
|
|
176
|
+
`NOMINAL_TYPE_VOCABULARY`); port names are for readability only. So `inputs`/`outputs` are flat
|
|
177
|
+
`port -> typeName` maps, and cardinality (list vs scalar) is NOT encoded — a port carrying many
|
|
178
|
+
candidates and one carrying a single value both use the element name (`chunk_id`, never `chunk_id[]`).
|
|
179
|
+
|
|
180
|
+
Plain `BaseModel`, NOT `_ArdModel`: the inner keys stay snake_case (`success_criterion`) even though the
|
|
181
|
+
surrounding manifest is camelCase, because GraphWright's mirror (`TypedInterface`) carries no ARD alias
|
|
182
|
+
and its `extra="forbid"` loader rejects camelCased inner keys. `extra="forbid"` here mirrors that —
|
|
183
|
+
exactly the three keys, nothing else (GraphWright ADR-0030 section 2).
|
|
184
|
+
"""
|
|
185
|
+
|
|
186
|
+
model_config = ConfigDict(extra="forbid")
|
|
187
|
+
|
|
188
|
+
inputs: dict[str, str]
|
|
189
|
+
outputs: dict[str, str]
|
|
190
|
+
success_criterion: str
|
|
191
|
+
|
|
192
|
+
@model_validator(mode="after")
|
|
193
|
+
def _check_ports(self) -> CapabilityInterface:
|
|
194
|
+
if not self.success_criterion.strip():
|
|
195
|
+
raise ValueError("success_criterion must be a non-empty one-line purpose")
|
|
196
|
+
for role, ports in (("inputs", self.inputs), ("outputs", self.outputs)):
|
|
197
|
+
for port, type_name in ports.items():
|
|
198
|
+
if not port.strip():
|
|
199
|
+
raise ValueError(f"{role} contains a blank port name")
|
|
200
|
+
if type_name not in NOMINAL_TYPE_VOCABULARY:
|
|
201
|
+
raise ValueError(
|
|
202
|
+
f"{role} port {port!r} has type {type_name!r}, not in the agreed nominal type "
|
|
203
|
+
f"vocabulary {sorted(NOMINAL_TYPE_VOCABULARY)} (GraphWright ADR-0030 section 3); "
|
|
204
|
+
"strip any list sugar like '[]' and use the agreed element name"
|
|
205
|
+
)
|
|
206
|
+
return self
|
|
207
|
+
|
|
208
|
+
|
|
209
|
+
class RegistryEntry(_ArdModel):
|
|
210
|
+
"""A governed registry record: the ARD envelope plus internal-only governance and eval fields.
|
|
211
|
+
|
|
212
|
+
`kind` is the internal discriminator; the envelope's ARD `type` media-type is the interoperable
|
|
213
|
+
face, kept consistent with `kind` here.
|
|
214
|
+
"""
|
|
215
|
+
|
|
216
|
+
kind: EntryKind
|
|
217
|
+
envelope: ArdEnvelope
|
|
218
|
+
golden_eval_ref: Optional[str] = None
|
|
219
|
+
response_bounds: Optional[ResponseBounds] = None # required for callable kinds
|
|
220
|
+
requires: list[str] = Field(default_factory=list) # closure; agent_skill only
|
|
221
|
+
skill_runtime: Optional[SkillRuntime] = None # intrinsic runtime; agent_skill only (like requires)
|
|
222
|
+
# GraphWright vendor extension (ADR-0030), optional: the governed typed I/O the compiler's lowering
|
|
223
|
+
# checker verifies a realization against. `capability_interface` -> `capabilityInterface` on the wire
|
|
224
|
+
# (to_camel), while the nested block keeps its snake_case keys (CapabilityInterface has no alias).
|
|
225
|
+
capability_interface: Optional[CapabilityInterface] = None
|
|
226
|
+
governance: GovernanceBlock
|
|
227
|
+
|
|
228
|
+
@model_validator(mode="after")
|
|
229
|
+
def _check_record_invariants(self) -> RegistryEntry:
|
|
230
|
+
if self.envelope.type != MEDIA_TYPE_BY_KIND[self.kind]:
|
|
231
|
+
raise ValueError(
|
|
232
|
+
f"envelope type {self.envelope.type!r} does not match kind {self.kind!r} "
|
|
233
|
+
f"(expected {MEDIA_TYPE_BY_KIND[self.kind]!r})"
|
|
234
|
+
)
|
|
235
|
+
if self.kind in CALLABLE_KINDS:
|
|
236
|
+
if self.response_bounds is None:
|
|
237
|
+
raise ValueError(f"callable kind {self.kind!r} requires response_bounds")
|
|
238
|
+
elif self.response_bounds is not None:
|
|
239
|
+
raise ValueError(f"kind {self.kind!r} is not callable and carries no response_bounds")
|
|
240
|
+
if self.requires and self.kind != "agent_skill":
|
|
241
|
+
raise ValueError(f"a requires closure is only valid on an agent_skill, not {self.kind!r}")
|
|
242
|
+
if self.skill_runtime is not None and self.kind != "agent_skill":
|
|
243
|
+
raise ValueError(f"skill_runtime is only valid on an agent_skill, not {self.kind!r}")
|
|
244
|
+
return self
|
|
245
|
+
|
|
246
|
+
|
|
247
|
+
# --- the shared ARD registry root (GraphWright docs/authoring/registry-root.md) ------------------
|
|
248
|
+
#
|
|
249
|
+
# One shared registry root, config-addressed by the ARD_REGISTRY_ROOT env var that GraphWright's
|
|
250
|
+
# RegistryStore also reads. RAG_Wright writes its manifests here; the compiler discovers them from
|
|
251
|
+
# the same directory. We never create a second or project-local root and never hardcode a path.
|
|
252
|
+
|
|
253
|
+
_DEFAULT_REGISTRY_ROOT = Path("~/.air/registry") # `.air` mirrors the urn:air: scheme (registry-root.md)
|
|
254
|
+
|
|
255
|
+
|
|
256
|
+
def registry_root() -> Path:
|
|
257
|
+
"""Resolve the shared ARD registry root from `ARD_REGISTRY_ROOT`, creating it if absent.
|
|
258
|
+
|
|
259
|
+
Defaults to `~/.air/registry` when the env var is unset (same default as GraphWright). A leading
|
|
260
|
+
`~` is expanded; no absolute path is baked into source. An empty root is a valid, zero-entry
|
|
261
|
+
catalog, so this always returns a usable directory.
|
|
262
|
+
"""
|
|
263
|
+
raw = os.environ.get("ARD_REGISTRY_ROOT", "").strip()
|
|
264
|
+
root = Path(raw).expanduser() if raw else _DEFAULT_REGISTRY_ROOT.expanduser()
|
|
265
|
+
root.mkdir(parents=True, exist_ok=True)
|
|
266
|
+
return root
|
|
267
|
+
|
|
268
|
+
|
|
269
|
+
def write_manifest(entry: RegistryEntry, *, root: Optional[Path] = None) -> Path:
|
|
270
|
+
"""Write an authored manifest to `<root>/<slug>.json`, the flat top-level ARD layout.
|
|
271
|
+
|
|
272
|
+
The identifier must be one of our own URNs (`urn:air:dreamai.io:rag_wright:<slug>`) — this is the
|
|
273
|
+
exact-publisher check for what we author, distinct from `ArdEnvelope`'s generic `urn:air:` schema
|
|
274
|
+
mirror. The file is named after the URN's final segment; identity is the `identifier` field
|
|
275
|
+
inside. Serialized ARD-shaped (camelCase on the wire) per ADR-0005.
|
|
276
|
+
"""
|
|
277
|
+
identifier = entry.envelope.identifier
|
|
278
|
+
if not identifier.startswith(RAG_URN_PREFIX):
|
|
279
|
+
raise ValueError(
|
|
280
|
+
f"manifest identifier {identifier!r} is not one of ours; expected it to start with "
|
|
281
|
+
f"{RAG_URN_PREFIX!r} (urn:air:dreamai.io:rag_wright:<slug>)"
|
|
282
|
+
)
|
|
283
|
+
slug = identifier.rsplit(":", 1)[-1]
|
|
284
|
+
target = (root if root is not None else registry_root()) / f"{slug}.json"
|
|
285
|
+
target.write_text(entry.model_dump_json(by_alias=True, indent=2), encoding="utf-8")
|
|
286
|
+
return target
|
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
"""SEG-3: the DOMAIN-NEUTRAL verbatim assertion extractor -- subject text -> `CheckableFact[]`.
|
|
2
|
+
|
|
3
|
+
The generic analog of `claim_extraction` (which is the ADVERTISING specialization): the SAME docling-graph
|
|
4
|
+
extraction act with a domain-neutral template (`ExtractedAssertions`, no `claim_type`), plus a deterministic
|
|
5
|
+
adaptation (`to_facts`). One LLM act reads the CHECKABLE ASSERTIONS out of a chunk of subject text, quoted
|
|
6
|
+
VERBATIM so each can be cited faithfully and mapped back to its source element (SEG-4). Used by the subject
|
|
7
|
+
compliance pipeline; `claim_extraction` remains the ad path (typed `Claim`s).
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
from typing import Any
|
|
13
|
+
|
|
14
|
+
from pydantic import BaseModel, ConfigDict, Field
|
|
15
|
+
|
|
16
|
+
from rag_wright.capabilities.dg_extraction import aextract_parties, edge
|
|
17
|
+
from rag_wright.contracts.compliance import CheckableFact
|
|
18
|
+
|
|
19
|
+
__all__ = ["ExtractedAssertion", "ExtractedAssertions", "to_facts", "aassertion_extraction"]
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
class ExtractedAssertion(BaseModel):
|
|
23
|
+
"""One checkable assertion the LLM reads out of a subject document (a docling-graph child entity)."""
|
|
24
|
+
|
|
25
|
+
model_config = ConfigDict(graph_id_fields=["assertion_text"], extra="ignore", populate_by_name=True)
|
|
26
|
+
|
|
27
|
+
assertion_text: str = Field(
|
|
28
|
+
description=("One checkable factual assertion / claim / statement the document makes, quoted VERBATIM "
|
|
29
|
+
"from the source text -- the exact words (one assertion per entry), so it can be cited "
|
|
30
|
+
"faithfully. Do NOT paraphrase, summarize, or merge multiple assertions."))
|
|
31
|
+
actor: str = Field(
|
|
32
|
+
default="",
|
|
33
|
+
description=("DEON-5: the ROLE of the party this assertion involves -- a role word, NOT a person's or "
|
|
34
|
+
"company's name. Choose the general role: advertiser, endorser, expert, manufacturer, "
|
|
35
|
+
"seller, employer, or party. (E.g. 'Dr. Miller recommends ...' -> endorser, not 'Dr. "
|
|
36
|
+
"Miller'.) Lets a rule that binds a role be gated to documents where that role appears. "
|
|
37
|
+
"Empty if no clear actor."))
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
class ExtractedAssertions(BaseModel):
|
|
41
|
+
"""The subject document and the distinct checkable assertions it makes (the docling-graph root entity)."""
|
|
42
|
+
|
|
43
|
+
model_config = ConfigDict(graph_id_fields=["subject"], extra="ignore", populate_by_name=True)
|
|
44
|
+
|
|
45
|
+
subject: str = Field(description="A short label for the subject (a headline phrase or the document topic)")
|
|
46
|
+
assertions: list[ExtractedAssertion] = edge(
|
|
47
|
+
"MAKES_ASSERTION", default_factory=list,
|
|
48
|
+
description="The distinct checkable assertions the document makes (one entry per assertion, verbatim)")
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def to_facts(extracted: ExtractedAssertions, *, source_doc: str) -> list[CheckableFact]:
|
|
52
|
+
"""DETERMINISTIC (no model): adapt extracted assertions to `CheckableFact`s. Blank assertion skipped;
|
|
53
|
+
`fact_id` = content-hash. Domain-neutral -- no `claim_type` (that is the `claim_extraction` specialization).
|
|
54
|
+
The structural locator (section / ¶ / bullet) is attached later, in SEG-4. DEON-5: an extracted `actor` rides
|
|
55
|
+
as a dimension-agnostic `Constraint("actor", ...)` on the fact's `scope`."""
|
|
56
|
+
from rag_wright.contracts.compliance import Constraint
|
|
57
|
+
|
|
58
|
+
out: list[CheckableFact] = []
|
|
59
|
+
for index, item in enumerate(extracted.assertions):
|
|
60
|
+
text = (item.assertion_text or "").strip()
|
|
61
|
+
if not text:
|
|
62
|
+
continue
|
|
63
|
+
actor = (getattr(item, "actor", "") or "").strip().lower()
|
|
64
|
+
scope = [Constraint(dimension="actor", value=actor)] if actor else []
|
|
65
|
+
out.append(CheckableFact(
|
|
66
|
+
fact_id=CheckableFact.make_id(source_doc, index, text), source_doc=source_doc, assertion_text=text,
|
|
67
|
+
scope=scope))
|
|
68
|
+
return out
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
async def aassertion_extraction(text: str, *, model: Any, source_doc: str,
|
|
72
|
+
aextract_fn: Any = aextract_parties) -> list[CheckableFact]:
|
|
73
|
+
"""Extract the checkable assertions (verbatim) from a chunk of subject text -> `CheckableFact`s: the
|
|
74
|
+
docling-graph verbatim-binding extraction act fills `ExtractedAssertions`, then `to_facts` adapts.
|
|
75
|
+
Returns [] if extraction yields nothing. `aextract_fn` is injected for hermetic tests."""
|
|
76
|
+
extracted = await aextract_fn(text, model, template=ExtractedAssertions, extraction_contract="direct")
|
|
77
|
+
if extracted is None:
|
|
78
|
+
return []
|
|
79
|
+
return to_facts(extracted, source_doc=source_doc)
|
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
"""chunk_read (FR-Q, T38): the governed text-rehydration step between fusion and synthesis.
|
|
2
|
+
|
|
3
|
+
Fusion (FR-Q.4) produces a capped evidence set of `chunk_id`s; synthesis (FR-Q.5) extracts over full
|
|
4
|
+
chunk text. But the retrieval index does not hold the text (it is dense-over-summary), so the `chunk_id`s
|
|
5
|
+
must be rehydrated to their text first. `chunk_read` is that step: it reads the chunk-text sidecar (T40,
|
|
6
|
+
`store/chunk_text.py`) and returns the full text per `chunk_id`, in the requested order.
|
|
7
|
+
|
|
8
|
+
It is a governed, discovered, bound capability, not caller-side plumbing: the compiler binds it as an
|
|
9
|
+
in-process `function` node under the `chunk_read` slug, so rehydration is a real registered capability with
|
|
10
|
+
its own contract, discoverable by representative queries. It drops nothing — a `chunk_id` that cannot be
|
|
11
|
+
rehydrated (an orphan, or an id the sidecar never received) is a pipeline inconsistency, raised loud,
|
|
12
|
+
never a silent evidence drop (the same no-silent-drop discipline as FR-Q.6's no-claim-without-a-citation).
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
from pydantic import BaseModel
|
|
18
|
+
|
|
19
|
+
from rag_wright.store.chunk_text import ChunkTextStore
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
class ChunkText(BaseModel):
|
|
23
|
+
"""One rehydrated chunk: its id, full text (from the T40 sidecar), and its source document.
|
|
24
|
+
|
|
25
|
+
`source_doc_id` is provenance, derived from the `chunk_id` (`<source_doc_id>:<index>:<hash>`), so it
|
|
26
|
+
travels with the text into synthesis without a second store read.
|
|
27
|
+
"""
|
|
28
|
+
|
|
29
|
+
chunk_id: str
|
|
30
|
+
text: str
|
|
31
|
+
source_doc_id: str
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
class ChunkReadResult(BaseModel):
|
|
35
|
+
"""The rehydrated evidence for synthesis: full text per `chunk_id`, in the requested order."""
|
|
36
|
+
|
|
37
|
+
chunks: list[ChunkText]
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def chunk_read(chunk_ids: list[str], *, text_store: ChunkTextStore) -> ChunkReadResult:
|
|
41
|
+
"""Rehydrate `chunk_ids` to their full text via the sidecar, order-preserving, dropping nothing.
|
|
42
|
+
|
|
43
|
+
A `chunk_id` with no persisted text raises `KeyError`: rehydration cannot silently drop evidence, so
|
|
44
|
+
a chunk that retrieval surfaced but the sidecar cannot supply is surfaced as an error, not skipped.
|
|
45
|
+
"""
|
|
46
|
+
chunks: list[ChunkText] = []
|
|
47
|
+
for chunk_id in chunk_ids:
|
|
48
|
+
text = text_store.get(chunk_id)
|
|
49
|
+
if text is None:
|
|
50
|
+
raise KeyError(
|
|
51
|
+
f"chunk_read: no persisted text for chunk_id {chunk_id!r} — cannot rehydrate "
|
|
52
|
+
"(orphaned chunk or an id the ingest sidecar never received); evidence is not dropped"
|
|
53
|
+
)
|
|
54
|
+
source_doc_id = chunk_id.rsplit(":", 2)[0] # <source_doc_id>:<chunk_index>:<content_hash>
|
|
55
|
+
chunks.append(ChunkText(chunk_id=chunk_id, text=text, source_doc_id=source_doc_id))
|
|
56
|
+
return ChunkReadResult(chunks=chunks)
|
|
57
|
+
|
|
58
|
+
|
|
@@ -0,0 +1,163 @@
|
|
|
1
|
+
"""Chunk write + incremental upsert (FR-I.3, FR-I.5): assemble chunk records and write them, gated.
|
|
2
|
+
|
|
3
|
+
This is the ingestion write leg: it assembles a `ChunkRecord` (T3) from a chunk (T17, summary +
|
|
4
|
+
text) and its embedding (T19, dense + sparse), then upserts it into the store by `chunk_id` (a
|
|
5
|
+
re-write updates in place, never duplicates). Ingestion is incremental, resumable, and idempotent: a
|
|
6
|
+
content-hash gate skips an unchanged document (effectively no work); per-document and per-chunk
|
|
7
|
+
checkpoints let a run resume where it stopped; and a document whose write fails lands in a
|
|
8
|
+
dead-letter queue.
|
|
9
|
+
|
|
10
|
+
Chunk write is a seam-bound pipeline node, not a capability discovered by representative queries, so
|
|
11
|
+
it registers nothing and authors no ARD manifest (SPEC section 5). It talks to the store only through
|
|
12
|
+
the swappable `Store` seam (T13), so it works against ArcadeDB or the LanceDB fallback unchanged.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import json
|
|
18
|
+
from dataclasses import dataclass
|
|
19
|
+
from pathlib import Path
|
|
20
|
+
from typing import Literal, Mapping, Optional
|
|
21
|
+
|
|
22
|
+
from rag_wright.capabilities.embedding import ChunkEmbedding
|
|
23
|
+
from rag_wright.capabilities.rlm_chunking import Chunk
|
|
24
|
+
from rag_wright.contracts.chunk import ChunkRecord, MetadataValue
|
|
25
|
+
from rag_wright.contracts.identifiers import ChunkId
|
|
26
|
+
from rag_wright.store.chunk_text import ChunkTextStore
|
|
27
|
+
from rag_wright.store.seam import Store
|
|
28
|
+
|
|
29
|
+
WriteStatus = Literal["written", "skipped", "dead_lettered"]
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
@dataclass(frozen=True)
|
|
33
|
+
class DocumentWriteResult:
|
|
34
|
+
"""The outcome of writing one document's chunks."""
|
|
35
|
+
|
|
36
|
+
source_doc_id: str
|
|
37
|
+
status: WriteStatus
|
|
38
|
+
written_count: int
|
|
39
|
+
error: Optional[str] = None
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def to_chunk_record(
|
|
43
|
+
chunk: Chunk,
|
|
44
|
+
embedding: ChunkEmbedding,
|
|
45
|
+
*,
|
|
46
|
+
source_doc_id: str,
|
|
47
|
+
source_metadata: Optional[dict[str, MetadataValue]] = None,
|
|
48
|
+
) -> ChunkRecord:
|
|
49
|
+
"""Assemble a `ChunkRecord` from a chunk and its embedding, checking the ids agree.
|
|
50
|
+
|
|
51
|
+
The `chunk_id` is recomputed deterministically from `(source_doc_id, chunk_index, text)` (T1) and
|
|
52
|
+
must match both the chunk's and the embedding's `chunk_id`. Keywords and entity mentions are empty
|
|
53
|
+
here; they are added by the metadata/extraction stages, not the write.
|
|
54
|
+
"""
|
|
55
|
+
chunk_id = ChunkId.of(source_doc_id, chunk.chunk_index, chunk.text)
|
|
56
|
+
if chunk_id.value != chunk.chunk_id or embedding.chunk_id != chunk.chunk_id:
|
|
57
|
+
raise ValueError(
|
|
58
|
+
f"chunk_id mismatch: recomputed {chunk_id.value!r}, chunk {chunk.chunk_id!r}, "
|
|
59
|
+
f"embedding {embedding.chunk_id!r}"
|
|
60
|
+
)
|
|
61
|
+
return ChunkRecord(
|
|
62
|
+
chunk_id=chunk_id,
|
|
63
|
+
summary=chunk.summary,
|
|
64
|
+
dense_vector=embedding.dense_vector,
|
|
65
|
+
sparse_vector=embedding.sparse_vector,
|
|
66
|
+
source_metadata=source_metadata or {},
|
|
67
|
+
)
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
class ChunkWriter:
|
|
71
|
+
"""Writes chunk records to the store, content-hash gated with checkpoints and a dead-letter queue.
|
|
72
|
+
|
|
73
|
+
Checkpoints and the dead-letter queue are files under `checkpoint_dir`; the store holds the chunk
|
|
74
|
+
records (summary + vectors). The full chunk text the index omits is persisted to `text_store`, the
|
|
75
|
+
chunk-text sidecar (T40), under the SAME content-hash gate, so the index and the sidecar are driven
|
|
76
|
+
by one decision and never diverge. On resume, chunks already recorded in a document's checkpoint are
|
|
77
|
+
skipped for both.
|
|
78
|
+
"""
|
|
79
|
+
|
|
80
|
+
def __init__(self, store: Store, *, text_store: ChunkTextStore, checkpoint_dir: Path) -> None:
|
|
81
|
+
self._store = store
|
|
82
|
+
self._text_store = text_store
|
|
83
|
+
self._checkpoints = Path(checkpoint_dir) / "checkpoints"
|
|
84
|
+
self._dead_letter = Path(checkpoint_dir) / "dead_letter"
|
|
85
|
+
self._checkpoints.mkdir(parents=True, exist_ok=True)
|
|
86
|
+
self._dead_letter.mkdir(parents=True, exist_ok=True)
|
|
87
|
+
|
|
88
|
+
def write_document(
|
|
89
|
+
self,
|
|
90
|
+
source_doc_id: str,
|
|
91
|
+
content_hash: str,
|
|
92
|
+
records: list[ChunkRecord],
|
|
93
|
+
*,
|
|
94
|
+
texts: Mapping[str, str],
|
|
95
|
+
) -> DocumentWriteResult:
|
|
96
|
+
"""Write a document's chunk records, upserting by `chunk_id`. Gated, resumable, dead-lettered.
|
|
97
|
+
|
|
98
|
+
`texts` maps each record's `chunk_id` to its full chunk text, persisted to the sidecar alongside
|
|
99
|
+
the index upsert. A record without a matching text is rejected before any write, so a chunk can
|
|
100
|
+
never land in the index without its text in the sidecar (the lifecycle-coupling guard).
|
|
101
|
+
"""
|
|
102
|
+
missing = [r.chunk_id.value for r in records if r.chunk_id.value not in texts]
|
|
103
|
+
if missing:
|
|
104
|
+
raise ValueError(f"no sidecar text supplied for chunk_ids: {missing}")
|
|
105
|
+
|
|
106
|
+
checkpoint = self._load_checkpoint(source_doc_id)
|
|
107
|
+
same_content = checkpoint is not None and checkpoint["content_hash"] == content_hash
|
|
108
|
+
if same_content and checkpoint["status"] == "complete":
|
|
109
|
+
return DocumentWriteResult(source_doc_id, "skipped", 0) # content-hash gate: no work
|
|
110
|
+
|
|
111
|
+
written = set(checkpoint["written"]) if same_content else set() # resume, or start fresh
|
|
112
|
+
newly = 0
|
|
113
|
+
try:
|
|
114
|
+
for record in records:
|
|
115
|
+
chunk_id = record.chunk_id.value
|
|
116
|
+
if chunk_id in written:
|
|
117
|
+
continue # already written on an earlier run (per-chunk checkpoint)
|
|
118
|
+
self._store.upsert_chunk(record)
|
|
119
|
+
self._text_store.put(record.chunk_id, texts[chunk_id]) # sidecar, same gate as the index
|
|
120
|
+
written.add(chunk_id)
|
|
121
|
+
newly += 1
|
|
122
|
+
self._save_checkpoint(source_doc_id, content_hash, written, "in_progress")
|
|
123
|
+
except Exception as exc: # noqa: BLE001 — a failed document is dead-lettered, not raised
|
|
124
|
+
self._save_checkpoint(source_doc_id, content_hash, written, "in_progress")
|
|
125
|
+
self._write_dead_letter(source_doc_id, content_hash, str(exc))
|
|
126
|
+
return DocumentWriteResult(source_doc_id, "dead_lettered", newly, error=str(exc))
|
|
127
|
+
|
|
128
|
+
self._save_checkpoint(source_doc_id, content_hash, written, "complete")
|
|
129
|
+
self._clear_dead_letter(source_doc_id)
|
|
130
|
+
return DocumentWriteResult(source_doc_id, "written", newly)
|
|
131
|
+
|
|
132
|
+
def dead_letter_ids(self) -> set[str]:
|
|
133
|
+
"""The source-doc ids currently in the dead-letter queue."""
|
|
134
|
+
return {path.stem for path in self._dead_letter.glob("*.json")}
|
|
135
|
+
|
|
136
|
+
# --- checkpoint / dead-letter files ---------------------------------------------------------
|
|
137
|
+
|
|
138
|
+
def _checkpoint_path(self, source_doc_id: str) -> Path:
|
|
139
|
+
return self._checkpoints / f"{source_doc_id}.json"
|
|
140
|
+
|
|
141
|
+
def _load_checkpoint(self, source_doc_id: str) -> Optional[dict]:
|
|
142
|
+
path = self._checkpoint_path(source_doc_id)
|
|
143
|
+
return json.loads(path.read_text(encoding="utf-8")) if path.exists() else None
|
|
144
|
+
|
|
145
|
+
def _save_checkpoint(
|
|
146
|
+
self, source_doc_id: str, content_hash: str, written: set[str], status: str
|
|
147
|
+
) -> None:
|
|
148
|
+
self._checkpoint_path(source_doc_id).write_text(
|
|
149
|
+
json.dumps(
|
|
150
|
+
{"content_hash": content_hash, "written": sorted(written), "status": status},
|
|
151
|
+
ensure_ascii=False,
|
|
152
|
+
),
|
|
153
|
+
encoding="utf-8",
|
|
154
|
+
)
|
|
155
|
+
|
|
156
|
+
def _write_dead_letter(self, source_doc_id: str, content_hash: str, error: str) -> None:
|
|
157
|
+
(self._dead_letter / f"{source_doc_id}.json").write_text(
|
|
158
|
+
json.dumps({"content_hash": content_hash, "error": error}, ensure_ascii=False),
|
|
159
|
+
encoding="utf-8",
|
|
160
|
+
)
|
|
161
|
+
|
|
162
|
+
def _clear_dead_letter(self, source_doc_id: str) -> None:
|
|
163
|
+
(self._dead_letter / f"{source_doc_id}.json").unlink(missing_ok=True)
|