rag-wright 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (184) hide show
  1. rag_wright/__init__.py +13 -0
  2. rag_wright/api/__init__.py +33 -0
  3. rag_wright/api/config.py +59 -0
  4. rag_wright/api/discover.py +70 -0
  5. rag_wright/api/documents.py +39 -0
  6. rag_wright/api/ids.py +31 -0
  7. rag_wright/api/invoke.py +99 -0
  8. rag_wright/api/kg.py +61 -0
  9. rag_wright/api/mcp.py +94 -0
  10. rag_wright/api/usage.py +30 -0
  11. rag_wright/api/workspace.py +85 -0
  12. rag_wright/capabilities/__init__.py +8 -0
  13. rag_wright/capabilities/answer_generator.py +427 -0
  14. rag_wright/capabilities/ard.py +286 -0
  15. rag_wright/capabilities/assertion_extraction.py +79 -0
  16. rag_wright/capabilities/chunk_read.py +58 -0
  17. rag_wright/capabilities/chunk_write.py +163 -0
  18. rag_wright/capabilities/claim_extraction.py +153 -0
  19. rag_wright/capabilities/clause_exception_linking.py +117 -0
  20. rag_wright/capabilities/compliance_judgment.py +322 -0
  21. rag_wright/capabilities/compliance_store.py +87 -0
  22. rag_wright/capabilities/contract_kg_serve.py +156 -0
  23. rag_wright/capabilities/contract_kg_store.py +251 -0
  24. rag_wright/capabilities/dg_extraction.py +585 -0
  25. rag_wright/capabilities/disambiguation.py +163 -0
  26. rag_wright/capabilities/document_parse.py +87 -0
  27. rag_wright/capabilities/document_scope.py +49 -0
  28. rag_wright/capabilities/embedding.py +164 -0
  29. rag_wright/capabilities/embedding_profiles.py +43 -0
  30. rag_wright/capabilities/entity_resolution.py +154 -0
  31. rag_wright/capabilities/fusion.py +64 -0
  32. rag_wright/capabilities/graph_extraction.py +243 -0
  33. rag_wright/capabilities/graph_query.py +73 -0
  34. rag_wright/capabilities/graph_storage.py +111 -0
  35. rag_wright/capabilities/highlight_serve.py +142 -0
  36. rag_wright/capabilities/hybrid_search.py +65 -0
  37. rag_wright/capabilities/invoke.py +31 -0
  38. rag_wright/capabilities/jev_decision.py +38 -0
  39. rag_wright/capabilities/manifests.py +872 -0
  40. rag_wright/capabilities/okf_navigate.py +456 -0
  41. rag_wright/capabilities/parsing.py +286 -0
  42. rag_wright/capabilities/property_boosted_retrieval.py +125 -0
  43. rag_wright/capabilities/query_function_classifier.py +94 -0
  44. rag_wright/capabilities/query_understanding.py +109 -0
  45. rag_wright/capabilities/registry.py +262 -0
  46. rag_wright/capabilities/remote_encoders.py +94 -0
  47. rag_wright/capabilities/requirement_extraction.py +247 -0
  48. rag_wright/capabilities/reranking.py +123 -0
  49. rag_wright/capabilities/retrieval_core.py +126 -0
  50. rag_wright/capabilities/rlm_chunking.py +808 -0
  51. rag_wright/capabilities/rlm_synthesis.py +316 -0
  52. rag_wright/capabilities/scan_quality.py +136 -0
  53. rag_wright/capabilities/span_relevance_judgment.py +191 -0
  54. rag_wright/capabilities/vision_to_text.py +85 -0
  55. rag_wright/capabilities/vlm_ocr.py +85 -0
  56. rag_wright/contracts/__init__.py +6 -0
  57. rag_wright/contracts/chunk.py +79 -0
  58. rag_wright/contracts/compliance.py +303 -0
  59. rag_wright/contracts/contract_meta.py +27 -0
  60. rag_wright/contracts/extraction.py +130 -0
  61. rag_wright/contracts/function.py +167 -0
  62. rag_wright/contracts/function_routing.py +91 -0
  63. rag_wright/contracts/highlight.py +74 -0
  64. rag_wright/contracts/identifiers.py +153 -0
  65. rag_wright/contracts/jurisdiction.py +96 -0
  66. rag_wright/contracts/ontology.py +142 -0
  67. rag_wright/contracts/property.py +201 -0
  68. rag_wright/contracts/provenance.py +78 -0
  69. rag_wright/contracts/query_intent.py +53 -0
  70. rag_wright/contracts/span.py +76 -0
  71. rag_wright/contracts/value_match.py +84 -0
  72. rag_wright/corpus/__init__.py +0 -0
  73. rag_wright/corpus/canonicalize.py +116 -0
  74. rag_wright/corpus/cuad.py +153 -0
  75. rag_wright/corpus/cuad_ingestion.py +72 -0
  76. rag_wright/corpus/document_parser.py +299 -0
  77. rag_wright/corpus/edgar.py +231 -0
  78. rag_wright/corpus/gcs_ingestion.py +120 -0
  79. rag_wright/corpus/http.py +110 -0
  80. rag_wright/corpus/selection.py +152 -0
  81. rag_wright/mcp/__init__.py +11 -0
  82. rag_wright/mcp/compliance_server.py +299 -0
  83. rag_wright/mcp/intra_document_qa_server.py +170 -0
  84. rag_wright/mcp/relational_qa_server.py +171 -0
  85. rag_wright/mcp/session_store.py +64 -0
  86. rag_wright/mcp/typed_property_retrieval_server.py +191 -0
  87. rag_wright/models/__init__.py +8 -0
  88. rag_wright/models/profiles.py +331 -0
  89. rag_wright/models/seam.py +497 -0
  90. rag_wright/models/tag_structured.py +285 -0
  91. rag_wright/models/tracing.py +179 -0
  92. rag_wright/models/usage.py +102 -0
  93. rag_wright/okf/__init__.py +11 -0
  94. rag_wright/okf/compile.py +292 -0
  95. rag_wright/okf/document.py +47 -0
  96. rag_wright/okf/enrich.py +176 -0
  97. rag_wright/okf/links.py +190 -0
  98. rag_wright/okf/lint.py +105 -0
  99. rag_wright/ontology/__init__.py +6 -0
  100. rag_wright/ontology/_generated_template_meta.py +60 -0
  101. rag_wright/ontology/_generated_vocab.py +52 -0
  102. rag_wright/ontology/clause_template.py +964 -0
  103. rag_wright/ontology/codegen.py +84 -0
  104. rag_wright/ontology/compliance_bridge.ttl +186 -0
  105. rag_wright/ontology/contract_bridge.ttl +2685 -0
  106. rag_wright/ontology/contract_taxonomy.py +24 -0
  107. rag_wright/ontology/derive.py +58 -0
  108. rag_wright/ontology/loader.py +435 -0
  109. rag_wright/ontology/packs/ftc_16cfr255.ttl +29 -0
  110. rag_wright/ontology/registry.py +87 -0
  111. rag_wright/ontology/template_introspect.py +100 -0
  112. rag_wright/py.typed +0 -0
  113. rag_wright/reference/__init__.py +2 -0
  114. rag_wright/reference/compliance.py +41 -0
  115. rag_wright/reference/contract_seam.py +123 -0
  116. rag_wright/skills/__init__.py +7 -0
  117. rag_wright/skills/claim_extraction/SKILL.md +47 -0
  118. rag_wright/skills/claim_extraction/__init__.py +1 -0
  119. rag_wright/skills/claim_extraction/template.py +50 -0
  120. rag_wright/skills/compliance_judgment/SKILL.md +59 -0
  121. rag_wright/skills/corpus_ingest/SKILL.md +106 -0
  122. rag_wright/skills/extraction_semantic_judge/SKILL.md +51 -0
  123. rag_wright/skills/extraction_semantic_judge/__init__.py +1 -0
  124. rag_wright/skills/generation/SKILL.md +64 -0
  125. rag_wright/skills/generation/__init__.py +1 -0
  126. rag_wright/skills/generic_compliance_judgment/SKILL.md +58 -0
  127. rag_wright/skills/okf_navigate/SKILL.md +137 -0
  128. rag_wright/skills/requirement_extraction/SKILL.md +47 -0
  129. rag_wright/skills/requirement_extraction/__init__.py +1 -0
  130. rag_wright/skills/requirement_extraction/template.py +50 -0
  131. rag_wright/skills/rlm/SKILL.md +186 -0
  132. rag_wright/skills/rlm/__init__.py +31 -0
  133. rag_wright/skills/rlm/agent.py +292 -0
  134. rag_wright/skills/span_relevance_judgment/SKILL.md +67 -0
  135. rag_wright/skills/vision_to_text/SKILL.md +36 -0
  136. rag_wright/skills/vision_to_text/__init__.py +1 -0
  137. rag_wright/spans/__init__.py +1 -0
  138. rag_wright/spans/boundary.py +78 -0
  139. rag_wright/spans/clause_function_classifier.py +490 -0
  140. rag_wright/spans/clause_kg_extractor.py +337 -0
  141. rag_wright/spans/cuad_labels.py +81 -0
  142. rag_wright/spans/dim_classifier.py +158 -0
  143. rag_wright/spans/dim_fleet.json +411 -0
  144. rag_wright/spans/function_classifier.py +77 -0
  145. rag_wright/spans/function_families.py +62 -0
  146. rag_wright/spans/hybrid_classifier.py +103 -0
  147. rag_wright/spans/legalbert_classifier.py +83 -0
  148. rag_wright/spans/model_capabilities.py +107 -0
  149. rag_wright/spans/new_function_labels.py +111 -0
  150. rag_wright/spans/page_map.py +68 -0
  151. rag_wright/spans/property_extractor.py +365 -0
  152. rag_wright/spans/property_grounding.py +182 -0
  153. rag_wright/spans/reclassify.py +77 -0
  154. rag_wright/spans/scarce_function_labels.py +105 -0
  155. rag_wright/spans/segment.py +341 -0
  156. rag_wright/spans/semantic_judge.py +197 -0
  157. rag_wright/spans/symbolic_validation.py +131 -0
  158. rag_wright/spans/tag_clause_extractor.py +182 -0
  159. rag_wright/store/__init__.py +6 -0
  160. rag_wright/store/arcadedb.py +1135 -0
  161. rag_wright/store/chunk_text.py +66 -0
  162. rag_wright/store/seam.py +213 -0
  163. rag_wright/subgraphs/__init__.py +0 -0
  164. rag_wright/subgraphs/async_ingestion.py +204 -0
  165. rag_wright/subgraphs/compliance_check.py +1042 -0
  166. rag_wright/subgraphs/compliance_ingestion.py +306 -0
  167. rag_wright/subgraphs/contract_ingestion_pipeline.py +999 -0
  168. rag_wright/subgraphs/graph_extraction.py +102 -0
  169. rag_wright/subgraphs/intra_document_qa.py +328 -0
  170. rag_wright/subgraphs/observability.py +140 -0
  171. rag_wright/subgraphs/query_constraint_extraction.py +73 -0
  172. rag_wright/subgraphs/relational_qa.py +165 -0
  173. rag_wright/subgraphs/requirement_extraction.py +137 -0
  174. rag_wright/subgraphs/scaffold.py +65 -0
  175. rag_wright/subgraphs/semantic_chunking.py +183 -0
  176. rag_wright/subgraphs/typed_clause_extraction.py +172 -0
  177. rag_wright/subgraphs/typed_property_retrieval.py +278 -0
  178. rag_wright/util/__init__.py +1 -0
  179. rag_wright/util/concurrent.py +153 -0
  180. rag_wright/util/spacy_model.py +45 -0
  181. rag_wright-0.1.0.dist-info/METADATA +168 -0
  182. rag_wright-0.1.0.dist-info/RECORD +184 -0
  183. rag_wright-0.1.0.dist-info/WHEEL +4 -0
  184. rag_wright-0.1.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,286 @@
1
+ """RAG-side mirror of GraphWright's ARD `RegistryEntry` schema (GraphWright ADR-0005).
2
+
3
+ This is a **shared wire-format contract**, mirrored here by **deliberate duplication** (see
4
+ docs/adr/0003). RAG_Wright authors Agentic Resource Discovery (ARD) manifests as JSON that
5
+ GraphWright's `RegistryStore` consumes; the capability half does not import the compiler half, so
6
+ its schema is duplicated rather than imported. A change to GraphWright's ADR-0005 schema is a
7
+ **cross-repo coordination point**: this mirror and GraphWright's `src/graphwright/registry/entry.py`
8
+ must be updated together, or a manifest that validates on one side fails on the other. The
9
+ conformance tests here catch a drift in RAG_Wright's own suite instead of only at GraphWright load.
10
+
11
+ Field names, shapes, and validators mirror GraphWright's `entry.py` verbatim (ARD v0.9, camelCase on
12
+ the wire via `to_camel`, `extra="forbid"`).
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ import os
18
+ from pathlib import Path
19
+ from typing import Literal, Optional
20
+
21
+ from pydantic import BaseModel, ConfigDict, Field, model_validator
22
+ from pydantic.alias_generators import to_camel
23
+
24
+ # The ARD identifier scheme (ARD v0.9 section 4.2.1, https://github.com/ards-project/ard-spec):
25
+ # urn:air:<publisher>:<namespace>:<name>. Our vertical's publisher is the FQDN dreamai.io and the
26
+ # namespace is rag_wright, so every capability we author carries `RAG_URN_PREFIX + <slug>`. The
27
+ # generic schema mirror (ArdEnvelope, below) validates only the `urn:air:` prefix, matching
28
+ # GraphWright's schema which accepts any publisher; the exact-publisher check is applied where we
29
+ # author/write OUR manifests (write_manifest, and the emitter in registry.py).
30
+ URN_PUBLISHER = "dreamai.io"
31
+ URN_NAMESPACE = "rag_wright"
32
+ RAG_URN_PREFIX = f"urn:air:{URN_PUBLISHER}:{URN_NAMESPACE}:"
33
+
34
+ # The six governed kinds (GraphWright FR-5.1). `agent_skill` is loaded by an agent node; the others
35
+ # are callables that declare response bounds (FR-5.2).
36
+ EntryKind = Literal["agent_skill", "mcp_tool", "function", "model", "subgraph", "dagster_asset"]
37
+ CALLABLE_KINDS: frozenset[str] = frozenset(
38
+ {"mcp_tool", "function", "model", "subgraph", "dagster_asset"}
39
+ )
40
+
41
+ # kind -> ARD `type` (an IANA media type). Standard types where one exists, vendor types otherwise.
42
+ MEDIA_TYPE_BY_KIND: dict[str, str] = {
43
+ "agent_skill": "application/ai-skill+md",
44
+ "mcp_tool": "application/mcp-server-card+json",
45
+ "function": "application/vnd.dreamai.graphwright.function+json",
46
+ "model": "application/vnd.dreamai.graphwright.model+json",
47
+ "subgraph": "application/vnd.dreamai.graphwright.subgraph+json",
48
+ "dagster_asset": "application/vnd.dreamai.graphwright.dagster-asset+json",
49
+ }
50
+
51
+
52
+ class _ArdModel(BaseModel):
53
+ """Base for the ARD-shaped models: camelCase on the wire, snake_case in Python."""
54
+
55
+ model_config = ConfigDict(alias_generator=to_camel, populate_by_name=True, extra="forbid")
56
+
57
+
58
+ class Attestation(_ArdModel):
59
+ """An ARD trust attestation: a category, where the document lives, and an optional digest."""
60
+
61
+ type: str
62
+ uri: str
63
+ digest: Optional[str] = None
64
+
65
+
66
+ class TrustManifest(_ArdModel):
67
+ """ARD trust manifest: a verifiable identity plus attestations. Attestations may be empty for a
68
+ new entry (GraphWright's control-level trust filter rejects an under-attested entry later)."""
69
+
70
+ identity: str
71
+ identity_type: str
72
+ attestations: list[Attestation] = Field(default_factory=list)
73
+
74
+
75
+ class ArdEnvelope(_ArdModel):
76
+ """The ARD v0.9 publishable face of a capability. Only ARD fields; no internal governance."""
77
+
78
+ identifier: str # domain-anchored URN: urn:air:<publisher>:<namespace>:<name>
79
+ display_name: str
80
+ type: str # IANA media type; must match MEDIA_TYPE_BY_KIND[kind] (checked on the record)
81
+ representative_queries: list[str] = Field(min_length=2, max_length=5)
82
+ trust_manifest: TrustManifest
83
+ description: Optional[str] = None
84
+ tags: list[str] = Field(default_factory=list)
85
+
86
+ @model_validator(mode="after")
87
+ def _check_identifier_is_urn(self) -> ArdEnvelope:
88
+ if not self.identifier.startswith("urn:air:"):
89
+ raise ValueError(
90
+ "identifier must be a domain-anchored ARD URN "
91
+ f"(urn:air:<publisher>:<namespace>:<name>), got {self.identifier!r}"
92
+ )
93
+ return self
94
+
95
+
96
+ class ResponseBounds(_ArdModel):
97
+ """Internal-only response bounds a callable entry declares (caps tool responses ~25,000 tokens)."""
98
+
99
+ max_tokens: int = 25_000
100
+ supports_pagination: bool = False
101
+ supports_filtering: bool = False
102
+
103
+
104
+ class GovernanceBlock(_ArdModel):
105
+ """Internal-only governance the ARD envelope does not carry."""
106
+
107
+ owner: str
108
+ control_level_min: Literal["high", "moderate", "low"] = "moderate"
109
+
110
+
111
+ class SkillRuntime(_ArdModel):
112
+ """An agent skill's intrinsic runtime requirements, so the compiler can hydrate an interpreter node
113
+ without fabricating anything (GraphWright RegistryEntry; mirrored per ADR-0003). Internal-only,
114
+ never the ARD envelope; `agent_skill` only (validated like `requires`); absent when a skill has no
115
+ special runtime needs. Intrinsic requirements only — the model is deployment config and stays out."""
116
+
117
+ needs_interpreter: bool = False # the skill runs code in the interpreter
118
+ rlm: bool = False # the auditable RLM-pattern marker (FR-1.4, FR-4.10)
119
+ granted_subagents: list[str] = Field(default_factory=list) # sub-agent names it may dispatch to
120
+ # This skill's execution requires the interpreter's code-driven fan-out (task() dispatch) to be
121
+ # triggered. A TYPED flag, NOT the trigger phrasing: the exact word (langchain-quickjs's "workflow")
122
+ # is owned by GraphWright's runtime, which translates this flag into whatever the installed
123
+ # interpreter version expects — so the magic word never enters the wire contract (ADR-0017). Implies
124
+ # `needs_interpreter` (dynamic dispatch is exposed by the interpreter), but kept a distinct field:
125
+ # a future interpreter-using skill might not need dynamic-dispatch triggering.
126
+ requires_dynamic_dispatch: bool = False
127
+
128
+ @model_validator(mode="after")
129
+ def _dispatch_implies_interpreter(self) -> SkillRuntime:
130
+ if self.requires_dynamic_dispatch and not self.needs_interpreter:
131
+ raise ValueError(
132
+ "requires_dynamic_dispatch implies needs_interpreter (task() fan-out is exposed by the "
133
+ "interpreter); set needs_interpreter=True"
134
+ )
135
+ return self
136
+
137
+
138
+ # The nominal type vocabulary GraphWright's lowering checker keys on (GraphWright ADR-0030 section 3). A
139
+ # MIRROR of a shared cross-repo contract, like the RegistryEntry schema above and the canonical-slug set in
140
+ # registry.py: the checker compares a port's type as an OPAQUE NAME string (two ports chain iff their type
141
+ # names are equal, no field-level reasoning), so the producer and the consumer of a chain-compatible shape
142
+ # MUST use the same name, and a typo silently breaks a chain check on GraphWright's side. We validate every
143
+ # declared type name against this set so the drift is caught in our own suite (the same discipline as the ARD
144
+ # schema mirror). Changing this set is a cross-repo coordination point with GraphWright's checker vocabulary.
145
+ NOMINAL_TYPE_VOCABULARY: frozenset[str] = frozenset(
146
+ {
147
+ # --- retrieval -> answer (T43) ---
148
+ "text", # a natural-language string (a query, an answer)
149
+ "chunk_id", # a chunk reference WITHOUT its text (id, plus provenance like source_doc_id)
150
+ "chunk_with_text", # a chunk reference WITH its text attached (only chunk_read produces it)
151
+ "scored_chunk", # a chunk reference carrying a relevance score (reranking's output)
152
+ "graph_answer", # the graph leg's cited answer ({answer?, evidence:[{entity_id, chunk_ids[]}]})
153
+ "cited_extract", # a citation: a chunk reference paired with the cited extract text ({chunk_id, extract})
154
+ # --- ingestion -> graph (T44); the ingestion data shapes the retrieval names don't cover ---
155
+ "document", # a raw source document (a file/scan to parse) — parsing's input
156
+ "parsed_doc", # a handle to the cached structured parse (DoclingDocument) — parsing out -> chunking in
157
+ "chunk", # the ingestion chunk: id + full text + SUMMARY + index (distinct from chunk_with_text,
158
+ # which is the query-side rehydrated id+text; embedding needs the summary this carries)
159
+ "embedding", # a chunk's dense+sparse vector record — embedding's output
160
+ "extraction", # chunk-anchored extracted facts (entity mentions + relationship facts) — graph_extraction out
161
+ "entity_cluster", # canonical mention clusters (human-verifiable proposals) — disambiguation's output
162
+ "resolved_entity", # entities + relationships linked to a canonical id (the knowledge graph) — resolution out
163
+ "image", # a raw image/scan (bytes) — vision_to_text's input
164
+ }
165
+ )
166
+
167
+
168
+ class CapabilityInterface(BaseModel):
169
+ """The governed typed I/O interface GraphWright's lowering checker verifies a realization against
170
+ (GraphWright ADR-0030; our T43). A GraphWright VENDOR EXTENSION, not part of the ARD envelope: it rides
171
+ on the `RegistryEntry` beside the internal governance blocks, so ARD-standard consumers ignore it. It is
172
+ the DATA that flows between orchestration steps as named channels (the query, chunk references, the
173
+ answer) — NOT the callable's config (model, keys, thresholds, top-k), which is deployment config.
174
+
175
+ Nominal typing: the checker compares the SET of input/output type NAMES (opaque strings from
176
+ `NOMINAL_TYPE_VOCABULARY`); port names are for readability only. So `inputs`/`outputs` are flat
177
+ `port -> typeName` maps, and cardinality (list vs scalar) is NOT encoded — a port carrying many
178
+ candidates and one carrying a single value both use the element name (`chunk_id`, never `chunk_id[]`).
179
+
180
+ Plain `BaseModel`, NOT `_ArdModel`: the inner keys stay snake_case (`success_criterion`) even though the
181
+ surrounding manifest is camelCase, because GraphWright's mirror (`TypedInterface`) carries no ARD alias
182
+ and its `extra="forbid"` loader rejects camelCased inner keys. `extra="forbid"` here mirrors that —
183
+ exactly the three keys, nothing else (GraphWright ADR-0030 section 2).
184
+ """
185
+
186
+ model_config = ConfigDict(extra="forbid")
187
+
188
+ inputs: dict[str, str]
189
+ outputs: dict[str, str]
190
+ success_criterion: str
191
+
192
+ @model_validator(mode="after")
193
+ def _check_ports(self) -> CapabilityInterface:
194
+ if not self.success_criterion.strip():
195
+ raise ValueError("success_criterion must be a non-empty one-line purpose")
196
+ for role, ports in (("inputs", self.inputs), ("outputs", self.outputs)):
197
+ for port, type_name in ports.items():
198
+ if not port.strip():
199
+ raise ValueError(f"{role} contains a blank port name")
200
+ if type_name not in NOMINAL_TYPE_VOCABULARY:
201
+ raise ValueError(
202
+ f"{role} port {port!r} has type {type_name!r}, not in the agreed nominal type "
203
+ f"vocabulary {sorted(NOMINAL_TYPE_VOCABULARY)} (GraphWright ADR-0030 section 3); "
204
+ "strip any list sugar like '[]' and use the agreed element name"
205
+ )
206
+ return self
207
+
208
+
209
+ class RegistryEntry(_ArdModel):
210
+ """A governed registry record: the ARD envelope plus internal-only governance and eval fields.
211
+
212
+ `kind` is the internal discriminator; the envelope's ARD `type` media-type is the interoperable
213
+ face, kept consistent with `kind` here.
214
+ """
215
+
216
+ kind: EntryKind
217
+ envelope: ArdEnvelope
218
+ golden_eval_ref: Optional[str] = None
219
+ response_bounds: Optional[ResponseBounds] = None # required for callable kinds
220
+ requires: list[str] = Field(default_factory=list) # closure; agent_skill only
221
+ skill_runtime: Optional[SkillRuntime] = None # intrinsic runtime; agent_skill only (like requires)
222
+ # GraphWright vendor extension (ADR-0030), optional: the governed typed I/O the compiler's lowering
223
+ # checker verifies a realization against. `capability_interface` -> `capabilityInterface` on the wire
224
+ # (to_camel), while the nested block keeps its snake_case keys (CapabilityInterface has no alias).
225
+ capability_interface: Optional[CapabilityInterface] = None
226
+ governance: GovernanceBlock
227
+
228
+ @model_validator(mode="after")
229
+ def _check_record_invariants(self) -> RegistryEntry:
230
+ if self.envelope.type != MEDIA_TYPE_BY_KIND[self.kind]:
231
+ raise ValueError(
232
+ f"envelope type {self.envelope.type!r} does not match kind {self.kind!r} "
233
+ f"(expected {MEDIA_TYPE_BY_KIND[self.kind]!r})"
234
+ )
235
+ if self.kind in CALLABLE_KINDS:
236
+ if self.response_bounds is None:
237
+ raise ValueError(f"callable kind {self.kind!r} requires response_bounds")
238
+ elif self.response_bounds is not None:
239
+ raise ValueError(f"kind {self.kind!r} is not callable and carries no response_bounds")
240
+ if self.requires and self.kind != "agent_skill":
241
+ raise ValueError(f"a requires closure is only valid on an agent_skill, not {self.kind!r}")
242
+ if self.skill_runtime is not None and self.kind != "agent_skill":
243
+ raise ValueError(f"skill_runtime is only valid on an agent_skill, not {self.kind!r}")
244
+ return self
245
+
246
+
247
+ # --- the shared ARD registry root (GraphWright docs/authoring/registry-root.md) ------------------
248
+ #
249
+ # One shared registry root, config-addressed by the ARD_REGISTRY_ROOT env var that GraphWright's
250
+ # RegistryStore also reads. RAG_Wright writes its manifests here; the compiler discovers them from
251
+ # the same directory. We never create a second or project-local root and never hardcode a path.
252
+
253
+ _DEFAULT_REGISTRY_ROOT = Path("~/.air/registry") # `.air` mirrors the urn:air: scheme (registry-root.md)
254
+
255
+
256
+ def registry_root() -> Path:
257
+ """Resolve the shared ARD registry root from `ARD_REGISTRY_ROOT`, creating it if absent.
258
+
259
+ Defaults to `~/.air/registry` when the env var is unset (same default as GraphWright). A leading
260
+ `~` is expanded; no absolute path is baked into source. An empty root is a valid, zero-entry
261
+ catalog, so this always returns a usable directory.
262
+ """
263
+ raw = os.environ.get("ARD_REGISTRY_ROOT", "").strip()
264
+ root = Path(raw).expanduser() if raw else _DEFAULT_REGISTRY_ROOT.expanduser()
265
+ root.mkdir(parents=True, exist_ok=True)
266
+ return root
267
+
268
+
269
+ def write_manifest(entry: RegistryEntry, *, root: Optional[Path] = None) -> Path:
270
+ """Write an authored manifest to `<root>/<slug>.json`, the flat top-level ARD layout.
271
+
272
+ The identifier must be one of our own URNs (`urn:air:dreamai.io:rag_wright:<slug>`) — this is the
273
+ exact-publisher check for what we author, distinct from `ArdEnvelope`'s generic `urn:air:` schema
274
+ mirror. The file is named after the URN's final segment; identity is the `identifier` field
275
+ inside. Serialized ARD-shaped (camelCase on the wire) per ADR-0005.
276
+ """
277
+ identifier = entry.envelope.identifier
278
+ if not identifier.startswith(RAG_URN_PREFIX):
279
+ raise ValueError(
280
+ f"manifest identifier {identifier!r} is not one of ours; expected it to start with "
281
+ f"{RAG_URN_PREFIX!r} (urn:air:dreamai.io:rag_wright:<slug>)"
282
+ )
283
+ slug = identifier.rsplit(":", 1)[-1]
284
+ target = (root if root is not None else registry_root()) / f"{slug}.json"
285
+ target.write_text(entry.model_dump_json(by_alias=True, indent=2), encoding="utf-8")
286
+ return target
@@ -0,0 +1,79 @@
1
+ """SEG-3: the DOMAIN-NEUTRAL verbatim assertion extractor -- subject text -> `CheckableFact[]`.
2
+
3
+ The generic analog of `claim_extraction` (which is the ADVERTISING specialization): the SAME docling-graph
4
+ extraction act with a domain-neutral template (`ExtractedAssertions`, no `claim_type`), plus a deterministic
5
+ adaptation (`to_facts`). One LLM act reads the CHECKABLE ASSERTIONS out of a chunk of subject text, quoted
6
+ VERBATIM so each can be cited faithfully and mapped back to its source element (SEG-4). Used by the subject
7
+ compliance pipeline; `claim_extraction` remains the ad path (typed `Claim`s).
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ from typing import Any
13
+
14
+ from pydantic import BaseModel, ConfigDict, Field
15
+
16
+ from rag_wright.capabilities.dg_extraction import aextract_parties, edge
17
+ from rag_wright.contracts.compliance import CheckableFact
18
+
19
+ __all__ = ["ExtractedAssertion", "ExtractedAssertions", "to_facts", "aassertion_extraction"]
20
+
21
+
22
+ class ExtractedAssertion(BaseModel):
23
+ """One checkable assertion the LLM reads out of a subject document (a docling-graph child entity)."""
24
+
25
+ model_config = ConfigDict(graph_id_fields=["assertion_text"], extra="ignore", populate_by_name=True)
26
+
27
+ assertion_text: str = Field(
28
+ description=("One checkable factual assertion / claim / statement the document makes, quoted VERBATIM "
29
+ "from the source text -- the exact words (one assertion per entry), so it can be cited "
30
+ "faithfully. Do NOT paraphrase, summarize, or merge multiple assertions."))
31
+ actor: str = Field(
32
+ default="",
33
+ description=("DEON-5: the ROLE of the party this assertion involves -- a role word, NOT a person's or "
34
+ "company's name. Choose the general role: advertiser, endorser, expert, manufacturer, "
35
+ "seller, employer, or party. (E.g. 'Dr. Miller recommends ...' -> endorser, not 'Dr. "
36
+ "Miller'.) Lets a rule that binds a role be gated to documents where that role appears. "
37
+ "Empty if no clear actor."))
38
+
39
+
40
+ class ExtractedAssertions(BaseModel):
41
+ """The subject document and the distinct checkable assertions it makes (the docling-graph root entity)."""
42
+
43
+ model_config = ConfigDict(graph_id_fields=["subject"], extra="ignore", populate_by_name=True)
44
+
45
+ subject: str = Field(description="A short label for the subject (a headline phrase or the document topic)")
46
+ assertions: list[ExtractedAssertion] = edge(
47
+ "MAKES_ASSERTION", default_factory=list,
48
+ description="The distinct checkable assertions the document makes (one entry per assertion, verbatim)")
49
+
50
+
51
+ def to_facts(extracted: ExtractedAssertions, *, source_doc: str) -> list[CheckableFact]:
52
+ """DETERMINISTIC (no model): adapt extracted assertions to `CheckableFact`s. Blank assertion skipped;
53
+ `fact_id` = content-hash. Domain-neutral -- no `claim_type` (that is the `claim_extraction` specialization).
54
+ The structural locator (section / ¶ / bullet) is attached later, in SEG-4. DEON-5: an extracted `actor` rides
55
+ as a dimension-agnostic `Constraint("actor", ...)` on the fact's `scope`."""
56
+ from rag_wright.contracts.compliance import Constraint
57
+
58
+ out: list[CheckableFact] = []
59
+ for index, item in enumerate(extracted.assertions):
60
+ text = (item.assertion_text or "").strip()
61
+ if not text:
62
+ continue
63
+ actor = (getattr(item, "actor", "") or "").strip().lower()
64
+ scope = [Constraint(dimension="actor", value=actor)] if actor else []
65
+ out.append(CheckableFact(
66
+ fact_id=CheckableFact.make_id(source_doc, index, text), source_doc=source_doc, assertion_text=text,
67
+ scope=scope))
68
+ return out
69
+
70
+
71
+ async def aassertion_extraction(text: str, *, model: Any, source_doc: str,
72
+ aextract_fn: Any = aextract_parties) -> list[CheckableFact]:
73
+ """Extract the checkable assertions (verbatim) from a chunk of subject text -> `CheckableFact`s: the
74
+ docling-graph verbatim-binding extraction act fills `ExtractedAssertions`, then `to_facts` adapts.
75
+ Returns [] if extraction yields nothing. `aextract_fn` is injected for hermetic tests."""
76
+ extracted = await aextract_fn(text, model, template=ExtractedAssertions, extraction_contract="direct")
77
+ if extracted is None:
78
+ return []
79
+ return to_facts(extracted, source_doc=source_doc)
@@ -0,0 +1,58 @@
1
+ """chunk_read (FR-Q, T38): the governed text-rehydration step between fusion and synthesis.
2
+
3
+ Fusion (FR-Q.4) produces a capped evidence set of `chunk_id`s; synthesis (FR-Q.5) extracts over full
4
+ chunk text. But the retrieval index does not hold the text (it is dense-over-summary), so the `chunk_id`s
5
+ must be rehydrated to their text first. `chunk_read` is that step: it reads the chunk-text sidecar (T40,
6
+ `store/chunk_text.py`) and returns the full text per `chunk_id`, in the requested order.
7
+
8
+ It is a governed, discovered, bound capability, not caller-side plumbing: the compiler binds it as an
9
+ in-process `function` node under the `chunk_read` slug, so rehydration is a real registered capability with
10
+ its own contract, discoverable by representative queries. It drops nothing — a `chunk_id` that cannot be
11
+ rehydrated (an orphan, or an id the sidecar never received) is a pipeline inconsistency, raised loud,
12
+ never a silent evidence drop (the same no-silent-drop discipline as FR-Q.6's no-claim-without-a-citation).
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ from pydantic import BaseModel
18
+
19
+ from rag_wright.store.chunk_text import ChunkTextStore
20
+
21
+
22
+ class ChunkText(BaseModel):
23
+ """One rehydrated chunk: its id, full text (from the T40 sidecar), and its source document.
24
+
25
+ `source_doc_id` is provenance, derived from the `chunk_id` (`<source_doc_id>:<index>:<hash>`), so it
26
+ travels with the text into synthesis without a second store read.
27
+ """
28
+
29
+ chunk_id: str
30
+ text: str
31
+ source_doc_id: str
32
+
33
+
34
+ class ChunkReadResult(BaseModel):
35
+ """The rehydrated evidence for synthesis: full text per `chunk_id`, in the requested order."""
36
+
37
+ chunks: list[ChunkText]
38
+
39
+
40
+ def chunk_read(chunk_ids: list[str], *, text_store: ChunkTextStore) -> ChunkReadResult:
41
+ """Rehydrate `chunk_ids` to their full text via the sidecar, order-preserving, dropping nothing.
42
+
43
+ A `chunk_id` with no persisted text raises `KeyError`: rehydration cannot silently drop evidence, so
44
+ a chunk that retrieval surfaced but the sidecar cannot supply is surfaced as an error, not skipped.
45
+ """
46
+ chunks: list[ChunkText] = []
47
+ for chunk_id in chunk_ids:
48
+ text = text_store.get(chunk_id)
49
+ if text is None:
50
+ raise KeyError(
51
+ f"chunk_read: no persisted text for chunk_id {chunk_id!r} — cannot rehydrate "
52
+ "(orphaned chunk or an id the ingest sidecar never received); evidence is not dropped"
53
+ )
54
+ source_doc_id = chunk_id.rsplit(":", 2)[0] # <source_doc_id>:<chunk_index>:<content_hash>
55
+ chunks.append(ChunkText(chunk_id=chunk_id, text=text, source_doc_id=source_doc_id))
56
+ return ChunkReadResult(chunks=chunks)
57
+
58
+
@@ -0,0 +1,163 @@
1
+ """Chunk write + incremental upsert (FR-I.3, FR-I.5): assemble chunk records and write them, gated.
2
+
3
+ This is the ingestion write leg: it assembles a `ChunkRecord` (T3) from a chunk (T17, summary +
4
+ text) and its embedding (T19, dense + sparse), then upserts it into the store by `chunk_id` (a
5
+ re-write updates in place, never duplicates). Ingestion is incremental, resumable, and idempotent: a
6
+ content-hash gate skips an unchanged document (effectively no work); per-document and per-chunk
7
+ checkpoints let a run resume where it stopped; and a document whose write fails lands in a
8
+ dead-letter queue.
9
+
10
+ Chunk write is a seam-bound pipeline node, not a capability discovered by representative queries, so
11
+ it registers nothing and authors no ARD manifest (SPEC section 5). It talks to the store only through
12
+ the swappable `Store` seam (T13), so it works against ArcadeDB or the LanceDB fallback unchanged.
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ import json
18
+ from dataclasses import dataclass
19
+ from pathlib import Path
20
+ from typing import Literal, Mapping, Optional
21
+
22
+ from rag_wright.capabilities.embedding import ChunkEmbedding
23
+ from rag_wright.capabilities.rlm_chunking import Chunk
24
+ from rag_wright.contracts.chunk import ChunkRecord, MetadataValue
25
+ from rag_wright.contracts.identifiers import ChunkId
26
+ from rag_wright.store.chunk_text import ChunkTextStore
27
+ from rag_wright.store.seam import Store
28
+
29
+ WriteStatus = Literal["written", "skipped", "dead_lettered"]
30
+
31
+
32
+ @dataclass(frozen=True)
33
+ class DocumentWriteResult:
34
+ """The outcome of writing one document's chunks."""
35
+
36
+ source_doc_id: str
37
+ status: WriteStatus
38
+ written_count: int
39
+ error: Optional[str] = None
40
+
41
+
42
+ def to_chunk_record(
43
+ chunk: Chunk,
44
+ embedding: ChunkEmbedding,
45
+ *,
46
+ source_doc_id: str,
47
+ source_metadata: Optional[dict[str, MetadataValue]] = None,
48
+ ) -> ChunkRecord:
49
+ """Assemble a `ChunkRecord` from a chunk and its embedding, checking the ids agree.
50
+
51
+ The `chunk_id` is recomputed deterministically from `(source_doc_id, chunk_index, text)` (T1) and
52
+ must match both the chunk's and the embedding's `chunk_id`. Keywords and entity mentions are empty
53
+ here; they are added by the metadata/extraction stages, not the write.
54
+ """
55
+ chunk_id = ChunkId.of(source_doc_id, chunk.chunk_index, chunk.text)
56
+ if chunk_id.value != chunk.chunk_id or embedding.chunk_id != chunk.chunk_id:
57
+ raise ValueError(
58
+ f"chunk_id mismatch: recomputed {chunk_id.value!r}, chunk {chunk.chunk_id!r}, "
59
+ f"embedding {embedding.chunk_id!r}"
60
+ )
61
+ return ChunkRecord(
62
+ chunk_id=chunk_id,
63
+ summary=chunk.summary,
64
+ dense_vector=embedding.dense_vector,
65
+ sparse_vector=embedding.sparse_vector,
66
+ source_metadata=source_metadata or {},
67
+ )
68
+
69
+
70
+ class ChunkWriter:
71
+ """Writes chunk records to the store, content-hash gated with checkpoints and a dead-letter queue.
72
+
73
+ Checkpoints and the dead-letter queue are files under `checkpoint_dir`; the store holds the chunk
74
+ records (summary + vectors). The full chunk text the index omits is persisted to `text_store`, the
75
+ chunk-text sidecar (T40), under the SAME content-hash gate, so the index and the sidecar are driven
76
+ by one decision and never diverge. On resume, chunks already recorded in a document's checkpoint are
77
+ skipped for both.
78
+ """
79
+
80
+ def __init__(self, store: Store, *, text_store: ChunkTextStore, checkpoint_dir: Path) -> None:
81
+ self._store = store
82
+ self._text_store = text_store
83
+ self._checkpoints = Path(checkpoint_dir) / "checkpoints"
84
+ self._dead_letter = Path(checkpoint_dir) / "dead_letter"
85
+ self._checkpoints.mkdir(parents=True, exist_ok=True)
86
+ self._dead_letter.mkdir(parents=True, exist_ok=True)
87
+
88
+ def write_document(
89
+ self,
90
+ source_doc_id: str,
91
+ content_hash: str,
92
+ records: list[ChunkRecord],
93
+ *,
94
+ texts: Mapping[str, str],
95
+ ) -> DocumentWriteResult:
96
+ """Write a document's chunk records, upserting by `chunk_id`. Gated, resumable, dead-lettered.
97
+
98
+ `texts` maps each record's `chunk_id` to its full chunk text, persisted to the sidecar alongside
99
+ the index upsert. A record without a matching text is rejected before any write, so a chunk can
100
+ never land in the index without its text in the sidecar (the lifecycle-coupling guard).
101
+ """
102
+ missing = [r.chunk_id.value for r in records if r.chunk_id.value not in texts]
103
+ if missing:
104
+ raise ValueError(f"no sidecar text supplied for chunk_ids: {missing}")
105
+
106
+ checkpoint = self._load_checkpoint(source_doc_id)
107
+ same_content = checkpoint is not None and checkpoint["content_hash"] == content_hash
108
+ if same_content and checkpoint["status"] == "complete":
109
+ return DocumentWriteResult(source_doc_id, "skipped", 0) # content-hash gate: no work
110
+
111
+ written = set(checkpoint["written"]) if same_content else set() # resume, or start fresh
112
+ newly = 0
113
+ try:
114
+ for record in records:
115
+ chunk_id = record.chunk_id.value
116
+ if chunk_id in written:
117
+ continue # already written on an earlier run (per-chunk checkpoint)
118
+ self._store.upsert_chunk(record)
119
+ self._text_store.put(record.chunk_id, texts[chunk_id]) # sidecar, same gate as the index
120
+ written.add(chunk_id)
121
+ newly += 1
122
+ self._save_checkpoint(source_doc_id, content_hash, written, "in_progress")
123
+ except Exception as exc: # noqa: BLE001 — a failed document is dead-lettered, not raised
124
+ self._save_checkpoint(source_doc_id, content_hash, written, "in_progress")
125
+ self._write_dead_letter(source_doc_id, content_hash, str(exc))
126
+ return DocumentWriteResult(source_doc_id, "dead_lettered", newly, error=str(exc))
127
+
128
+ self._save_checkpoint(source_doc_id, content_hash, written, "complete")
129
+ self._clear_dead_letter(source_doc_id)
130
+ return DocumentWriteResult(source_doc_id, "written", newly)
131
+
132
+ def dead_letter_ids(self) -> set[str]:
133
+ """The source-doc ids currently in the dead-letter queue."""
134
+ return {path.stem for path in self._dead_letter.glob("*.json")}
135
+
136
+ # --- checkpoint / dead-letter files ---------------------------------------------------------
137
+
138
+ def _checkpoint_path(self, source_doc_id: str) -> Path:
139
+ return self._checkpoints / f"{source_doc_id}.json"
140
+
141
+ def _load_checkpoint(self, source_doc_id: str) -> Optional[dict]:
142
+ path = self._checkpoint_path(source_doc_id)
143
+ return json.loads(path.read_text(encoding="utf-8")) if path.exists() else None
144
+
145
+ def _save_checkpoint(
146
+ self, source_doc_id: str, content_hash: str, written: set[str], status: str
147
+ ) -> None:
148
+ self._checkpoint_path(source_doc_id).write_text(
149
+ json.dumps(
150
+ {"content_hash": content_hash, "written": sorted(written), "status": status},
151
+ ensure_ascii=False,
152
+ ),
153
+ encoding="utf-8",
154
+ )
155
+
156
+ def _write_dead_letter(self, source_doc_id: str, content_hash: str, error: str) -> None:
157
+ (self._dead_letter / f"{source_doc_id}.json").write_text(
158
+ json.dumps({"content_hash": content_hash, "error": error}, ensure_ascii=False),
159
+ encoding="utf-8",
160
+ )
161
+
162
+ def _clear_dead_letter(self, source_doc_id: str) -> None:
163
+ (self._dead_letter / f"{source_doc_id}.json").unlink(missing_ok=True)