rag-wright 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (184) hide show
  1. rag_wright/__init__.py +13 -0
  2. rag_wright/api/__init__.py +33 -0
  3. rag_wright/api/config.py +59 -0
  4. rag_wright/api/discover.py +70 -0
  5. rag_wright/api/documents.py +39 -0
  6. rag_wright/api/ids.py +31 -0
  7. rag_wright/api/invoke.py +99 -0
  8. rag_wright/api/kg.py +61 -0
  9. rag_wright/api/mcp.py +94 -0
  10. rag_wright/api/usage.py +30 -0
  11. rag_wright/api/workspace.py +85 -0
  12. rag_wright/capabilities/__init__.py +8 -0
  13. rag_wright/capabilities/answer_generator.py +427 -0
  14. rag_wright/capabilities/ard.py +286 -0
  15. rag_wright/capabilities/assertion_extraction.py +79 -0
  16. rag_wright/capabilities/chunk_read.py +58 -0
  17. rag_wright/capabilities/chunk_write.py +163 -0
  18. rag_wright/capabilities/claim_extraction.py +153 -0
  19. rag_wright/capabilities/clause_exception_linking.py +117 -0
  20. rag_wright/capabilities/compliance_judgment.py +322 -0
  21. rag_wright/capabilities/compliance_store.py +87 -0
  22. rag_wright/capabilities/contract_kg_serve.py +156 -0
  23. rag_wright/capabilities/contract_kg_store.py +251 -0
  24. rag_wright/capabilities/dg_extraction.py +585 -0
  25. rag_wright/capabilities/disambiguation.py +163 -0
  26. rag_wright/capabilities/document_parse.py +87 -0
  27. rag_wright/capabilities/document_scope.py +49 -0
  28. rag_wright/capabilities/embedding.py +164 -0
  29. rag_wright/capabilities/embedding_profiles.py +43 -0
  30. rag_wright/capabilities/entity_resolution.py +154 -0
  31. rag_wright/capabilities/fusion.py +64 -0
  32. rag_wright/capabilities/graph_extraction.py +243 -0
  33. rag_wright/capabilities/graph_query.py +73 -0
  34. rag_wright/capabilities/graph_storage.py +111 -0
  35. rag_wright/capabilities/highlight_serve.py +142 -0
  36. rag_wright/capabilities/hybrid_search.py +65 -0
  37. rag_wright/capabilities/invoke.py +31 -0
  38. rag_wright/capabilities/jev_decision.py +38 -0
  39. rag_wright/capabilities/manifests.py +872 -0
  40. rag_wright/capabilities/okf_navigate.py +456 -0
  41. rag_wright/capabilities/parsing.py +286 -0
  42. rag_wright/capabilities/property_boosted_retrieval.py +125 -0
  43. rag_wright/capabilities/query_function_classifier.py +94 -0
  44. rag_wright/capabilities/query_understanding.py +109 -0
  45. rag_wright/capabilities/registry.py +262 -0
  46. rag_wright/capabilities/remote_encoders.py +94 -0
  47. rag_wright/capabilities/requirement_extraction.py +247 -0
  48. rag_wright/capabilities/reranking.py +123 -0
  49. rag_wright/capabilities/retrieval_core.py +126 -0
  50. rag_wright/capabilities/rlm_chunking.py +808 -0
  51. rag_wright/capabilities/rlm_synthesis.py +316 -0
  52. rag_wright/capabilities/scan_quality.py +136 -0
  53. rag_wright/capabilities/span_relevance_judgment.py +191 -0
  54. rag_wright/capabilities/vision_to_text.py +85 -0
  55. rag_wright/capabilities/vlm_ocr.py +85 -0
  56. rag_wright/contracts/__init__.py +6 -0
  57. rag_wright/contracts/chunk.py +79 -0
  58. rag_wright/contracts/compliance.py +303 -0
  59. rag_wright/contracts/contract_meta.py +27 -0
  60. rag_wright/contracts/extraction.py +130 -0
  61. rag_wright/contracts/function.py +167 -0
  62. rag_wright/contracts/function_routing.py +91 -0
  63. rag_wright/contracts/highlight.py +74 -0
  64. rag_wright/contracts/identifiers.py +153 -0
  65. rag_wright/contracts/jurisdiction.py +96 -0
  66. rag_wright/contracts/ontology.py +142 -0
  67. rag_wright/contracts/property.py +201 -0
  68. rag_wright/contracts/provenance.py +78 -0
  69. rag_wright/contracts/query_intent.py +53 -0
  70. rag_wright/contracts/span.py +76 -0
  71. rag_wright/contracts/value_match.py +84 -0
  72. rag_wright/corpus/__init__.py +0 -0
  73. rag_wright/corpus/canonicalize.py +116 -0
  74. rag_wright/corpus/cuad.py +153 -0
  75. rag_wright/corpus/cuad_ingestion.py +72 -0
  76. rag_wright/corpus/document_parser.py +299 -0
  77. rag_wright/corpus/edgar.py +231 -0
  78. rag_wright/corpus/gcs_ingestion.py +120 -0
  79. rag_wright/corpus/http.py +110 -0
  80. rag_wright/corpus/selection.py +152 -0
  81. rag_wright/mcp/__init__.py +11 -0
  82. rag_wright/mcp/compliance_server.py +299 -0
  83. rag_wright/mcp/intra_document_qa_server.py +170 -0
  84. rag_wright/mcp/relational_qa_server.py +171 -0
  85. rag_wright/mcp/session_store.py +64 -0
  86. rag_wright/mcp/typed_property_retrieval_server.py +191 -0
  87. rag_wright/models/__init__.py +8 -0
  88. rag_wright/models/profiles.py +331 -0
  89. rag_wright/models/seam.py +497 -0
  90. rag_wright/models/tag_structured.py +285 -0
  91. rag_wright/models/tracing.py +179 -0
  92. rag_wright/models/usage.py +102 -0
  93. rag_wright/okf/__init__.py +11 -0
  94. rag_wright/okf/compile.py +292 -0
  95. rag_wright/okf/document.py +47 -0
  96. rag_wright/okf/enrich.py +176 -0
  97. rag_wright/okf/links.py +190 -0
  98. rag_wright/okf/lint.py +105 -0
  99. rag_wright/ontology/__init__.py +6 -0
  100. rag_wright/ontology/_generated_template_meta.py +60 -0
  101. rag_wright/ontology/_generated_vocab.py +52 -0
  102. rag_wright/ontology/clause_template.py +964 -0
  103. rag_wright/ontology/codegen.py +84 -0
  104. rag_wright/ontology/compliance_bridge.ttl +186 -0
  105. rag_wright/ontology/contract_bridge.ttl +2685 -0
  106. rag_wright/ontology/contract_taxonomy.py +24 -0
  107. rag_wright/ontology/derive.py +58 -0
  108. rag_wright/ontology/loader.py +435 -0
  109. rag_wright/ontology/packs/ftc_16cfr255.ttl +29 -0
  110. rag_wright/ontology/registry.py +87 -0
  111. rag_wright/ontology/template_introspect.py +100 -0
  112. rag_wright/py.typed +0 -0
  113. rag_wright/reference/__init__.py +2 -0
  114. rag_wright/reference/compliance.py +41 -0
  115. rag_wright/reference/contract_seam.py +123 -0
  116. rag_wright/skills/__init__.py +7 -0
  117. rag_wright/skills/claim_extraction/SKILL.md +47 -0
  118. rag_wright/skills/claim_extraction/__init__.py +1 -0
  119. rag_wright/skills/claim_extraction/template.py +50 -0
  120. rag_wright/skills/compliance_judgment/SKILL.md +59 -0
  121. rag_wright/skills/corpus_ingest/SKILL.md +106 -0
  122. rag_wright/skills/extraction_semantic_judge/SKILL.md +51 -0
  123. rag_wright/skills/extraction_semantic_judge/__init__.py +1 -0
  124. rag_wright/skills/generation/SKILL.md +64 -0
  125. rag_wright/skills/generation/__init__.py +1 -0
  126. rag_wright/skills/generic_compliance_judgment/SKILL.md +58 -0
  127. rag_wright/skills/okf_navigate/SKILL.md +137 -0
  128. rag_wright/skills/requirement_extraction/SKILL.md +47 -0
  129. rag_wright/skills/requirement_extraction/__init__.py +1 -0
  130. rag_wright/skills/requirement_extraction/template.py +50 -0
  131. rag_wright/skills/rlm/SKILL.md +186 -0
  132. rag_wright/skills/rlm/__init__.py +31 -0
  133. rag_wright/skills/rlm/agent.py +292 -0
  134. rag_wright/skills/span_relevance_judgment/SKILL.md +67 -0
  135. rag_wright/skills/vision_to_text/SKILL.md +36 -0
  136. rag_wright/skills/vision_to_text/__init__.py +1 -0
  137. rag_wright/spans/__init__.py +1 -0
  138. rag_wright/spans/boundary.py +78 -0
  139. rag_wright/spans/clause_function_classifier.py +490 -0
  140. rag_wright/spans/clause_kg_extractor.py +337 -0
  141. rag_wright/spans/cuad_labels.py +81 -0
  142. rag_wright/spans/dim_classifier.py +158 -0
  143. rag_wright/spans/dim_fleet.json +411 -0
  144. rag_wright/spans/function_classifier.py +77 -0
  145. rag_wright/spans/function_families.py +62 -0
  146. rag_wright/spans/hybrid_classifier.py +103 -0
  147. rag_wright/spans/legalbert_classifier.py +83 -0
  148. rag_wright/spans/model_capabilities.py +107 -0
  149. rag_wright/spans/new_function_labels.py +111 -0
  150. rag_wright/spans/page_map.py +68 -0
  151. rag_wright/spans/property_extractor.py +365 -0
  152. rag_wright/spans/property_grounding.py +182 -0
  153. rag_wright/spans/reclassify.py +77 -0
  154. rag_wright/spans/scarce_function_labels.py +105 -0
  155. rag_wright/spans/segment.py +341 -0
  156. rag_wright/spans/semantic_judge.py +197 -0
  157. rag_wright/spans/symbolic_validation.py +131 -0
  158. rag_wright/spans/tag_clause_extractor.py +182 -0
  159. rag_wright/store/__init__.py +6 -0
  160. rag_wright/store/arcadedb.py +1135 -0
  161. rag_wright/store/chunk_text.py +66 -0
  162. rag_wright/store/seam.py +213 -0
  163. rag_wright/subgraphs/__init__.py +0 -0
  164. rag_wright/subgraphs/async_ingestion.py +204 -0
  165. rag_wright/subgraphs/compliance_check.py +1042 -0
  166. rag_wright/subgraphs/compliance_ingestion.py +306 -0
  167. rag_wright/subgraphs/contract_ingestion_pipeline.py +999 -0
  168. rag_wright/subgraphs/graph_extraction.py +102 -0
  169. rag_wright/subgraphs/intra_document_qa.py +328 -0
  170. rag_wright/subgraphs/observability.py +140 -0
  171. rag_wright/subgraphs/query_constraint_extraction.py +73 -0
  172. rag_wright/subgraphs/relational_qa.py +165 -0
  173. rag_wright/subgraphs/requirement_extraction.py +137 -0
  174. rag_wright/subgraphs/scaffold.py +65 -0
  175. rag_wright/subgraphs/semantic_chunking.py +183 -0
  176. rag_wright/subgraphs/typed_clause_extraction.py +172 -0
  177. rag_wright/subgraphs/typed_property_retrieval.py +278 -0
  178. rag_wright/util/__init__.py +1 -0
  179. rag_wright/util/concurrent.py +153 -0
  180. rag_wright/util/spacy_model.py +45 -0
  181. rag_wright-0.1.0.dist-info/METADATA +168 -0
  182. rag_wright-0.1.0.dist-info/RECORD +184 -0
  183. rag_wright-0.1.0.dist-info/WHEEL +4 -0
  184. rag_wright-0.1.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,292 @@
1
+ """RLM machinery (FR-C.10, `rlm_method`): the Deep Agents assembly the RLM method runs on.
2
+
3
+ This is the reusable RLM runtime the RLM chunking (T17) and RLM synthesis (T28) capabilities apply.
4
+ It is authored software, not a build-tool feature: the RLM node is a Deep Agent whose interpreter
5
+ (`CodeInterpreterMiddleware`) holds the working set and runs a recursive `decompose()` *workflow* that
6
+ dispatches sub-agents with `task()` — a **fresh `rlm_decomposer`** at each over-budget internal level
7
+ (per-level fresh context) and an `rlm_slice_worker` at each leaf. Arbitrary depth comes from the
8
+ interpreter re-entering `decompose()`; the interpreter holds the recursion stack.
9
+
10
+ Two named sub-agents, dispatched by name (ADR-0015):
11
+ - `rlm_decomposer` — decides one level's split for the slice it is handed; a fresh agent per dispatch.
12
+ - `rlm_slice_worker` — handles one leaf slice; per-slice tool use and per-slice skills live here.
13
+
14
+ The recursion is driven by the interpreter, **not** by an agent dispatching itself: a self-referential
15
+ sub-agent is not constructible on the pinned `deepagents==0.6.12` (its `SubAgentMiddleware.__init__`
16
+ compiles its roster eagerly, so a config whose roster contains itself recurses at construction). See
17
+ ADR-0015 (Q2, corrected — design B') for the grounded reason and why interpreter-driven recursion is the
18
+ faithful realization, not a workaround.
19
+
20
+ Models resolve through the model-profile seam (T11): the RLM reasoning role (orchestrator + decomposer)
21
+ is a Gemma 4 class model (`ModelRole.GENERAL`); the leaf worker's model is chosen by the applying
22
+ capability (a smaller model for chunking summaries, the structured-reasoning model for synthesis). No
23
+ provider or model flag lives here.
24
+ """
25
+
26
+ from __future__ import annotations
27
+
28
+ import threading
29
+ from collections.abc import Callable, Iterator, Sequence
30
+ from contextlib import contextmanager
31
+ from pathlib import Path
32
+ from typing import Any, Optional, Union
33
+
34
+ from langchain_core.language_models.chat_models import BaseChatModel
35
+ from langchain_core.tools import BaseTool
36
+ from langchain_quickjs import CodeInterpreterMiddleware
37
+ from langgraph.graph.state import CompiledStateGraph
38
+
39
+ from deepagents import create_deep_agent
40
+ from deepagents.middleware.subagents import SubAgent
41
+ from rag_wright.models.profiles import ModelRole, model_for
42
+ from rag_wright.models.seam import build_model
43
+
44
+ # The two named sub-agents (ADR-0015 Q1). Also the exact `grantedSubagents` the manifests declare.
45
+ RLM_DECOMPOSER = "rlm_decomposer"
46
+ RLM_SLICE_WORKER = "rlm_slice_worker"
47
+ GRANTED_SUBAGENTS: tuple[str, str] = (RLM_DECOMPOSER, RLM_SLICE_WORKER)
48
+
49
+ # LOAD-BEARING, do NOT remove as an "uncontended lock in a serial path" (KI-1, ADR-0020). Two QuickJS
50
+ # interpreter runtimes coexisting in one process race on shared Rust state and silently complete without
51
+ # dispatching ~half the time, with zero exceptions. This process-wide semaphore serializes the FULL
52
+ # interpreter-session lifetime (build -> run -> teardown) so no two RLM runtimes are ever alive at once.
53
+ # It is uncontended (zero cost) while ingestion is serial; it makes the safe behaviour the DEFAULT so that
54
+ # adding concurrency later turns a silent-correctness failure into a visible-performance one (slower, not
55
+ # wrong). It lifts only when the upstream coexistence bug is fixed (ADR-0020 exit path). Task T35 designs
56
+ # the real concurrent batch; this is the always-on correctness floor beneath it, not throughput tuning.
57
+ _INTERPRETER_SEMAPHORE = threading.BoundedSemaphore(1)
58
+
59
+
60
+ @contextmanager
61
+ def rlm_interpreter_session(
62
+ *, ptc: Sequence[BaseTool] = (), max_result_chars: Optional[int] = None
63
+ ) -> Iterator[CodeInterpreterMiddleware]:
64
+ """Own the process for exactly one RLM interpreter session (KI-1, ADR-0020).
65
+
66
+ Holds `_INTERPRETER_SEMAPHORE` from before the interpreter runtime is built (the middleware is created
67
+ here; its QuickJS runtime is built lazily on first `eval`, inside the lock) until after it is torn down
68
+ (the registry is closed here, deterministically, not on GC timing) — the whole coexistence window, not
69
+ just the dispatch call. Build the RLM agent with the yielded middleware and run it inside the `with`;
70
+ do parsing/String work outside it. Reentrant call from within one session would deadlock — one session
71
+ per call stack, which the discoverer/extractor honour (they invoke once, then parse outside).
72
+
73
+ `ptc` are Programmatic-Tool-Calling tools exposed **inside** the interpreter as `tools.<camelCase>()`
74
+ and never as top-level tools — this is how the working set is delivered as a JS value that stays out of
75
+ the model's context (`working_set` → `tools.workingSet()`, T36 / GraphWright working-set contract).
76
+ """
77
+ # `max_result_chars` overrides the interpreter's 4,000-char eval-result cap (which truncates a large
78
+ # JSON return mid-string). A capability whose workflow returns a big result (e.g. okf_navigate's shortlist
79
+ # + decision log at a high frontier budget) raises it; RLM leaves it at the default.
80
+ _INTERPRETER_SEMAPHORE.acquire()
81
+ kwargs: dict = {"subagents": True, "ptc": list(ptc) or None}
82
+ if max_result_chars is not None:
83
+ kwargs["max_result_chars"] = max_result_chars
84
+ interpreter = CodeInterpreterMiddleware(**kwargs)
85
+ try:
86
+ yield interpreter
87
+ finally:
88
+ try:
89
+ interpreter._registry.close() # tear the QuickJS runtime down inside the lock (no coexistence)
90
+ finally:
91
+ _INTERPRETER_SEMAPHORE.release()
92
+
93
+ _DECOMPOSER_PROMPT = (
94
+ "You decide how to split ONE working-set slice (a list of items) for a recursive divide-and-conquer. "
95
+ "If the slice is small and focused enough to handle directly, mark it a leaf; otherwise return the "
96
+ "split offsets that partition it into coherent contiguous groups. Reply as JSON: {\"leaf\": true} for "
97
+ "a leaf, or {\"leaf\": false, \"cuts\": [i, j, ...]} where each cut is an ascending index into the "
98
+ "slice at which a new group begins. The cuts partition the slice, so no item is lost. Decide only THIS "
99
+ "level; the interpreter re-dispatches you on each group."
100
+ )
101
+ _SLICE_WORKER_PROMPT = (
102
+ "You handle ONE focused working-set slice end to end. Use the tools and skills you are given as "
103
+ "needed, then return the result for this slice only. You never see the whole working set."
104
+ )
105
+
106
+ # The canonical interpreter-driven recursive workflow (ADR-0015 Q2 corrected, design B'). Reads the
107
+ # working set from the runtime PTC tool `tools.workingSet()` (T36) — a JS value that never enters context —
108
+ # and dispatches sub-agents by name via `task()`. Recursion lives HERE, in the interpreter: a fresh
109
+ # `rlm_decomposer` decides each level's split, `decompose()` re-enters itself on the returned parts
110
+ # (depth capped at _MAX_DEPTH = 3, the interpreter holds the stack), and `rlm_slice_worker` handles each leaf. The RLM
111
+ # skill (SKILL.md) teaches this workflow; the node writes it into the `eval` tool when its request asks
112
+ # for a "workflow" (the trigger GraphWright's applier guarantees, requiresDynamicDispatch). It returns
113
+ # the per-leaf results plus the depths at which splitting occurred, so the descent is inspectable.
114
+ RLM_WORKFLOW_JS = r"""
115
+ // Recursive divide-and-conquer over a working set of items. Read it from the runtime tool (a JS value
116
+ // that stays in the interpreter and never enters the model's context), then fan the work out to
117
+ // sub-agents in code (a "workflow"), never one grinding tool call at a time. The interpreter holds the
118
+ // working set and the recursion stack; the model is only ever called on a focused slice.
119
+ const workingSet = await tools.workingSet(); // [{id, ...}, ...] delivered by the runtime, never in context
120
+ // LOAD-COMPLETENESS ASSERTION (T37): verify you loaded the WHOLE delivered set before anything else. The
121
+ // size comes from the runtime (a scalar it cannot under-read); if the load is short, fail loud rather than
122
+ // silently working over a truncated set — the coverage tail below only guarantees coverage over what you
123
+ // loaded, so an under-read here is a silent evidence drop nothing downstream catches.
124
+ const _delivered = await tools.workingSetSize();
125
+ if (workingSet.length !== _delivered) {
126
+ throw new Error("LOAD UNDER-READ: loaded " + workingSet.length + " of " + _delivered + " delivered items");
127
+ }
128
+ const _MAX_DEPTH = 3; // hard cap on recursion — beyond this, force leaf (no runaway splits)
129
+ const _splitDepths = []; // the depths at which decompose() re-entered itself (proof of descent)
130
+ const _handled = new Set(); // ids of working-set items a leaf worker covered
131
+ async function decompose(items, depth) {
132
+ if (items.length === 0) return []; // empty slice — nothing to dispatch (no-op leaf)
133
+ if (depth >= _MAX_DEPTH) {
134
+ for (const it of items) _handled.add(it.id);
135
+ return [await task({ description: "handle leaf depth " + depth + " over " + items.length + " items: " + JSON.stringify(items), subagentType: "rlm_slice_worker" })];
136
+ }
137
+ const decision = JSON.parse(await task({
138
+ description: "decompose depth " + depth + " over " + items.length + " items",
139
+ subagentType: "rlm_decomposer",
140
+ }));
141
+ if (decision.leaf) {
142
+ for (const it of items) _handled.add(it.id);
143
+ return [await task({ description: "handle leaf depth " + depth + " over " + items.length + " items: " + JSON.stringify(items), subagentType: "rlm_slice_worker" })];
144
+ }
145
+ _splitDepths.push(depth);
146
+ // decision.cuts partition `items` into contiguous groups (no item lost); recurse on every group.
147
+ const bounds = [0, ...decision.cuts, items.length];
148
+ const groups = [];
149
+ for (let i = 0; i < bounds.length - 1; i++) groups.push(items.slice(bounds[i], bounds[i + 1]));
150
+ const handled = await Promise.all(groups.map((g) => decompose(g, depth + 1)));
151
+ return handled.flat();
152
+ }
153
+ const _leaves = await decompose(workingSet, 0);
154
+ // COVERAGE TAIL (structural, in-interpreter): the code holds EVERY item, so it guarantees coverage even
155
+ // if the recursion missed a deep leaf out of context — a silent drop otherwise (T37). Any uncovered item
156
+ // is dispatched now, not dropped. This is code checking coverage, not the model asked to be thorough.
157
+ const _missed = workingSet.filter((it) => !_handled.has(it.id));
158
+ if (_missed.length) {
159
+ for (const it of _missed) _handled.add(it.id);
160
+ _leaves.push(await task({ description: "cover " + _missed.length + " missed items: " + JSON.stringify(_missed), subagentType: "rlm_slice_worker" }));
161
+ }
162
+ JSON.stringify({
163
+ leaves: _leaves,
164
+ leafCount: _leaves.length,
165
+ covered: _handled.size,
166
+ total: workingSet.length,
167
+ missed: _missed.length,
168
+ maxSplitDepth: _splitDepths.length ? Math.max(..._splitDepths) : -1,
169
+ });
170
+ """.strip()
171
+
172
+
173
+ _SKILL_PATH = Path(__file__).parent / "SKILL.md"
174
+
175
+
176
+ def method_prompt() -> str:
177
+ """The RLM method (SKILL.md body, YAML frontmatter stripped) as the orchestrator's instructions.
178
+
179
+ The method is loaded into the orchestrator's **system prompt**, not wired as a lazy `skills=` source:
180
+ grounded on the pinned stack (2026-07-15), a real model does NOT proactively open a lazy skill source
181
+ before writing its `eval` workflow, so a lazy method never reaches it and it improvises a flat split.
182
+ The RLM method IS this node's defining job, so it belongs in the system prompt, always in front of the
183
+ model. (`skills=`/`worker_skills=` remain for auxiliary per-slice worker skills, which the worker may
184
+ open on demand.) See ADR-0018.
185
+ """
186
+ text = _SKILL_PATH.read_text(encoding="utf-8")
187
+ if text.startswith("---"): # strip YAML frontmatter
188
+ end = text.find("\n---", 3)
189
+ if end != -1:
190
+ text = text[end + 4 :]
191
+ return text.strip()
192
+
193
+
194
+ _ModelArg = Union[BaseChatModel, str, None]
195
+
196
+
197
+ def _resolve_model(model: _ModelArg, default_role: ModelRole) -> BaseChatModel:
198
+ """Resolve a model argument to a concrete `BaseChatModel` instance.
199
+
200
+ An instance passes through (tests inject fakes here); a string is a model id built through the seam;
201
+ `None` falls back to the profile for `default_role`. Always an instance — `create_deep_agent` would
202
+ otherwise resolve a bare id through its own provider path, which our OpenRouter-via-`langchain_openai`
203
+ seam does not use.
204
+ """
205
+ if isinstance(model, BaseChatModel):
206
+ return model
207
+ return build_model(model or model_for(default_role))
208
+
209
+
210
+ def decomposer_config(*, model: _ModelArg = None) -> SubAgent:
211
+ """The `rlm_decomposer` sub-agent: decides one level's split; a fresh agent per dispatch.
212
+
213
+ Reasoning work, so it defaults to the Gemma 4 class RLM role (`ModelRole.GENERAL`). It carries no
214
+ roster of its own — the interpreter, not the decomposer, drives the recursion (ADR-0015 Q2).
215
+ """
216
+ return {
217
+ "name": RLM_DECOMPOSER,
218
+ "description": "Decides how to split one working-set slice for recursive RLM decomposition.",
219
+ "system_prompt": _DECOMPOSER_PROMPT,
220
+ "model": _resolve_model(model, ModelRole.GENERAL),
221
+ }
222
+
223
+
224
+ def slice_worker_config(
225
+ *,
226
+ model: _ModelArg = None,
227
+ system_prompt: str = _SLICE_WORKER_PROMPT,
228
+ tools: Sequence[Union[BaseTool, Callable[..., Any], dict[str, Any]]] = (),
229
+ skills: Sequence[str] = (),
230
+ ) -> SubAgent:
231
+ """The `rlm_slice_worker` sub-agent: handles one leaf slice, with per-slice tools and skills.
232
+
233
+ The applying capability supplies `tools`, `skills`, and (via the `task()` description) the per-call
234
+ specialization (summarize for chunking, extract-and-synthesize for synthesis), so one config serves
235
+ both. The worker model is the applying capability's choice; it defaults to the RLM role.
236
+ """
237
+ config: SubAgent = {
238
+ "name": RLM_SLICE_WORKER,
239
+ "description": "Handles one focused leaf slice end to end, using per-slice tools and skills.",
240
+ "system_prompt": system_prompt,
241
+ "model": _resolve_model(model, ModelRole.GENERAL),
242
+ "tools": list(tools),
243
+ }
244
+ if skills:
245
+ config["skills"] = list(skills)
246
+ return config
247
+
248
+
249
+ def build_rlm_agent(
250
+ *,
251
+ reasoning_model: _ModelArg = None,
252
+ decomposer_model: _ModelArg = None,
253
+ worker_model: _ModelArg = None,
254
+ worker_system_prompt: str = _SLICE_WORKER_PROMPT,
255
+ worker_tools: Sequence[Union[BaseTool, Callable[..., Any], dict[str, Any]]] = (),
256
+ worker_skills: Sequence[str] = (),
257
+ tools: Sequence[Union[BaseTool, Callable[..., Any], dict[str, Any]]] = (),
258
+ system_prompt: Optional[str] = None,
259
+ skills: Optional[Sequence[str]] = None,
260
+ interpreter: Optional[CodeInterpreterMiddleware] = None,
261
+ ) -> CompiledStateGraph:
262
+ """Assemble the RLM Deep Agent: an interpreter orchestrator over the two named sub-agents.
263
+
264
+ The orchestrator holds the working set in the interpreter and runs the recursive `decompose()`
265
+ workflow (`RLM_WORKFLOW_JS`) via the `eval` tool, dispatching `rlm_decomposer` per level and
266
+ `rlm_slice_worker` per leaf. `reasoning_model` is the orchestrator; `decomposer_model`/`worker_model`
267
+ default to it / the profile. `system_prompt` defaults to the RLM method (`method_prompt()`), always in
268
+ front of the orchestrator; `skills`/`worker_skills`/`worker_tools` are auxiliary (the leaf worker's).
269
+ Tests inject fake models per role.
270
+
271
+ Returns the compiled agent. This is the reference assembly RAG_Wright's own tests and evals run and
272
+ the applying capabilities (T17, T28) build on; the graph half (GraphWright) assembles the production
273
+ node equivalently from the same skill and `grantedSubagents`.
274
+ """
275
+ orchestrator = _resolve_model(reasoning_model, ModelRole.GENERAL)
276
+ subagents: list[SubAgent] = [
277
+ decomposer_config(model=decomposer_model if decomposer_model is not None else reasoning_model),
278
+ slice_worker_config(
279
+ model=worker_model,
280
+ system_prompt=worker_system_prompt,
281
+ tools=worker_tools,
282
+ skills=worker_skills,
283
+ ),
284
+ ]
285
+ return create_deep_agent(
286
+ model=orchestrator,
287
+ tools=list(tools),
288
+ system_prompt=system_prompt if system_prompt is not None else method_prompt(),
289
+ subagents=subagents,
290
+ middleware=[interpreter or CodeInterpreterMiddleware(subagents=True)],
291
+ skills=list(skills) if skills else None,
292
+ )
@@ -0,0 +1,67 @@
1
+ ---
2
+ name: span_relevance_judgment
3
+ description: >
4
+ The corpus-retrieval RELEVANCE method: given ONE retrieved span (a contract clause's operative text) and ONE
5
+ structured condition being searched for (a clause type, optionally a specific value condition, with the user's
6
+ question as context), decide whether the span ACTUALLY ADDRESSES that condition -- relevant, not_relevant, or
7
+ uncertain from the text. This is the retrieval analog of the compliance judge (does this text satisfy this
8
+ thing?) and of answer abstention (does the evidence support an answer?): it returns a VERDICT, not a score, so
9
+ no caller has to choose a similarity threshold. The applying capability owns the deterministic guarantees
10
+ (verdict vocab, conservative default) -- this skill teaches only the reading.
11
+ ---
12
+
13
+ # Span relevance: does this retrieved span address this condition?
14
+
15
+ `typed_property_retrieval` always returns the top-k nearest spans, so a nonsense query still comes back with a
16
+ full page of clauses. Retrieval ranks by similarity; it never asks whether a span is *about* the thing searched
17
+ for. This skill supplies that missing judgement, so a sweep can honestly say "nothing in this corpus addresses
18
+ this condition" instead of surfacing eight near-misses.
19
+
20
+ ## The one hard constraint: you see only the span text
21
+
22
+ You are given the condition and ONE span's text. You judge whether **that span text** addresses the condition.
23
+ You cannot see the rest of the contract or the corpus. Judge only what the span itself shows.
24
+
25
+ ## What the condition is (and how to weigh its parts)
26
+
27
+ - **clause type** (primary) — the kind of clause being searched for, e.g. "Cap On Liability", "Renewal Term",
28
+ "Source Code Escrow". This is the main test: is the span a clause of, or squarely about, this type?
29
+ - **specific condition** (when present) — a narrower test within the type, e.g. "capped at a multiple of fees",
30
+ "auto-renews unless notice is given". When given, the span must address THIS, not just the general type.
31
+ - **the user's question** — CONTEXT ONLY. In a multi-condition sweep one question is shared across several
32
+ conditions, so it may be broader than, or only loosely tied to, this particular condition. Never treat a span
33
+ as relevant just because it echoes a word from the question; anchor on the clause type and the specific
34
+ condition.
35
+
36
+ ## Typed properties are evidence, not proof
37
+
38
+ The span may arrive with typed properties already detected on it (e.g. `cap_basis = multiple_of_fees`). These
39
+ were extracted from the question ONCE and reused across every condition in the sweep, so a property being
40
+ present does **not** prove the span is about *this* condition. A Source Code Escrow clause can carry
41
+ `cap_basis = multiple_of_fees` and still have nothing to do with a Cap On Liability search. Read the properties
42
+ as a hint, then decide from the span text itself.
43
+
44
+ ## The three verdicts
45
+
46
+ - **relevant** — the span is clearly a clause of the condition's type, or squarely addresses the specific
47
+ condition. A lawyer scanning results would say "yes, this is the clause you were looking for."
48
+
49
+ - **not_relevant** — the span is about something else. It was returned because it was among the nearest by
50
+ similarity (or shares an incidental property), but it does not address this clause type / condition. This is
51
+ the verdict that makes an honest "not found" reachable: when every returned span is not_relevant, the corpus
52
+ does not contain the condition.
53
+
54
+ - **uncertain** — the span text is too ambiguous, partial, or truncated to tell whether it addresses the
55
+ condition. Reserve this for genuine ambiguity, not for "probably not" (that is not_relevant) and not for
56
+ "probably yes" (that is relevant). Do not use uncertain to avoid a decision the text supports.
57
+
58
+ ## Output
59
+
60
+ Return the verdict (relevant / not_relevant / uncertain), a one- or two-sentence rationale grounded in the span
61
+ text, and a confidence in [0, 1].
62
+
63
+ ## What this skill does NOT own
64
+
65
+ The vocabulary mapping, the conservative default when you cannot be read, and how the verdicts roll up into a
66
+ product's matched / possible / not-found grouping are the APPLYING capability's and the caller's concern, not
67
+ this method's. This skill decides one span against one condition and explains why.
@@ -0,0 +1,36 @@
1
+ ---
2
+ name: vision_to_text
3
+ description: >
4
+ The scanned-image transcription method: given a single page image from an image-only filing, transcribe all
5
+ visible text exactly, preserving reading order, and output only the transcribed text. A single grounded
6
+ vision-language act on the GENERAL model (Gemma 4 class) through the model seam; the ingestion-side twin of
7
+ answer generation (both are the FR-C.9 generation capability, ADR-0014). Applied by the vision_to_text skill
8
+ runtime at ingestion for the image-only PDF subset.
9
+ ---
10
+
11
+ # Vision-to-text: transcribe the text visible in this image
12
+
13
+ This skill teaches a **method** for turning a page image into its text, so an image-only filing can be chunked,
14
+ embedded, and reasoned over like any parsed document. It is a single vision-language reading, not a workflow.
15
+
16
+ ## The method
17
+
18
+ - **Transcribe all text visible in the image exactly.** Reproduce the characters as written -- do not
19
+ paraphrase, summarize, correct, translate, or complete anything.
20
+ - **Preserve reading order.** Follow the page's natural top-to-bottom, left-to-right flow (and column order
21
+ where the page is multi-column), so the transcription reads as the document reads.
22
+ - **Output only the transcribed text.** No commentary, no description of the layout, no headings you invent,
23
+ no "here is the text" preamble -- just the text itself.
24
+
25
+ ## Boundaries
26
+
27
+ - If a region is unreadable, transcribe what is legible and do not fabricate the rest.
28
+ - The transcription is downstream evidence: it must be faithful to the page, because everything built on top
29
+ of it (chunks, embeddings, extracted facts, citations) inherits its errors.
30
+
31
+ ## What this skill does NOT own (the applying capability's job)
32
+
33
+ - the **model choice and the seam** (the GENERAL role via the model-profile seam -- product = self-hosted
34
+ Gemma-class, ADR-0039) and the **multimodal message assembly** (base64 data URI) are the capability runtime's
35
+ plumbing, not the method;
36
+ - the **output contract** (`VisionTranscription`) is attached by the capability.
@@ -0,0 +1 @@
1
+ """vision_to_text agent-skill folder: SKILL.md (the scanned-image transcription method)."""
@@ -0,0 +1 @@
1
+ """Operative-span segmentation (FR-R, ADR-0025)."""
@@ -0,0 +1,78 @@
1
+ """Provision-boundary decision: deterministic-first, with a decision-model (Jev) fallback for the UNCERTAIN residue.
2
+
3
+ `provision_boundary_verdict` (``spans.segment``) classifies each span deterministically into ``start`` / ``continue``
4
+ / ``uncertain``. The ``uncertain`` residue -- short, plausibly-heading lines in styles the deterministic rules do not
5
+ confidently classify (a new document convention, a colon/odd heading) -- is adjudicated by a pluggable async decider,
6
+ the ``jev_decision`` capability by default (invoked THROUGH the engine invoker, one batched call for the whole
7
+ residue). The decider is OPTIONAL: with no decision model configured, or on any error, an uncertain span GRACEFULLY
8
+ DEGRADES to "not a new provision" (the prior deterministic behavior) -- so ingestion never requires a decision model
9
+ and the hermetic tests stay offline. This is the neuro-symbolic shape (ADR-0040): a cheap, exact symbolic layer for
10
+ the clear majority + a model only for the ambiguous part -- flexible where regex is rigid, bounded in cost.
11
+ """
12
+ from __future__ import annotations
13
+
14
+ import os
15
+ from typing import Any, Awaitable, Callable, Optional
16
+
17
+ from rag_wright.spans.segment import provision_boundary_verdict
18
+
19
+ # A residue decider: candidate texts (the UNCERTAIN spans) -> a start flag each (True = begins a new provision).
20
+ BoundaryDecider = Callable[[list[str]], Awaitable[list[bool]]]
21
+
22
+ _THRESHOLD = 0.5
23
+
24
+
25
+ async def adecide_provision_starts(
26
+ texts: list[str], *, decider: Optional[BoundaryDecider] = None
27
+ ) -> list[bool]:
28
+ """Per-span "starts a new provision?" flags, aligned to ``texts``. Deterministic ``start``/``continue`` are
29
+ decided for free; only the ``uncertain`` residue is passed to ``decider`` (one batched call). No decider, or a
30
+ decider error, degrades the residue to False (fold in) -- never an exception, never a lost span."""
31
+ verdicts = [provision_boundary_verdict(t) for t in texts]
32
+ starts = [v == "start" for v in verdicts]
33
+ residue = [i for i, v in enumerate(verdicts) if v == "uncertain"]
34
+ if residue and decider is not None:
35
+ try:
36
+ decided = await decider([texts[i] for i in residue])
37
+ except Exception: # noqa: BLE001 - a decision-model blip must not fail ingestion; degrade to deterministic
38
+ decided = [False] * len(residue)
39
+ for i, flag in zip(residue, decided):
40
+ starts[i] = bool(flag)
41
+ return starts
42
+
43
+
44
+ def jev_boundary_decider(resources: Any = None) -> Optional[BoundaryDecider]:
45
+ """Build the Jev-backed residue decider, or ``None`` when no decision model is configured/registered (then the
46
+ boundary stays purely deterministic). It invokes the ``jev_decision`` capability THROUGH the engine invoker --
47
+ one batched call with all candidate lines in the ``state`` and one ``noul`` question per line."""
48
+ from rag_wright.api import ainvoke_model, capability_index
49
+ from rag_wright.models.profiles import decision_profile
50
+
51
+ model = os.environ.get("RAG_DECISION_MODEL")
52
+ prof = decision_profile(model)
53
+ if not os.environ.get(prof.api_key_env) or "jev_decision" not in capability_index():
54
+ return None # no decision model available -> deterministic-only
55
+
56
+ async def _decide(texts: list[str]) -> list[bool]:
57
+ state = "\n".join(f"[{i}] {t.strip()}" for i, t in enumerate(texts))
58
+ questions = {
59
+ f"c{i}": {
60
+ "type": "noul",
61
+ "instructions": (
62
+ f"In the numbered lines above, does line [{i}] BEGIN a new numbered section or provision of a "
63
+ "contract (a section heading / start), rather than continue the previous provision's text?"
64
+ ),
65
+ "criteria": {
66
+ "true": f"line [{i}] begins a new section or provision",
67
+ "false": f"line [{i}] continues the current provision",
68
+ },
69
+ }
70
+ for i in range(len(texts))
71
+ }
72
+ out = await ainvoke_model(
73
+ "jev_decision", {"state": state, "questions": questions, "model": model}, resources=resources
74
+ )
75
+ answers = out.get("answers", {}) if isinstance(out, dict) else {}
76
+ return [float((answers.get(f"c{i}") or {}).get("noul", 0.0)) >= _THRESHOLD for i in range(len(texts))]
77
+
78
+ return _decide