rag-wright 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- rag_wright/__init__.py +13 -0
- rag_wright/api/__init__.py +33 -0
- rag_wright/api/config.py +59 -0
- rag_wright/api/discover.py +70 -0
- rag_wright/api/documents.py +39 -0
- rag_wright/api/ids.py +31 -0
- rag_wright/api/invoke.py +99 -0
- rag_wright/api/kg.py +61 -0
- rag_wright/api/mcp.py +94 -0
- rag_wright/api/usage.py +30 -0
- rag_wright/api/workspace.py +85 -0
- rag_wright/capabilities/__init__.py +8 -0
- rag_wright/capabilities/answer_generator.py +427 -0
- rag_wright/capabilities/ard.py +286 -0
- rag_wright/capabilities/assertion_extraction.py +79 -0
- rag_wright/capabilities/chunk_read.py +58 -0
- rag_wright/capabilities/chunk_write.py +163 -0
- rag_wright/capabilities/claim_extraction.py +153 -0
- rag_wright/capabilities/clause_exception_linking.py +117 -0
- rag_wright/capabilities/compliance_judgment.py +322 -0
- rag_wright/capabilities/compliance_store.py +87 -0
- rag_wright/capabilities/contract_kg_serve.py +156 -0
- rag_wright/capabilities/contract_kg_store.py +251 -0
- rag_wright/capabilities/dg_extraction.py +585 -0
- rag_wright/capabilities/disambiguation.py +163 -0
- rag_wright/capabilities/document_parse.py +87 -0
- rag_wright/capabilities/document_scope.py +49 -0
- rag_wright/capabilities/embedding.py +164 -0
- rag_wright/capabilities/embedding_profiles.py +43 -0
- rag_wright/capabilities/entity_resolution.py +154 -0
- rag_wright/capabilities/fusion.py +64 -0
- rag_wright/capabilities/graph_extraction.py +243 -0
- rag_wright/capabilities/graph_query.py +73 -0
- rag_wright/capabilities/graph_storage.py +111 -0
- rag_wright/capabilities/highlight_serve.py +142 -0
- rag_wright/capabilities/hybrid_search.py +65 -0
- rag_wright/capabilities/invoke.py +31 -0
- rag_wright/capabilities/jev_decision.py +38 -0
- rag_wright/capabilities/manifests.py +872 -0
- rag_wright/capabilities/okf_navigate.py +456 -0
- rag_wright/capabilities/parsing.py +286 -0
- rag_wright/capabilities/property_boosted_retrieval.py +125 -0
- rag_wright/capabilities/query_function_classifier.py +94 -0
- rag_wright/capabilities/query_understanding.py +109 -0
- rag_wright/capabilities/registry.py +262 -0
- rag_wright/capabilities/remote_encoders.py +94 -0
- rag_wright/capabilities/requirement_extraction.py +247 -0
- rag_wright/capabilities/reranking.py +123 -0
- rag_wright/capabilities/retrieval_core.py +126 -0
- rag_wright/capabilities/rlm_chunking.py +808 -0
- rag_wright/capabilities/rlm_synthesis.py +316 -0
- rag_wright/capabilities/scan_quality.py +136 -0
- rag_wright/capabilities/span_relevance_judgment.py +191 -0
- rag_wright/capabilities/vision_to_text.py +85 -0
- rag_wright/capabilities/vlm_ocr.py +85 -0
- rag_wright/contracts/__init__.py +6 -0
- rag_wright/contracts/chunk.py +79 -0
- rag_wright/contracts/compliance.py +303 -0
- rag_wright/contracts/contract_meta.py +27 -0
- rag_wright/contracts/extraction.py +130 -0
- rag_wright/contracts/function.py +167 -0
- rag_wright/contracts/function_routing.py +91 -0
- rag_wright/contracts/highlight.py +74 -0
- rag_wright/contracts/identifiers.py +153 -0
- rag_wright/contracts/jurisdiction.py +96 -0
- rag_wright/contracts/ontology.py +142 -0
- rag_wright/contracts/property.py +201 -0
- rag_wright/contracts/provenance.py +78 -0
- rag_wright/contracts/query_intent.py +53 -0
- rag_wright/contracts/span.py +76 -0
- rag_wright/contracts/value_match.py +84 -0
- rag_wright/corpus/__init__.py +0 -0
- rag_wright/corpus/canonicalize.py +116 -0
- rag_wright/corpus/cuad.py +153 -0
- rag_wright/corpus/cuad_ingestion.py +72 -0
- rag_wright/corpus/document_parser.py +299 -0
- rag_wright/corpus/edgar.py +231 -0
- rag_wright/corpus/gcs_ingestion.py +120 -0
- rag_wright/corpus/http.py +110 -0
- rag_wright/corpus/selection.py +152 -0
- rag_wright/mcp/__init__.py +11 -0
- rag_wright/mcp/compliance_server.py +299 -0
- rag_wright/mcp/intra_document_qa_server.py +170 -0
- rag_wright/mcp/relational_qa_server.py +171 -0
- rag_wright/mcp/session_store.py +64 -0
- rag_wright/mcp/typed_property_retrieval_server.py +191 -0
- rag_wright/models/__init__.py +8 -0
- rag_wright/models/profiles.py +331 -0
- rag_wright/models/seam.py +497 -0
- rag_wright/models/tag_structured.py +285 -0
- rag_wright/models/tracing.py +179 -0
- rag_wright/models/usage.py +102 -0
- rag_wright/okf/__init__.py +11 -0
- rag_wright/okf/compile.py +292 -0
- rag_wright/okf/document.py +47 -0
- rag_wright/okf/enrich.py +176 -0
- rag_wright/okf/links.py +190 -0
- rag_wright/okf/lint.py +105 -0
- rag_wright/ontology/__init__.py +6 -0
- rag_wright/ontology/_generated_template_meta.py +60 -0
- rag_wright/ontology/_generated_vocab.py +52 -0
- rag_wright/ontology/clause_template.py +964 -0
- rag_wright/ontology/codegen.py +84 -0
- rag_wright/ontology/compliance_bridge.ttl +186 -0
- rag_wright/ontology/contract_bridge.ttl +2685 -0
- rag_wright/ontology/contract_taxonomy.py +24 -0
- rag_wright/ontology/derive.py +58 -0
- rag_wright/ontology/loader.py +435 -0
- rag_wright/ontology/packs/ftc_16cfr255.ttl +29 -0
- rag_wright/ontology/registry.py +87 -0
- rag_wright/ontology/template_introspect.py +100 -0
- rag_wright/py.typed +0 -0
- rag_wright/reference/__init__.py +2 -0
- rag_wright/reference/compliance.py +41 -0
- rag_wright/reference/contract_seam.py +123 -0
- rag_wright/skills/__init__.py +7 -0
- rag_wright/skills/claim_extraction/SKILL.md +47 -0
- rag_wright/skills/claim_extraction/__init__.py +1 -0
- rag_wright/skills/claim_extraction/template.py +50 -0
- rag_wright/skills/compliance_judgment/SKILL.md +59 -0
- rag_wright/skills/corpus_ingest/SKILL.md +106 -0
- rag_wright/skills/extraction_semantic_judge/SKILL.md +51 -0
- rag_wright/skills/extraction_semantic_judge/__init__.py +1 -0
- rag_wright/skills/generation/SKILL.md +64 -0
- rag_wright/skills/generation/__init__.py +1 -0
- rag_wright/skills/generic_compliance_judgment/SKILL.md +58 -0
- rag_wright/skills/okf_navigate/SKILL.md +137 -0
- rag_wright/skills/requirement_extraction/SKILL.md +47 -0
- rag_wright/skills/requirement_extraction/__init__.py +1 -0
- rag_wright/skills/requirement_extraction/template.py +50 -0
- rag_wright/skills/rlm/SKILL.md +186 -0
- rag_wright/skills/rlm/__init__.py +31 -0
- rag_wright/skills/rlm/agent.py +292 -0
- rag_wright/skills/span_relevance_judgment/SKILL.md +67 -0
- rag_wright/skills/vision_to_text/SKILL.md +36 -0
- rag_wright/skills/vision_to_text/__init__.py +1 -0
- rag_wright/spans/__init__.py +1 -0
- rag_wright/spans/boundary.py +78 -0
- rag_wright/spans/clause_function_classifier.py +490 -0
- rag_wright/spans/clause_kg_extractor.py +337 -0
- rag_wright/spans/cuad_labels.py +81 -0
- rag_wright/spans/dim_classifier.py +158 -0
- rag_wright/spans/dim_fleet.json +411 -0
- rag_wright/spans/function_classifier.py +77 -0
- rag_wright/spans/function_families.py +62 -0
- rag_wright/spans/hybrid_classifier.py +103 -0
- rag_wright/spans/legalbert_classifier.py +83 -0
- rag_wright/spans/model_capabilities.py +107 -0
- rag_wright/spans/new_function_labels.py +111 -0
- rag_wright/spans/page_map.py +68 -0
- rag_wright/spans/property_extractor.py +365 -0
- rag_wright/spans/property_grounding.py +182 -0
- rag_wright/spans/reclassify.py +77 -0
- rag_wright/spans/scarce_function_labels.py +105 -0
- rag_wright/spans/segment.py +341 -0
- rag_wright/spans/semantic_judge.py +197 -0
- rag_wright/spans/symbolic_validation.py +131 -0
- rag_wright/spans/tag_clause_extractor.py +182 -0
- rag_wright/store/__init__.py +6 -0
- rag_wright/store/arcadedb.py +1135 -0
- rag_wright/store/chunk_text.py +66 -0
- rag_wright/store/seam.py +213 -0
- rag_wright/subgraphs/__init__.py +0 -0
- rag_wright/subgraphs/async_ingestion.py +204 -0
- rag_wright/subgraphs/compliance_check.py +1042 -0
- rag_wright/subgraphs/compliance_ingestion.py +306 -0
- rag_wright/subgraphs/contract_ingestion_pipeline.py +999 -0
- rag_wright/subgraphs/graph_extraction.py +102 -0
- rag_wright/subgraphs/intra_document_qa.py +328 -0
- rag_wright/subgraphs/observability.py +140 -0
- rag_wright/subgraphs/query_constraint_extraction.py +73 -0
- rag_wright/subgraphs/relational_qa.py +165 -0
- rag_wright/subgraphs/requirement_extraction.py +137 -0
- rag_wright/subgraphs/scaffold.py +65 -0
- rag_wright/subgraphs/semantic_chunking.py +183 -0
- rag_wright/subgraphs/typed_clause_extraction.py +172 -0
- rag_wright/subgraphs/typed_property_retrieval.py +278 -0
- rag_wright/util/__init__.py +1 -0
- rag_wright/util/concurrent.py +153 -0
- rag_wright/util/spacy_model.py +45 -0
- rag_wright-0.1.0.dist-info/METADATA +168 -0
- rag_wright-0.1.0.dist-info/RECORD +184 -0
- rag_wright-0.1.0.dist-info/WHEEL +4 -0
- rag_wright-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,292 @@
|
|
|
1
|
+
"""RLM machinery (FR-C.10, `rlm_method`): the Deep Agents assembly the RLM method runs on.
|
|
2
|
+
|
|
3
|
+
This is the reusable RLM runtime the RLM chunking (T17) and RLM synthesis (T28) capabilities apply.
|
|
4
|
+
It is authored software, not a build-tool feature: the RLM node is a Deep Agent whose interpreter
|
|
5
|
+
(`CodeInterpreterMiddleware`) holds the working set and runs a recursive `decompose()` *workflow* that
|
|
6
|
+
dispatches sub-agents with `task()` — a **fresh `rlm_decomposer`** at each over-budget internal level
|
|
7
|
+
(per-level fresh context) and an `rlm_slice_worker` at each leaf. Arbitrary depth comes from the
|
|
8
|
+
interpreter re-entering `decompose()`; the interpreter holds the recursion stack.
|
|
9
|
+
|
|
10
|
+
Two named sub-agents, dispatched by name (ADR-0015):
|
|
11
|
+
- `rlm_decomposer` — decides one level's split for the slice it is handed; a fresh agent per dispatch.
|
|
12
|
+
- `rlm_slice_worker` — handles one leaf slice; per-slice tool use and per-slice skills live here.
|
|
13
|
+
|
|
14
|
+
The recursion is driven by the interpreter, **not** by an agent dispatching itself: a self-referential
|
|
15
|
+
sub-agent is not constructible on the pinned `deepagents==0.6.12` (its `SubAgentMiddleware.__init__`
|
|
16
|
+
compiles its roster eagerly, so a config whose roster contains itself recurses at construction). See
|
|
17
|
+
ADR-0015 (Q2, corrected — design B') for the grounded reason and why interpreter-driven recursion is the
|
|
18
|
+
faithful realization, not a workaround.
|
|
19
|
+
|
|
20
|
+
Models resolve through the model-profile seam (T11): the RLM reasoning role (orchestrator + decomposer)
|
|
21
|
+
is a Gemma 4 class model (`ModelRole.GENERAL`); the leaf worker's model is chosen by the applying
|
|
22
|
+
capability (a smaller model for chunking summaries, the structured-reasoning model for synthesis). No
|
|
23
|
+
provider or model flag lives here.
|
|
24
|
+
"""
|
|
25
|
+
|
|
26
|
+
from __future__ import annotations
|
|
27
|
+
|
|
28
|
+
import threading
|
|
29
|
+
from collections.abc import Callable, Iterator, Sequence
|
|
30
|
+
from contextlib import contextmanager
|
|
31
|
+
from pathlib import Path
|
|
32
|
+
from typing import Any, Optional, Union
|
|
33
|
+
|
|
34
|
+
from langchain_core.language_models.chat_models import BaseChatModel
|
|
35
|
+
from langchain_core.tools import BaseTool
|
|
36
|
+
from langchain_quickjs import CodeInterpreterMiddleware
|
|
37
|
+
from langgraph.graph.state import CompiledStateGraph
|
|
38
|
+
|
|
39
|
+
from deepagents import create_deep_agent
|
|
40
|
+
from deepagents.middleware.subagents import SubAgent
|
|
41
|
+
from rag_wright.models.profiles import ModelRole, model_for
|
|
42
|
+
from rag_wright.models.seam import build_model
|
|
43
|
+
|
|
44
|
+
# The two named sub-agents (ADR-0015 Q1). Also the exact `grantedSubagents` the manifests declare.
|
|
45
|
+
RLM_DECOMPOSER = "rlm_decomposer"
|
|
46
|
+
RLM_SLICE_WORKER = "rlm_slice_worker"
|
|
47
|
+
GRANTED_SUBAGENTS: tuple[str, str] = (RLM_DECOMPOSER, RLM_SLICE_WORKER)
|
|
48
|
+
|
|
49
|
+
# LOAD-BEARING, do NOT remove as an "uncontended lock in a serial path" (KI-1, ADR-0020). Two QuickJS
|
|
50
|
+
# interpreter runtimes coexisting in one process race on shared Rust state and silently complete without
|
|
51
|
+
# dispatching ~half the time, with zero exceptions. This process-wide semaphore serializes the FULL
|
|
52
|
+
# interpreter-session lifetime (build -> run -> teardown) so no two RLM runtimes are ever alive at once.
|
|
53
|
+
# It is uncontended (zero cost) while ingestion is serial; it makes the safe behaviour the DEFAULT so that
|
|
54
|
+
# adding concurrency later turns a silent-correctness failure into a visible-performance one (slower, not
|
|
55
|
+
# wrong). It lifts only when the upstream coexistence bug is fixed (ADR-0020 exit path). Task T35 designs
|
|
56
|
+
# the real concurrent batch; this is the always-on correctness floor beneath it, not throughput tuning.
|
|
57
|
+
_INTERPRETER_SEMAPHORE = threading.BoundedSemaphore(1)
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
@contextmanager
|
|
61
|
+
def rlm_interpreter_session(
|
|
62
|
+
*, ptc: Sequence[BaseTool] = (), max_result_chars: Optional[int] = None
|
|
63
|
+
) -> Iterator[CodeInterpreterMiddleware]:
|
|
64
|
+
"""Own the process for exactly one RLM interpreter session (KI-1, ADR-0020).
|
|
65
|
+
|
|
66
|
+
Holds `_INTERPRETER_SEMAPHORE` from before the interpreter runtime is built (the middleware is created
|
|
67
|
+
here; its QuickJS runtime is built lazily on first `eval`, inside the lock) until after it is torn down
|
|
68
|
+
(the registry is closed here, deterministically, not on GC timing) — the whole coexistence window, not
|
|
69
|
+
just the dispatch call. Build the RLM agent with the yielded middleware and run it inside the `with`;
|
|
70
|
+
do parsing/String work outside it. Reentrant call from within one session would deadlock — one session
|
|
71
|
+
per call stack, which the discoverer/extractor honour (they invoke once, then parse outside).
|
|
72
|
+
|
|
73
|
+
`ptc` are Programmatic-Tool-Calling tools exposed **inside** the interpreter as `tools.<camelCase>()`
|
|
74
|
+
and never as top-level tools — this is how the working set is delivered as a JS value that stays out of
|
|
75
|
+
the model's context (`working_set` → `tools.workingSet()`, T36 / GraphWright working-set contract).
|
|
76
|
+
"""
|
|
77
|
+
# `max_result_chars` overrides the interpreter's 4,000-char eval-result cap (which truncates a large
|
|
78
|
+
# JSON return mid-string). A capability whose workflow returns a big result (e.g. okf_navigate's shortlist
|
|
79
|
+
# + decision log at a high frontier budget) raises it; RLM leaves it at the default.
|
|
80
|
+
_INTERPRETER_SEMAPHORE.acquire()
|
|
81
|
+
kwargs: dict = {"subagents": True, "ptc": list(ptc) or None}
|
|
82
|
+
if max_result_chars is not None:
|
|
83
|
+
kwargs["max_result_chars"] = max_result_chars
|
|
84
|
+
interpreter = CodeInterpreterMiddleware(**kwargs)
|
|
85
|
+
try:
|
|
86
|
+
yield interpreter
|
|
87
|
+
finally:
|
|
88
|
+
try:
|
|
89
|
+
interpreter._registry.close() # tear the QuickJS runtime down inside the lock (no coexistence)
|
|
90
|
+
finally:
|
|
91
|
+
_INTERPRETER_SEMAPHORE.release()
|
|
92
|
+
|
|
93
|
+
_DECOMPOSER_PROMPT = (
|
|
94
|
+
"You decide how to split ONE working-set slice (a list of items) for a recursive divide-and-conquer. "
|
|
95
|
+
"If the slice is small and focused enough to handle directly, mark it a leaf; otherwise return the "
|
|
96
|
+
"split offsets that partition it into coherent contiguous groups. Reply as JSON: {\"leaf\": true} for "
|
|
97
|
+
"a leaf, or {\"leaf\": false, \"cuts\": [i, j, ...]} where each cut is an ascending index into the "
|
|
98
|
+
"slice at which a new group begins. The cuts partition the slice, so no item is lost. Decide only THIS "
|
|
99
|
+
"level; the interpreter re-dispatches you on each group."
|
|
100
|
+
)
|
|
101
|
+
_SLICE_WORKER_PROMPT = (
|
|
102
|
+
"You handle ONE focused working-set slice end to end. Use the tools and skills you are given as "
|
|
103
|
+
"needed, then return the result for this slice only. You never see the whole working set."
|
|
104
|
+
)
|
|
105
|
+
|
|
106
|
+
# The canonical interpreter-driven recursive workflow (ADR-0015 Q2 corrected, design B'). Reads the
|
|
107
|
+
# working set from the runtime PTC tool `tools.workingSet()` (T36) — a JS value that never enters context —
|
|
108
|
+
# and dispatches sub-agents by name via `task()`. Recursion lives HERE, in the interpreter: a fresh
|
|
109
|
+
# `rlm_decomposer` decides each level's split, `decompose()` re-enters itself on the returned parts
|
|
110
|
+
# (depth capped at _MAX_DEPTH = 3, the interpreter holds the stack), and `rlm_slice_worker` handles each leaf. The RLM
|
|
111
|
+
# skill (SKILL.md) teaches this workflow; the node writes it into the `eval` tool when its request asks
|
|
112
|
+
# for a "workflow" (the trigger GraphWright's applier guarantees, requiresDynamicDispatch). It returns
|
|
113
|
+
# the per-leaf results plus the depths at which splitting occurred, so the descent is inspectable.
|
|
114
|
+
RLM_WORKFLOW_JS = r"""
|
|
115
|
+
// Recursive divide-and-conquer over a working set of items. Read it from the runtime tool (a JS value
|
|
116
|
+
// that stays in the interpreter and never enters the model's context), then fan the work out to
|
|
117
|
+
// sub-agents in code (a "workflow"), never one grinding tool call at a time. The interpreter holds the
|
|
118
|
+
// working set and the recursion stack; the model is only ever called on a focused slice.
|
|
119
|
+
const workingSet = await tools.workingSet(); // [{id, ...}, ...] delivered by the runtime, never in context
|
|
120
|
+
// LOAD-COMPLETENESS ASSERTION (T37): verify you loaded the WHOLE delivered set before anything else. The
|
|
121
|
+
// size comes from the runtime (a scalar it cannot under-read); if the load is short, fail loud rather than
|
|
122
|
+
// silently working over a truncated set — the coverage tail below only guarantees coverage over what you
|
|
123
|
+
// loaded, so an under-read here is a silent evidence drop nothing downstream catches.
|
|
124
|
+
const _delivered = await tools.workingSetSize();
|
|
125
|
+
if (workingSet.length !== _delivered) {
|
|
126
|
+
throw new Error("LOAD UNDER-READ: loaded " + workingSet.length + " of " + _delivered + " delivered items");
|
|
127
|
+
}
|
|
128
|
+
const _MAX_DEPTH = 3; // hard cap on recursion — beyond this, force leaf (no runaway splits)
|
|
129
|
+
const _splitDepths = []; // the depths at which decompose() re-entered itself (proof of descent)
|
|
130
|
+
const _handled = new Set(); // ids of working-set items a leaf worker covered
|
|
131
|
+
async function decompose(items, depth) {
|
|
132
|
+
if (items.length === 0) return []; // empty slice — nothing to dispatch (no-op leaf)
|
|
133
|
+
if (depth >= _MAX_DEPTH) {
|
|
134
|
+
for (const it of items) _handled.add(it.id);
|
|
135
|
+
return [await task({ description: "handle leaf depth " + depth + " over " + items.length + " items: " + JSON.stringify(items), subagentType: "rlm_slice_worker" })];
|
|
136
|
+
}
|
|
137
|
+
const decision = JSON.parse(await task({
|
|
138
|
+
description: "decompose depth " + depth + " over " + items.length + " items",
|
|
139
|
+
subagentType: "rlm_decomposer",
|
|
140
|
+
}));
|
|
141
|
+
if (decision.leaf) {
|
|
142
|
+
for (const it of items) _handled.add(it.id);
|
|
143
|
+
return [await task({ description: "handle leaf depth " + depth + " over " + items.length + " items: " + JSON.stringify(items), subagentType: "rlm_slice_worker" })];
|
|
144
|
+
}
|
|
145
|
+
_splitDepths.push(depth);
|
|
146
|
+
// decision.cuts partition `items` into contiguous groups (no item lost); recurse on every group.
|
|
147
|
+
const bounds = [0, ...decision.cuts, items.length];
|
|
148
|
+
const groups = [];
|
|
149
|
+
for (let i = 0; i < bounds.length - 1; i++) groups.push(items.slice(bounds[i], bounds[i + 1]));
|
|
150
|
+
const handled = await Promise.all(groups.map((g) => decompose(g, depth + 1)));
|
|
151
|
+
return handled.flat();
|
|
152
|
+
}
|
|
153
|
+
const _leaves = await decompose(workingSet, 0);
|
|
154
|
+
// COVERAGE TAIL (structural, in-interpreter): the code holds EVERY item, so it guarantees coverage even
|
|
155
|
+
// if the recursion missed a deep leaf out of context — a silent drop otherwise (T37). Any uncovered item
|
|
156
|
+
// is dispatched now, not dropped. This is code checking coverage, not the model asked to be thorough.
|
|
157
|
+
const _missed = workingSet.filter((it) => !_handled.has(it.id));
|
|
158
|
+
if (_missed.length) {
|
|
159
|
+
for (const it of _missed) _handled.add(it.id);
|
|
160
|
+
_leaves.push(await task({ description: "cover " + _missed.length + " missed items: " + JSON.stringify(_missed), subagentType: "rlm_slice_worker" }));
|
|
161
|
+
}
|
|
162
|
+
JSON.stringify({
|
|
163
|
+
leaves: _leaves,
|
|
164
|
+
leafCount: _leaves.length,
|
|
165
|
+
covered: _handled.size,
|
|
166
|
+
total: workingSet.length,
|
|
167
|
+
missed: _missed.length,
|
|
168
|
+
maxSplitDepth: _splitDepths.length ? Math.max(..._splitDepths) : -1,
|
|
169
|
+
});
|
|
170
|
+
""".strip()
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
_SKILL_PATH = Path(__file__).parent / "SKILL.md"
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
def method_prompt() -> str:
|
|
177
|
+
"""The RLM method (SKILL.md body, YAML frontmatter stripped) as the orchestrator's instructions.
|
|
178
|
+
|
|
179
|
+
The method is loaded into the orchestrator's **system prompt**, not wired as a lazy `skills=` source:
|
|
180
|
+
grounded on the pinned stack (2026-07-15), a real model does NOT proactively open a lazy skill source
|
|
181
|
+
before writing its `eval` workflow, so a lazy method never reaches it and it improvises a flat split.
|
|
182
|
+
The RLM method IS this node's defining job, so it belongs in the system prompt, always in front of the
|
|
183
|
+
model. (`skills=`/`worker_skills=` remain for auxiliary per-slice worker skills, which the worker may
|
|
184
|
+
open on demand.) See ADR-0018.
|
|
185
|
+
"""
|
|
186
|
+
text = _SKILL_PATH.read_text(encoding="utf-8")
|
|
187
|
+
if text.startswith("---"): # strip YAML frontmatter
|
|
188
|
+
end = text.find("\n---", 3)
|
|
189
|
+
if end != -1:
|
|
190
|
+
text = text[end + 4 :]
|
|
191
|
+
return text.strip()
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
_ModelArg = Union[BaseChatModel, str, None]
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
def _resolve_model(model: _ModelArg, default_role: ModelRole) -> BaseChatModel:
|
|
198
|
+
"""Resolve a model argument to a concrete `BaseChatModel` instance.
|
|
199
|
+
|
|
200
|
+
An instance passes through (tests inject fakes here); a string is a model id built through the seam;
|
|
201
|
+
`None` falls back to the profile for `default_role`. Always an instance — `create_deep_agent` would
|
|
202
|
+
otherwise resolve a bare id through its own provider path, which our OpenRouter-via-`langchain_openai`
|
|
203
|
+
seam does not use.
|
|
204
|
+
"""
|
|
205
|
+
if isinstance(model, BaseChatModel):
|
|
206
|
+
return model
|
|
207
|
+
return build_model(model or model_for(default_role))
|
|
208
|
+
|
|
209
|
+
|
|
210
|
+
def decomposer_config(*, model: _ModelArg = None) -> SubAgent:
|
|
211
|
+
"""The `rlm_decomposer` sub-agent: decides one level's split; a fresh agent per dispatch.
|
|
212
|
+
|
|
213
|
+
Reasoning work, so it defaults to the Gemma 4 class RLM role (`ModelRole.GENERAL`). It carries no
|
|
214
|
+
roster of its own — the interpreter, not the decomposer, drives the recursion (ADR-0015 Q2).
|
|
215
|
+
"""
|
|
216
|
+
return {
|
|
217
|
+
"name": RLM_DECOMPOSER,
|
|
218
|
+
"description": "Decides how to split one working-set slice for recursive RLM decomposition.",
|
|
219
|
+
"system_prompt": _DECOMPOSER_PROMPT,
|
|
220
|
+
"model": _resolve_model(model, ModelRole.GENERAL),
|
|
221
|
+
}
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
def slice_worker_config(
|
|
225
|
+
*,
|
|
226
|
+
model: _ModelArg = None,
|
|
227
|
+
system_prompt: str = _SLICE_WORKER_PROMPT,
|
|
228
|
+
tools: Sequence[Union[BaseTool, Callable[..., Any], dict[str, Any]]] = (),
|
|
229
|
+
skills: Sequence[str] = (),
|
|
230
|
+
) -> SubAgent:
|
|
231
|
+
"""The `rlm_slice_worker` sub-agent: handles one leaf slice, with per-slice tools and skills.
|
|
232
|
+
|
|
233
|
+
The applying capability supplies `tools`, `skills`, and (via the `task()` description) the per-call
|
|
234
|
+
specialization (summarize for chunking, extract-and-synthesize for synthesis), so one config serves
|
|
235
|
+
both. The worker model is the applying capability's choice; it defaults to the RLM role.
|
|
236
|
+
"""
|
|
237
|
+
config: SubAgent = {
|
|
238
|
+
"name": RLM_SLICE_WORKER,
|
|
239
|
+
"description": "Handles one focused leaf slice end to end, using per-slice tools and skills.",
|
|
240
|
+
"system_prompt": system_prompt,
|
|
241
|
+
"model": _resolve_model(model, ModelRole.GENERAL),
|
|
242
|
+
"tools": list(tools),
|
|
243
|
+
}
|
|
244
|
+
if skills:
|
|
245
|
+
config["skills"] = list(skills)
|
|
246
|
+
return config
|
|
247
|
+
|
|
248
|
+
|
|
249
|
+
def build_rlm_agent(
|
|
250
|
+
*,
|
|
251
|
+
reasoning_model: _ModelArg = None,
|
|
252
|
+
decomposer_model: _ModelArg = None,
|
|
253
|
+
worker_model: _ModelArg = None,
|
|
254
|
+
worker_system_prompt: str = _SLICE_WORKER_PROMPT,
|
|
255
|
+
worker_tools: Sequence[Union[BaseTool, Callable[..., Any], dict[str, Any]]] = (),
|
|
256
|
+
worker_skills: Sequence[str] = (),
|
|
257
|
+
tools: Sequence[Union[BaseTool, Callable[..., Any], dict[str, Any]]] = (),
|
|
258
|
+
system_prompt: Optional[str] = None,
|
|
259
|
+
skills: Optional[Sequence[str]] = None,
|
|
260
|
+
interpreter: Optional[CodeInterpreterMiddleware] = None,
|
|
261
|
+
) -> CompiledStateGraph:
|
|
262
|
+
"""Assemble the RLM Deep Agent: an interpreter orchestrator over the two named sub-agents.
|
|
263
|
+
|
|
264
|
+
The orchestrator holds the working set in the interpreter and runs the recursive `decompose()`
|
|
265
|
+
workflow (`RLM_WORKFLOW_JS`) via the `eval` tool, dispatching `rlm_decomposer` per level and
|
|
266
|
+
`rlm_slice_worker` per leaf. `reasoning_model` is the orchestrator; `decomposer_model`/`worker_model`
|
|
267
|
+
default to it / the profile. `system_prompt` defaults to the RLM method (`method_prompt()`), always in
|
|
268
|
+
front of the orchestrator; `skills`/`worker_skills`/`worker_tools` are auxiliary (the leaf worker's).
|
|
269
|
+
Tests inject fake models per role.
|
|
270
|
+
|
|
271
|
+
Returns the compiled agent. This is the reference assembly RAG_Wright's own tests and evals run and
|
|
272
|
+
the applying capabilities (T17, T28) build on; the graph half (GraphWright) assembles the production
|
|
273
|
+
node equivalently from the same skill and `grantedSubagents`.
|
|
274
|
+
"""
|
|
275
|
+
orchestrator = _resolve_model(reasoning_model, ModelRole.GENERAL)
|
|
276
|
+
subagents: list[SubAgent] = [
|
|
277
|
+
decomposer_config(model=decomposer_model if decomposer_model is not None else reasoning_model),
|
|
278
|
+
slice_worker_config(
|
|
279
|
+
model=worker_model,
|
|
280
|
+
system_prompt=worker_system_prompt,
|
|
281
|
+
tools=worker_tools,
|
|
282
|
+
skills=worker_skills,
|
|
283
|
+
),
|
|
284
|
+
]
|
|
285
|
+
return create_deep_agent(
|
|
286
|
+
model=orchestrator,
|
|
287
|
+
tools=list(tools),
|
|
288
|
+
system_prompt=system_prompt if system_prompt is not None else method_prompt(),
|
|
289
|
+
subagents=subagents,
|
|
290
|
+
middleware=[interpreter or CodeInterpreterMiddleware(subagents=True)],
|
|
291
|
+
skills=list(skills) if skills else None,
|
|
292
|
+
)
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: span_relevance_judgment
|
|
3
|
+
description: >
|
|
4
|
+
The corpus-retrieval RELEVANCE method: given ONE retrieved span (a contract clause's operative text) and ONE
|
|
5
|
+
structured condition being searched for (a clause type, optionally a specific value condition, with the user's
|
|
6
|
+
question as context), decide whether the span ACTUALLY ADDRESSES that condition -- relevant, not_relevant, or
|
|
7
|
+
uncertain from the text. This is the retrieval analog of the compliance judge (does this text satisfy this
|
|
8
|
+
thing?) and of answer abstention (does the evidence support an answer?): it returns a VERDICT, not a score, so
|
|
9
|
+
no caller has to choose a similarity threshold. The applying capability owns the deterministic guarantees
|
|
10
|
+
(verdict vocab, conservative default) -- this skill teaches only the reading.
|
|
11
|
+
---
|
|
12
|
+
|
|
13
|
+
# Span relevance: does this retrieved span address this condition?
|
|
14
|
+
|
|
15
|
+
`typed_property_retrieval` always returns the top-k nearest spans, so a nonsense query still comes back with a
|
|
16
|
+
full page of clauses. Retrieval ranks by similarity; it never asks whether a span is *about* the thing searched
|
|
17
|
+
for. This skill supplies that missing judgement, so a sweep can honestly say "nothing in this corpus addresses
|
|
18
|
+
this condition" instead of surfacing eight near-misses.
|
|
19
|
+
|
|
20
|
+
## The one hard constraint: you see only the span text
|
|
21
|
+
|
|
22
|
+
You are given the condition and ONE span's text. You judge whether **that span text** addresses the condition.
|
|
23
|
+
You cannot see the rest of the contract or the corpus. Judge only what the span itself shows.
|
|
24
|
+
|
|
25
|
+
## What the condition is (and how to weigh its parts)
|
|
26
|
+
|
|
27
|
+
- **clause type** (primary) — the kind of clause being searched for, e.g. "Cap On Liability", "Renewal Term",
|
|
28
|
+
"Source Code Escrow". This is the main test: is the span a clause of, or squarely about, this type?
|
|
29
|
+
- **specific condition** (when present) — a narrower test within the type, e.g. "capped at a multiple of fees",
|
|
30
|
+
"auto-renews unless notice is given". When given, the span must address THIS, not just the general type.
|
|
31
|
+
- **the user's question** — CONTEXT ONLY. In a multi-condition sweep one question is shared across several
|
|
32
|
+
conditions, so it may be broader than, or only loosely tied to, this particular condition. Never treat a span
|
|
33
|
+
as relevant just because it echoes a word from the question; anchor on the clause type and the specific
|
|
34
|
+
condition.
|
|
35
|
+
|
|
36
|
+
## Typed properties are evidence, not proof
|
|
37
|
+
|
|
38
|
+
The span may arrive with typed properties already detected on it (e.g. `cap_basis = multiple_of_fees`). These
|
|
39
|
+
were extracted from the question ONCE and reused across every condition in the sweep, so a property being
|
|
40
|
+
present does **not** prove the span is about *this* condition. A Source Code Escrow clause can carry
|
|
41
|
+
`cap_basis = multiple_of_fees` and still have nothing to do with a Cap On Liability search. Read the properties
|
|
42
|
+
as a hint, then decide from the span text itself.
|
|
43
|
+
|
|
44
|
+
## The three verdicts
|
|
45
|
+
|
|
46
|
+
- **relevant** — the span is clearly a clause of the condition's type, or squarely addresses the specific
|
|
47
|
+
condition. A lawyer scanning results would say "yes, this is the clause you were looking for."
|
|
48
|
+
|
|
49
|
+
- **not_relevant** — the span is about something else. It was returned because it was among the nearest by
|
|
50
|
+
similarity (or shares an incidental property), but it does not address this clause type / condition. This is
|
|
51
|
+
the verdict that makes an honest "not found" reachable: when every returned span is not_relevant, the corpus
|
|
52
|
+
does not contain the condition.
|
|
53
|
+
|
|
54
|
+
- **uncertain** — the span text is too ambiguous, partial, or truncated to tell whether it addresses the
|
|
55
|
+
condition. Reserve this for genuine ambiguity, not for "probably not" (that is not_relevant) and not for
|
|
56
|
+
"probably yes" (that is relevant). Do not use uncertain to avoid a decision the text supports.
|
|
57
|
+
|
|
58
|
+
## Output
|
|
59
|
+
|
|
60
|
+
Return the verdict (relevant / not_relevant / uncertain), a one- or two-sentence rationale grounded in the span
|
|
61
|
+
text, and a confidence in [0, 1].
|
|
62
|
+
|
|
63
|
+
## What this skill does NOT own
|
|
64
|
+
|
|
65
|
+
The vocabulary mapping, the conservative default when you cannot be read, and how the verdicts roll up into a
|
|
66
|
+
product's matched / possible / not-found grouping are the APPLYING capability's and the caller's concern, not
|
|
67
|
+
this method's. This skill decides one span against one condition and explains why.
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: vision_to_text
|
|
3
|
+
description: >
|
|
4
|
+
The scanned-image transcription method: given a single page image from an image-only filing, transcribe all
|
|
5
|
+
visible text exactly, preserving reading order, and output only the transcribed text. A single grounded
|
|
6
|
+
vision-language act on the GENERAL model (Gemma 4 class) through the model seam; the ingestion-side twin of
|
|
7
|
+
answer generation (both are the FR-C.9 generation capability, ADR-0014). Applied by the vision_to_text skill
|
|
8
|
+
runtime at ingestion for the image-only PDF subset.
|
|
9
|
+
---
|
|
10
|
+
|
|
11
|
+
# Vision-to-text: transcribe the text visible in this image
|
|
12
|
+
|
|
13
|
+
This skill teaches a **method** for turning a page image into its text, so an image-only filing can be chunked,
|
|
14
|
+
embedded, and reasoned over like any parsed document. It is a single vision-language reading, not a workflow.
|
|
15
|
+
|
|
16
|
+
## The method
|
|
17
|
+
|
|
18
|
+
- **Transcribe all text visible in the image exactly.** Reproduce the characters as written -- do not
|
|
19
|
+
paraphrase, summarize, correct, translate, or complete anything.
|
|
20
|
+
- **Preserve reading order.** Follow the page's natural top-to-bottom, left-to-right flow (and column order
|
|
21
|
+
where the page is multi-column), so the transcription reads as the document reads.
|
|
22
|
+
- **Output only the transcribed text.** No commentary, no description of the layout, no headings you invent,
|
|
23
|
+
no "here is the text" preamble -- just the text itself.
|
|
24
|
+
|
|
25
|
+
## Boundaries
|
|
26
|
+
|
|
27
|
+
- If a region is unreadable, transcribe what is legible and do not fabricate the rest.
|
|
28
|
+
- The transcription is downstream evidence: it must be faithful to the page, because everything built on top
|
|
29
|
+
of it (chunks, embeddings, extracted facts, citations) inherits its errors.
|
|
30
|
+
|
|
31
|
+
## What this skill does NOT own (the applying capability's job)
|
|
32
|
+
|
|
33
|
+
- the **model choice and the seam** (the GENERAL role via the model-profile seam -- product = self-hosted
|
|
34
|
+
Gemma-class, ADR-0039) and the **multimodal message assembly** (base64 data URI) are the capability runtime's
|
|
35
|
+
plumbing, not the method;
|
|
36
|
+
- the **output contract** (`VisionTranscription`) is attached by the capability.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""vision_to_text agent-skill folder: SKILL.md (the scanned-image transcription method)."""
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""Operative-span segmentation (FR-R, ADR-0025)."""
|
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
"""Provision-boundary decision: deterministic-first, with a decision-model (Jev) fallback for the UNCERTAIN residue.
|
|
2
|
+
|
|
3
|
+
`provision_boundary_verdict` (``spans.segment``) classifies each span deterministically into ``start`` / ``continue``
|
|
4
|
+
/ ``uncertain``. The ``uncertain`` residue -- short, plausibly-heading lines in styles the deterministic rules do not
|
|
5
|
+
confidently classify (a new document convention, a colon/odd heading) -- is adjudicated by a pluggable async decider,
|
|
6
|
+
the ``jev_decision`` capability by default (invoked THROUGH the engine invoker, one batched call for the whole
|
|
7
|
+
residue). The decider is OPTIONAL: with no decision model configured, or on any error, an uncertain span GRACEFULLY
|
|
8
|
+
DEGRADES to "not a new provision" (the prior deterministic behavior) -- so ingestion never requires a decision model
|
|
9
|
+
and the hermetic tests stay offline. This is the neuro-symbolic shape (ADR-0040): a cheap, exact symbolic layer for
|
|
10
|
+
the clear majority + a model only for the ambiguous part -- flexible where regex is rigid, bounded in cost.
|
|
11
|
+
"""
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import os
|
|
15
|
+
from typing import Any, Awaitable, Callable, Optional
|
|
16
|
+
|
|
17
|
+
from rag_wright.spans.segment import provision_boundary_verdict
|
|
18
|
+
|
|
19
|
+
# A residue decider: candidate texts (the UNCERTAIN spans) -> a start flag each (True = begins a new provision).
|
|
20
|
+
BoundaryDecider = Callable[[list[str]], Awaitable[list[bool]]]
|
|
21
|
+
|
|
22
|
+
_THRESHOLD = 0.5
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
async def adecide_provision_starts(
|
|
26
|
+
texts: list[str], *, decider: Optional[BoundaryDecider] = None
|
|
27
|
+
) -> list[bool]:
|
|
28
|
+
"""Per-span "starts a new provision?" flags, aligned to ``texts``. Deterministic ``start``/``continue`` are
|
|
29
|
+
decided for free; only the ``uncertain`` residue is passed to ``decider`` (one batched call). No decider, or a
|
|
30
|
+
decider error, degrades the residue to False (fold in) -- never an exception, never a lost span."""
|
|
31
|
+
verdicts = [provision_boundary_verdict(t) for t in texts]
|
|
32
|
+
starts = [v == "start" for v in verdicts]
|
|
33
|
+
residue = [i for i, v in enumerate(verdicts) if v == "uncertain"]
|
|
34
|
+
if residue and decider is not None:
|
|
35
|
+
try:
|
|
36
|
+
decided = await decider([texts[i] for i in residue])
|
|
37
|
+
except Exception: # noqa: BLE001 - a decision-model blip must not fail ingestion; degrade to deterministic
|
|
38
|
+
decided = [False] * len(residue)
|
|
39
|
+
for i, flag in zip(residue, decided):
|
|
40
|
+
starts[i] = bool(flag)
|
|
41
|
+
return starts
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def jev_boundary_decider(resources: Any = None) -> Optional[BoundaryDecider]:
|
|
45
|
+
"""Build the Jev-backed residue decider, or ``None`` when no decision model is configured/registered (then the
|
|
46
|
+
boundary stays purely deterministic). It invokes the ``jev_decision`` capability THROUGH the engine invoker --
|
|
47
|
+
one batched call with all candidate lines in the ``state`` and one ``noul`` question per line."""
|
|
48
|
+
from rag_wright.api import ainvoke_model, capability_index
|
|
49
|
+
from rag_wright.models.profiles import decision_profile
|
|
50
|
+
|
|
51
|
+
model = os.environ.get("RAG_DECISION_MODEL")
|
|
52
|
+
prof = decision_profile(model)
|
|
53
|
+
if not os.environ.get(prof.api_key_env) or "jev_decision" not in capability_index():
|
|
54
|
+
return None # no decision model available -> deterministic-only
|
|
55
|
+
|
|
56
|
+
async def _decide(texts: list[str]) -> list[bool]:
|
|
57
|
+
state = "\n".join(f"[{i}] {t.strip()}" for i, t in enumerate(texts))
|
|
58
|
+
questions = {
|
|
59
|
+
f"c{i}": {
|
|
60
|
+
"type": "noul",
|
|
61
|
+
"instructions": (
|
|
62
|
+
f"In the numbered lines above, does line [{i}] BEGIN a new numbered section or provision of a "
|
|
63
|
+
"contract (a section heading / start), rather than continue the previous provision's text?"
|
|
64
|
+
),
|
|
65
|
+
"criteria": {
|
|
66
|
+
"true": f"line [{i}] begins a new section or provision",
|
|
67
|
+
"false": f"line [{i}] continues the current provision",
|
|
68
|
+
},
|
|
69
|
+
}
|
|
70
|
+
for i in range(len(texts))
|
|
71
|
+
}
|
|
72
|
+
out = await ainvoke_model(
|
|
73
|
+
"jev_decision", {"state": state, "questions": questions, "model": model}, resources=resources
|
|
74
|
+
)
|
|
75
|
+
answers = out.get("answers", {}) if isinstance(out, dict) else {}
|
|
76
|
+
return [float((answers.get(f"c{i}") or {}).get("noul", 0.0)) >= _THRESHOLD for i in range(len(texts))]
|
|
77
|
+
|
|
78
|
+
return _decide
|