rag-wright 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- rag_wright/__init__.py +13 -0
- rag_wright/api/__init__.py +33 -0
- rag_wright/api/config.py +59 -0
- rag_wright/api/discover.py +70 -0
- rag_wright/api/documents.py +39 -0
- rag_wright/api/ids.py +31 -0
- rag_wright/api/invoke.py +99 -0
- rag_wright/api/kg.py +61 -0
- rag_wright/api/mcp.py +94 -0
- rag_wright/api/usage.py +30 -0
- rag_wright/api/workspace.py +85 -0
- rag_wright/capabilities/__init__.py +8 -0
- rag_wright/capabilities/answer_generator.py +427 -0
- rag_wright/capabilities/ard.py +286 -0
- rag_wright/capabilities/assertion_extraction.py +79 -0
- rag_wright/capabilities/chunk_read.py +58 -0
- rag_wright/capabilities/chunk_write.py +163 -0
- rag_wright/capabilities/claim_extraction.py +153 -0
- rag_wright/capabilities/clause_exception_linking.py +117 -0
- rag_wright/capabilities/compliance_judgment.py +322 -0
- rag_wright/capabilities/compliance_store.py +87 -0
- rag_wright/capabilities/contract_kg_serve.py +156 -0
- rag_wright/capabilities/contract_kg_store.py +251 -0
- rag_wright/capabilities/dg_extraction.py +585 -0
- rag_wright/capabilities/disambiguation.py +163 -0
- rag_wright/capabilities/document_parse.py +87 -0
- rag_wright/capabilities/document_scope.py +49 -0
- rag_wright/capabilities/embedding.py +164 -0
- rag_wright/capabilities/embedding_profiles.py +43 -0
- rag_wright/capabilities/entity_resolution.py +154 -0
- rag_wright/capabilities/fusion.py +64 -0
- rag_wright/capabilities/graph_extraction.py +243 -0
- rag_wright/capabilities/graph_query.py +73 -0
- rag_wright/capabilities/graph_storage.py +111 -0
- rag_wright/capabilities/highlight_serve.py +142 -0
- rag_wright/capabilities/hybrid_search.py +65 -0
- rag_wright/capabilities/invoke.py +31 -0
- rag_wright/capabilities/jev_decision.py +38 -0
- rag_wright/capabilities/manifests.py +872 -0
- rag_wright/capabilities/okf_navigate.py +456 -0
- rag_wright/capabilities/parsing.py +286 -0
- rag_wright/capabilities/property_boosted_retrieval.py +125 -0
- rag_wright/capabilities/query_function_classifier.py +94 -0
- rag_wright/capabilities/query_understanding.py +109 -0
- rag_wright/capabilities/registry.py +262 -0
- rag_wright/capabilities/remote_encoders.py +94 -0
- rag_wright/capabilities/requirement_extraction.py +247 -0
- rag_wright/capabilities/reranking.py +123 -0
- rag_wright/capabilities/retrieval_core.py +126 -0
- rag_wright/capabilities/rlm_chunking.py +808 -0
- rag_wright/capabilities/rlm_synthesis.py +316 -0
- rag_wright/capabilities/scan_quality.py +136 -0
- rag_wright/capabilities/span_relevance_judgment.py +191 -0
- rag_wright/capabilities/vision_to_text.py +85 -0
- rag_wright/capabilities/vlm_ocr.py +85 -0
- rag_wright/contracts/__init__.py +6 -0
- rag_wright/contracts/chunk.py +79 -0
- rag_wright/contracts/compliance.py +303 -0
- rag_wright/contracts/contract_meta.py +27 -0
- rag_wright/contracts/extraction.py +130 -0
- rag_wright/contracts/function.py +167 -0
- rag_wright/contracts/function_routing.py +91 -0
- rag_wright/contracts/highlight.py +74 -0
- rag_wright/contracts/identifiers.py +153 -0
- rag_wright/contracts/jurisdiction.py +96 -0
- rag_wright/contracts/ontology.py +142 -0
- rag_wright/contracts/property.py +201 -0
- rag_wright/contracts/provenance.py +78 -0
- rag_wright/contracts/query_intent.py +53 -0
- rag_wright/contracts/span.py +76 -0
- rag_wright/contracts/value_match.py +84 -0
- rag_wright/corpus/__init__.py +0 -0
- rag_wright/corpus/canonicalize.py +116 -0
- rag_wright/corpus/cuad.py +153 -0
- rag_wright/corpus/cuad_ingestion.py +72 -0
- rag_wright/corpus/document_parser.py +299 -0
- rag_wright/corpus/edgar.py +231 -0
- rag_wright/corpus/gcs_ingestion.py +120 -0
- rag_wright/corpus/http.py +110 -0
- rag_wright/corpus/selection.py +152 -0
- rag_wright/mcp/__init__.py +11 -0
- rag_wright/mcp/compliance_server.py +299 -0
- rag_wright/mcp/intra_document_qa_server.py +170 -0
- rag_wright/mcp/relational_qa_server.py +171 -0
- rag_wright/mcp/session_store.py +64 -0
- rag_wright/mcp/typed_property_retrieval_server.py +191 -0
- rag_wright/models/__init__.py +8 -0
- rag_wright/models/profiles.py +331 -0
- rag_wright/models/seam.py +497 -0
- rag_wright/models/tag_structured.py +285 -0
- rag_wright/models/tracing.py +179 -0
- rag_wright/models/usage.py +102 -0
- rag_wright/okf/__init__.py +11 -0
- rag_wright/okf/compile.py +292 -0
- rag_wright/okf/document.py +47 -0
- rag_wright/okf/enrich.py +176 -0
- rag_wright/okf/links.py +190 -0
- rag_wright/okf/lint.py +105 -0
- rag_wright/ontology/__init__.py +6 -0
- rag_wright/ontology/_generated_template_meta.py +60 -0
- rag_wright/ontology/_generated_vocab.py +52 -0
- rag_wright/ontology/clause_template.py +964 -0
- rag_wright/ontology/codegen.py +84 -0
- rag_wright/ontology/compliance_bridge.ttl +186 -0
- rag_wright/ontology/contract_bridge.ttl +2685 -0
- rag_wright/ontology/contract_taxonomy.py +24 -0
- rag_wright/ontology/derive.py +58 -0
- rag_wright/ontology/loader.py +435 -0
- rag_wright/ontology/packs/ftc_16cfr255.ttl +29 -0
- rag_wright/ontology/registry.py +87 -0
- rag_wright/ontology/template_introspect.py +100 -0
- rag_wright/py.typed +0 -0
- rag_wright/reference/__init__.py +2 -0
- rag_wright/reference/compliance.py +41 -0
- rag_wright/reference/contract_seam.py +123 -0
- rag_wright/skills/__init__.py +7 -0
- rag_wright/skills/claim_extraction/SKILL.md +47 -0
- rag_wright/skills/claim_extraction/__init__.py +1 -0
- rag_wright/skills/claim_extraction/template.py +50 -0
- rag_wright/skills/compliance_judgment/SKILL.md +59 -0
- rag_wright/skills/corpus_ingest/SKILL.md +106 -0
- rag_wright/skills/extraction_semantic_judge/SKILL.md +51 -0
- rag_wright/skills/extraction_semantic_judge/__init__.py +1 -0
- rag_wright/skills/generation/SKILL.md +64 -0
- rag_wright/skills/generation/__init__.py +1 -0
- rag_wright/skills/generic_compliance_judgment/SKILL.md +58 -0
- rag_wright/skills/okf_navigate/SKILL.md +137 -0
- rag_wright/skills/requirement_extraction/SKILL.md +47 -0
- rag_wright/skills/requirement_extraction/__init__.py +1 -0
- rag_wright/skills/requirement_extraction/template.py +50 -0
- rag_wright/skills/rlm/SKILL.md +186 -0
- rag_wright/skills/rlm/__init__.py +31 -0
- rag_wright/skills/rlm/agent.py +292 -0
- rag_wright/skills/span_relevance_judgment/SKILL.md +67 -0
- rag_wright/skills/vision_to_text/SKILL.md +36 -0
- rag_wright/skills/vision_to_text/__init__.py +1 -0
- rag_wright/spans/__init__.py +1 -0
- rag_wright/spans/boundary.py +78 -0
- rag_wright/spans/clause_function_classifier.py +490 -0
- rag_wright/spans/clause_kg_extractor.py +337 -0
- rag_wright/spans/cuad_labels.py +81 -0
- rag_wright/spans/dim_classifier.py +158 -0
- rag_wright/spans/dim_fleet.json +411 -0
- rag_wright/spans/function_classifier.py +77 -0
- rag_wright/spans/function_families.py +62 -0
- rag_wright/spans/hybrid_classifier.py +103 -0
- rag_wright/spans/legalbert_classifier.py +83 -0
- rag_wright/spans/model_capabilities.py +107 -0
- rag_wright/spans/new_function_labels.py +111 -0
- rag_wright/spans/page_map.py +68 -0
- rag_wright/spans/property_extractor.py +365 -0
- rag_wright/spans/property_grounding.py +182 -0
- rag_wright/spans/reclassify.py +77 -0
- rag_wright/spans/scarce_function_labels.py +105 -0
- rag_wright/spans/segment.py +341 -0
- rag_wright/spans/semantic_judge.py +197 -0
- rag_wright/spans/symbolic_validation.py +131 -0
- rag_wright/spans/tag_clause_extractor.py +182 -0
- rag_wright/store/__init__.py +6 -0
- rag_wright/store/arcadedb.py +1135 -0
- rag_wright/store/chunk_text.py +66 -0
- rag_wright/store/seam.py +213 -0
- rag_wright/subgraphs/__init__.py +0 -0
- rag_wright/subgraphs/async_ingestion.py +204 -0
- rag_wright/subgraphs/compliance_check.py +1042 -0
- rag_wright/subgraphs/compliance_ingestion.py +306 -0
- rag_wright/subgraphs/contract_ingestion_pipeline.py +999 -0
- rag_wright/subgraphs/graph_extraction.py +102 -0
- rag_wright/subgraphs/intra_document_qa.py +328 -0
- rag_wright/subgraphs/observability.py +140 -0
- rag_wright/subgraphs/query_constraint_extraction.py +73 -0
- rag_wright/subgraphs/relational_qa.py +165 -0
- rag_wright/subgraphs/requirement_extraction.py +137 -0
- rag_wright/subgraphs/scaffold.py +65 -0
- rag_wright/subgraphs/semantic_chunking.py +183 -0
- rag_wright/subgraphs/typed_clause_extraction.py +172 -0
- rag_wright/subgraphs/typed_property_retrieval.py +278 -0
- rag_wright/util/__init__.py +1 -0
- rag_wright/util/concurrent.py +153 -0
- rag_wright/util/spacy_model.py +45 -0
- rag_wright-0.1.0.dist-info/METADATA +168 -0
- rag_wright-0.1.0.dist-info/RECORD +184 -0
- rag_wright-0.1.0.dist-info/WHEEL +4 -0
- rag_wright-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,316 @@
|
|
|
1
|
+
"""RLM synthesis (FR-Q.5, T28): apply the recursive RLM method to the retrieved candidate chunks.
|
|
2
|
+
|
|
3
|
+
The query-side RLM. The rebuild (ADR-0015/0016) makes synthesis genuinely recursive dynamic sub-agents,
|
|
4
|
+
in two halves:
|
|
5
|
+
|
|
6
|
+
- DESCENT (new, recursive): a `SliceExtractor` decomposes the candidate set through the T15 machinery
|
|
7
|
+
(`build_rlm_agent`) — the interpreter holds the candidate chunks, a fresh `rlm_decomposer` splits an
|
|
8
|
+
over-large group, and an `rlm_slice_worker` extracts the query-relevant facts from each leaf slice
|
|
9
|
+
(per-slice tool use and per-slice skills live in the worker). A model is only ever called on a focused
|
|
10
|
+
slice, never over the full candidate volume.
|
|
11
|
+
- ASCENT (kept, ADR-0016): the Python `_reduce` fan-in combines the per-slice extracts into the final
|
|
12
|
+
synthesis, recursively (each combine sees at most `fanout` notes), so a model is never called over the
|
|
13
|
+
whole set of extracts either.
|
|
14
|
+
|
|
15
|
+
Unlike chunking, recursion IS gated for synthesis (ADR-0019): the ADR-0016 fail-if-absent recursion
|
|
16
|
+
discipline applies. The extractor and the combine both sit behind seams so the ascent is tested
|
|
17
|
+
hermetically with stubs and the descent machinery is driven by scripted fake models; the live extract +
|
|
18
|
+
synthesize is opt-in. No claim leaves without a citation: every `SliceOutput` carries its `chunk_id`
|
|
19
|
+
(FR-Q.6).
|
|
20
|
+
"""
|
|
21
|
+
|
|
22
|
+
from __future__ import annotations
|
|
23
|
+
|
|
24
|
+
import asyncio
|
|
25
|
+
import json
|
|
26
|
+
from typing import Optional, Protocol, runtime_checkable
|
|
27
|
+
|
|
28
|
+
from langchain_core.messages import AIMessage, HumanMessage, ToolMessage
|
|
29
|
+
from langchain_core.tools import tool
|
|
30
|
+
from pydantic import BaseModel
|
|
31
|
+
|
|
32
|
+
from rag_wright.capabilities.registry import CapabilityRegistry
|
|
33
|
+
from rag_wright.models.profiles import ModelRole, model_for
|
|
34
|
+
from rag_wright.models.seam import build_model
|
|
35
|
+
from rag_wright.skills.rlm.agent import (
|
|
36
|
+
RLM_DECOMPOSER,
|
|
37
|
+
RLM_SLICE_WORKER,
|
|
38
|
+
build_rlm_agent,
|
|
39
|
+
rlm_interpreter_session,
|
|
40
|
+
)
|
|
41
|
+
|
|
42
|
+
__all__ = [
|
|
43
|
+
"RLM_DECOMPOSER",
|
|
44
|
+
"RLM_SLICE_WORKER",
|
|
45
|
+
"SynthesisChunk",
|
|
46
|
+
"SliceOutput",
|
|
47
|
+
"SynthesisResult",
|
|
48
|
+
"Synthesizer",
|
|
49
|
+
"SeamSynthesizer",
|
|
50
|
+
"SliceExtractor",
|
|
51
|
+
"SeamSliceExtractor",
|
|
52
|
+
"rlm_synthesize",
|
|
53
|
+
"register_rlm_synthesis",
|
|
54
|
+
]
|
|
55
|
+
|
|
56
|
+
DEFAULT_REDUCE_CONCURRENCY = 8 # in-flight combine calls (backpressure); network-bound
|
|
57
|
+
DEFAULT_FANOUT = 8 # code-side reduce fan-in: a combine call sees at most this many notes at once
|
|
58
|
+
|
|
59
|
+
_EXTRACT_WORKER_PROMPT = (
|
|
60
|
+
"You handle ONE slice of retrieved candidate passages. Extract only the facts in this slice that help "
|
|
61
|
+
"answer the question, with any figures and named entities, faithfully and concisely, and keep each "
|
|
62
|
+
"fact tied to the chunk_id it came from. If the slice is irrelevant, say so briefly. You never see the "
|
|
63
|
+
"whole candidate set."
|
|
64
|
+
)
|
|
65
|
+
_COMBINE_PROMPT = (
|
|
66
|
+
"Combine these notes into a single faithful synthesis that answers the question, keeping figures and "
|
|
67
|
+
"named entities. Do not add facts not present in the notes."
|
|
68
|
+
)
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
class SynthesisChunk(BaseModel):
|
|
72
|
+
"""A candidate chunk to synthesize over: its id and full text (rehydrated by `chunk_read`, T38,
|
|
73
|
+
from the chunk-text sidecar the ingest write leg persisted, `store/chunk_text.py`, T40)."""
|
|
74
|
+
|
|
75
|
+
chunk_id: str
|
|
76
|
+
text: str
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
class SliceOutput(BaseModel):
|
|
80
|
+
"""One slice's focused extract (the divide step), tied to its chunk_id (no claim without a citation)."""
|
|
81
|
+
|
|
82
|
+
chunk_id: str
|
|
83
|
+
extract: str
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
class SynthesisResult(BaseModel):
|
|
87
|
+
"""The RLM synthesis output: per-slice extracts, the combined synthesis, and the cited chunk_ids."""
|
|
88
|
+
|
|
89
|
+
query: str
|
|
90
|
+
slice_outputs: list[SliceOutput]
|
|
91
|
+
synthesis: str
|
|
92
|
+
chunk_ids: list[str]
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
# --- the ascent: the kept Python `_reduce` fan-in (ADR-0016) --------------------------------------
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
@runtime_checkable
|
|
99
|
+
class Synthesizer(Protocol):
|
|
100
|
+
"""The combine seam: fold a small group of already-reduced notes (<= fanout) into one synthesis."""
|
|
101
|
+
|
|
102
|
+
def combine(self, query: str, extracts: list[str]) -> str: ...
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
class SeamSynthesizer:
|
|
106
|
+
"""The real combine: a free-text call through the model-profile seam (DeepSeek V4 Pro, ADR-0006)."""
|
|
107
|
+
|
|
108
|
+
def __init__(self, model_id: str | None = None) -> None:
|
|
109
|
+
self._model_id = model_id or model_for(ModelRole.STRUCTURED_REASONING)
|
|
110
|
+
|
|
111
|
+
def combine(self, query: str, extracts: list[str]) -> str:
|
|
112
|
+
notes = "\n\n---\n\n".join(extracts)
|
|
113
|
+
message = build_model(self._model_id).invoke(f"{_COMBINE_PROMPT}\nQuestion: {query}\n\nNotes:\n{notes}")
|
|
114
|
+
return message.content if hasattr(message, "content") else str(message)
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
async def _reduce(
|
|
118
|
+
query: str, extracts: list[str], synthesizer: Synthesizer, semaphore: asyncio.Semaphore, fanout: int
|
|
119
|
+
) -> str:
|
|
120
|
+
"""Fan-in reduce in code: combine at most `fanout` notes per call, recursing on the results so a
|
|
121
|
+
model is never called over the whole set. Groups at one level are combined concurrently. Kept from
|
|
122
|
+
the pre-rebuild implementation as the synthesis combine step (the ascent complements the descent)."""
|
|
123
|
+
if not extracts:
|
|
124
|
+
return ""
|
|
125
|
+
if len(extracts) <= fanout:
|
|
126
|
+
async with semaphore:
|
|
127
|
+
return await asyncio.to_thread(synthesizer.combine, query, extracts)
|
|
128
|
+
groups = [extracts[i : i + fanout] for i in range(0, len(extracts), fanout)]
|
|
129
|
+
|
|
130
|
+
async def _combine(group: list[str]) -> str:
|
|
131
|
+
async with semaphore:
|
|
132
|
+
return await asyncio.to_thread(synthesizer.combine, query, group)
|
|
133
|
+
|
|
134
|
+
partials = list(await asyncio.gather(*(_combine(group) for group in groups)))
|
|
135
|
+
return await _reduce(query, partials, synthesizer, semaphore, fanout)
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
# --- the descent: the recursive RLM extractor (the T15 machinery) ---------------------------------
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
@runtime_checkable
|
|
142
|
+
class SliceExtractor(Protocol):
|
|
143
|
+
"""The recursive-descent seam: decompose the candidate set and extract query-relevant facts per leaf."""
|
|
144
|
+
|
|
145
|
+
def extract(self, query: str, chunks: list[SynthesisChunk]) -> list[SliceOutput]: ...
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
def _final_text(messages) -> str:
|
|
149
|
+
for message in reversed(messages):
|
|
150
|
+
if isinstance(message, AIMessage) and (message.text or "").strip():
|
|
151
|
+
return message.text
|
|
152
|
+
for message in reversed(messages):
|
|
153
|
+
if isinstance(message, ToolMessage) and message.name == "eval":
|
|
154
|
+
return str(message.content)
|
|
155
|
+
return ""
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
def _parse_slice_outputs(text: str) -> list[SliceOutput]:
|
|
159
|
+
"""Parse the JSON array of {chunk_id, extract} objects from the model's final output.
|
|
160
|
+
|
|
161
|
+
Scans every ``[`` and JSON-`raw_decode`s from it, keeping the LONGEST array whose elements are all
|
|
162
|
+
``{chunk_id, extract}`` objects. Bracket-matching from the end (rfind) is wrong: real legal clauses put
|
|
163
|
+
``[`` inside the extract text (e.g. "as set out in [Section 5]"), so the last ``[`` before the closing
|
|
164
|
+
``]`` lands inside a string and the slice is invalid JSON — throwing a valid answer away. A ``[`` inside
|
|
165
|
+
a string cannot itself decode into a citation array, so scanning-and-validating cannot be fooled by it.
|
|
166
|
+
Returns ``[]`` when no valid citation array is present (never raises).
|
|
167
|
+
"""
|
|
168
|
+
decoder = json.JSONDecoder()
|
|
169
|
+
best: list[SliceOutput] = []
|
|
170
|
+
for i, ch in enumerate(text):
|
|
171
|
+
if ch != "[":
|
|
172
|
+
continue
|
|
173
|
+
try:
|
|
174
|
+
value, _ = decoder.raw_decode(text, i)
|
|
175
|
+
except json.JSONDecodeError:
|
|
176
|
+
continue
|
|
177
|
+
if not isinstance(value, list) or not value:
|
|
178
|
+
continue
|
|
179
|
+
if not all(isinstance(o, dict) and "chunk_id" in o and "extract" in o for o in value):
|
|
180
|
+
continue
|
|
181
|
+
if len(value) > len(best):
|
|
182
|
+
best = [SliceOutput(chunk_id=str(o["chunk_id"]), extract=str(o["extract"])) for o in value]
|
|
183
|
+
return best
|
|
184
|
+
|
|
185
|
+
|
|
186
|
+
class SeamSliceExtractor:
|
|
187
|
+
"""The real extractor: the candidate set is decomposed recursively via `build_rlm_agent` and each leaf
|
|
188
|
+
slice is extracted by an `rlm_slice_worker`. Per-role models resolve through the profile seam
|
|
189
|
+
(STRUCTURED_REASONING / DeepSeek V4 Pro) or are injected as instances (tests). `working_set` overrides
|
|
190
|
+
the candidate view handed to the orchestrator (tests use an opaque handle to force recursion)."""
|
|
191
|
+
|
|
192
|
+
def __init__(
|
|
193
|
+
self,
|
|
194
|
+
model: object = None,
|
|
195
|
+
*,
|
|
196
|
+
decomposer_model: object = None,
|
|
197
|
+
worker_model: object = None,
|
|
198
|
+
worker_tools=(),
|
|
199
|
+
worker_skills=(),
|
|
200
|
+
working_set: object = None,
|
|
201
|
+
) -> None:
|
|
202
|
+
self._model = model
|
|
203
|
+
self._decomposer_model = decomposer_model
|
|
204
|
+
self._worker_model = worker_model
|
|
205
|
+
self._worker_tools = worker_tools
|
|
206
|
+
self._worker_skills = worker_skills
|
|
207
|
+
self._working_set = working_set
|
|
208
|
+
|
|
209
|
+
def _build_agent(self, chunks: list[SynthesisChunk], query: str, *, interpreter=None):
|
|
210
|
+
model = self._model if self._model is not None else model_for(ModelRole.STRUCTURED_REASONING)
|
|
211
|
+
# Enforce the query reaching every worker: bake it into the worker's system prompt (T42). The leaf
|
|
212
|
+
# dispatch delivers the SLICE (the workflow threads `items`); the capability delivers the QUERY here,
|
|
213
|
+
# so query-relevant extraction no longer depends on the orchestrator choosing to thread it.
|
|
214
|
+
worker_prompt = (
|
|
215
|
+
f"{_EXTRACT_WORKER_PROMPT}\n\n"
|
|
216
|
+
f"The question to answer (extract facts relevant to THIS, verbatim):\n{query}"
|
|
217
|
+
)
|
|
218
|
+
return build_rlm_agent(
|
|
219
|
+
reasoning_model=model,
|
|
220
|
+
decomposer_model=self._decomposer_model if self._decomposer_model is not None else model,
|
|
221
|
+
worker_model=self._worker_model if self._worker_model is not None else model,
|
|
222
|
+
worker_system_prompt=worker_prompt,
|
|
223
|
+
worker_tools=self._worker_tools,
|
|
224
|
+
worker_skills=self._worker_skills,
|
|
225
|
+
interpreter=interpreter,
|
|
226
|
+
)
|
|
227
|
+
|
|
228
|
+
def _working_set_ptc(self, chunks: list[SynthesisChunk]):
|
|
229
|
+
value = self._working_set if self._working_set is not None else [
|
|
230
|
+
{"id": c.chunk_id, "chunk_id": c.chunk_id, "text": c.text} for c in chunks
|
|
231
|
+
]
|
|
232
|
+
delivered = len(value)
|
|
233
|
+
|
|
234
|
+
@tool
|
|
235
|
+
def working_set() -> object:
|
|
236
|
+
"""Return the working set: the retrieved candidate chunks (id, chunk_id, text)."""
|
|
237
|
+
return value
|
|
238
|
+
|
|
239
|
+
@tool
|
|
240
|
+
def working_set_size() -> int:
|
|
241
|
+
"""Return the number of items in the delivered working set (a truthful count the workflow's
|
|
242
|
+
load-completeness assertion checks against, so an under-read fails loud, not silent)."""
|
|
243
|
+
return delivered
|
|
244
|
+
|
|
245
|
+
return [working_set, working_set_size]
|
|
246
|
+
|
|
247
|
+
def _request(self, query: str) -> str:
|
|
248
|
+
return (
|
|
249
|
+
"Run this as a workflow. Call `const workingSet = await tools.workingSet();` to get the "
|
|
250
|
+
"retrieved candidate set (a list of {id, chunk_id, text}); it is a JavaScript value in the "
|
|
251
|
+
"interpreter, never in your context. Decompose it (dispatch rlm_decomposer with `cuts` for an "
|
|
252
|
+
"over-large group and recurse), hand each leaf slice to an rlm_slice_worker that extracts the "
|
|
253
|
+
"facts relevant to the question, keeping each fact tied to its chunk_id. Apply the coverage "
|
|
254
|
+
"tail from your instructions: before returning, cover any candidate the recursion missed. Then "
|
|
255
|
+
"return ONLY a JSON array [{\"chunk_id\": ..., \"extract\": ...}], one entry per candidate.\n\n"
|
|
256
|
+
f"Question: {query}"
|
|
257
|
+
)
|
|
258
|
+
|
|
259
|
+
def extract(self, query: str, chunks: list[SynthesisChunk]) -> list[SliceOutput]:
|
|
260
|
+
if not chunks:
|
|
261
|
+
return []
|
|
262
|
+
request = self._request(query)
|
|
263
|
+
# The candidate set is delivered as a PTC value (`tools.workingSet()`, T36): it stays in the
|
|
264
|
+
# interpreter and never enters the model's context. The interpreter session is serialized
|
|
265
|
+
# process-wide (KI-1, ADR-0020): build + run + teardown inside the lock.
|
|
266
|
+
with rlm_interpreter_session(ptc=self._working_set_ptc(chunks)) as interpreter:
|
|
267
|
+
agent = self._build_agent(chunks, query, interpreter=interpreter)
|
|
268
|
+
messages = agent.invoke({"messages": [HumanMessage(content=request)]})["messages"]
|
|
269
|
+
return _parse_slice_outputs(_final_text(messages))
|
|
270
|
+
|
|
271
|
+
|
|
272
|
+
# --- the capability: descent then ascent ---------------------------------------------------------
|
|
273
|
+
|
|
274
|
+
|
|
275
|
+
def rlm_synthesize(
|
|
276
|
+
query: str,
|
|
277
|
+
chunks: list[SynthesisChunk],
|
|
278
|
+
*,
|
|
279
|
+
extractor: Optional[SliceExtractor] = None,
|
|
280
|
+
synthesizer: Optional[Synthesizer] = None,
|
|
281
|
+
max_concurrency: int = DEFAULT_REDUCE_CONCURRENCY,
|
|
282
|
+
fanout: int = DEFAULT_FANOUT,
|
|
283
|
+
) -> SynthesisResult:
|
|
284
|
+
"""Synthesize an answer over the candidate chunks: recursive extract (descent) then reduce (ascent).
|
|
285
|
+
|
|
286
|
+
`extractor` decomposes the candidate set and extracts per leaf, with the coverage guarantee **in the
|
|
287
|
+
interpreter workflow** (the skill's coverage tail covers any candidate the recursion missed, so it runs
|
|
288
|
+
in every consumer including GraphWright's node, not just here — T37); `_reduce` combines the extracts
|
|
289
|
+
via `synthesizer` (defaults to `SeamSynthesizer`). Every extract keeps its `chunk_id`, cited (FR-Q.6).
|
|
290
|
+
"""
|
|
291
|
+
if not chunks:
|
|
292
|
+
return SynthesisResult(query=query, slice_outputs=[], synthesis="", chunk_ids=[])
|
|
293
|
+
extractor = extractor if extractor is not None else SeamSliceExtractor()
|
|
294
|
+
synthesizer = synthesizer if synthesizer is not None else SeamSynthesizer()
|
|
295
|
+
|
|
296
|
+
slice_outputs = extractor.extract(query, chunks)
|
|
297
|
+
semaphore = asyncio.Semaphore(max_concurrency)
|
|
298
|
+
synthesis = asyncio.run(
|
|
299
|
+
_reduce(query, [o.extract for o in slice_outputs], synthesizer, semaphore, fanout)
|
|
300
|
+
)
|
|
301
|
+
return SynthesisResult(
|
|
302
|
+
query=query,
|
|
303
|
+
slice_outputs=slice_outputs,
|
|
304
|
+
synthesis=synthesis,
|
|
305
|
+
chunk_ids=[chunk.chunk_id for chunk in chunks],
|
|
306
|
+
)
|
|
307
|
+
|
|
308
|
+
|
|
309
|
+
def register_rlm_synthesis(registry: CapabilityRegistry) -> None:
|
|
310
|
+
"""Register RLM synthesis under FR-Q.5 (`rlm_synthesis`, an `agent_skill` applying the RLM method)."""
|
|
311
|
+
registry.register(
|
|
312
|
+
"rlm_synthesis",
|
|
313
|
+
contract=SynthesisResult,
|
|
314
|
+
kind="agent_skill",
|
|
315
|
+
display_name="RLM synthesis",
|
|
316
|
+
)
|
|
@@ -0,0 +1,136 @@
|
|
|
1
|
+
"""Issue 0009-GATE: the scan-quality gate.
|
|
2
|
+
|
|
3
|
+
Decide whether a scanned page's OCR is trustworthy, so the tiered OCR path can react: a DEGRADED page is
|
|
4
|
+
escalated to a VLM (which reads blurred/faded text by language context), and a genuinely UNREADABLE page is
|
|
5
|
+
flagged PARTIAL / needs-rescan rather than ingested as gibberish (the 0006-C / ENG-1 lossless principle applied
|
|
6
|
+
to OCR).
|
|
7
|
+
|
|
8
|
+
Signals (any subset; the strongest DEGRADED/UNREADABLE verdict wins):
|
|
9
|
+
- `text_readability` -- common-word hit rate of the OCR text. Garbage OCR ("upareaia aes jo sa ...") almost
|
|
10
|
+
never hits common English words; real prose is dense with them. The strongest post-OCR signal, no dep.
|
|
11
|
+
- `image_quality` -- Laplacian variance (blur) + dark_frac (faint ink), pre-OCR, via OpenCV. Predicts an
|
|
12
|
+
unreadable page before OCR is even run.
|
|
13
|
+
- docling `confidence` -- the parser's own per-page `PageConfidenceScores` (0..1), when available.
|
|
14
|
+
|
|
15
|
+
Thresholds are the ones separated in the OCR benchmark (docs/eval/ocr_benchmark.md): clean/moderate scans read
|
|
16
|
+
at word-hit ~0.2-0.4, Laplacian 2440/549, dark_frac 0.03/0.028; the heavy scan that defeats all OCR sits at
|
|
17
|
+
word-hit ~0, Laplacian 54, dark_frac 0.003.
|
|
18
|
+
"""
|
|
19
|
+
from __future__ import annotations
|
|
20
|
+
|
|
21
|
+
import re
|
|
22
|
+
from enum import Enum
|
|
23
|
+
|
|
24
|
+
from pydantic import BaseModel
|
|
25
|
+
|
|
26
|
+
# validated thresholds (see docs/eval/ocr_benchmark.md)
|
|
27
|
+
_WORD_HIT_DEGRADED = 0.08 # readable text >> this; garbage OCR ~ 0
|
|
28
|
+
_LAPLACIAN_DEGRADED = 150.0 # heavy 54 (blur) vs moderate 549 / clean 2440
|
|
29
|
+
_DARK_FRAC_FAINT = 0.008 # heavy 0.003 (almost no ink) vs ~0.03 readable
|
|
30
|
+
_CONFIDENCE_DEGRADED = 0.5 # docling PageConfidenceScores below this -> distrust
|
|
31
|
+
|
|
32
|
+
# a small, dependency-free set of the most common English words -- their hit rate in OCR text cleanly separates
|
|
33
|
+
# real prose (dense with these) from OCR garbage (almost none). Deliberately generic, not domain-specific.
|
|
34
|
+
_COMMON_WORDS = frozenset((
|
|
35
|
+
"the of and to a in that is was he for it with as his on be at by i this had not are but from or have an "
|
|
36
|
+
"they which one you were her all she there would their we him been has when who will more no if out so said "
|
|
37
|
+
"what up its about into than them can only other new some could time these two may then do first any my now "
|
|
38
|
+
"such like our over me after also did many shall not any such under upon herein hereof any all each party "
|
|
39
|
+
"parties agreement pursuant provided further including without between whether"
|
|
40
|
+
).split())
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
class ScanQuality(str, Enum):
|
|
44
|
+
READABLE = "readable" # trust the OCR text
|
|
45
|
+
DEGRADED = "degraded" # low-quality -> escalate to a VLM
|
|
46
|
+
UNREADABLE = "unreadable" # nothing recovered -> flag PARTIAL / needs-rescan
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
class ScanAssessment(BaseModel):
|
|
50
|
+
quality: ScanQuality
|
|
51
|
+
reason: str
|
|
52
|
+
word_hit_rate: float | None = None
|
|
53
|
+
laplacian_var: float | None = None
|
|
54
|
+
dark_frac: float | None = None
|
|
55
|
+
confidence: float | None = None
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def text_readability(text: str) -> float:
|
|
59
|
+
"""Fraction of alphabetic OCR tokens that are common English words (0..1). ~0 for OCR garbage."""
|
|
60
|
+
toks = re.findall(r"[A-Za-z]{2,}", (text or "").lower())
|
|
61
|
+
if not toks:
|
|
62
|
+
return 0.0
|
|
63
|
+
return sum(t in _COMMON_WORDS for t in toks) / len(toks)
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def image_quality(gray) -> dict:
|
|
67
|
+
"""Pre-OCR page metrics from a grayscale image (numpy 2-D uint8): Laplacian variance (blur; higher = sharper)
|
|
68
|
+
and dark_frac (fraction of ink-dark pixels; a faint/washed-out scan is near zero)."""
|
|
69
|
+
import cv2
|
|
70
|
+
import numpy as np
|
|
71
|
+
|
|
72
|
+
g = np.asarray(gray)
|
|
73
|
+
return {"laplacian_var": float(cv2.Laplacian(g, cv2.CV_64F).var()), "dark_frac": float((g < 128).mean())}
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def assess_scan(text: str | None, *, laplacian_var: float | None = None, dark_frac: float | None = None,
|
|
77
|
+
confidence: float | None = None) -> ScanAssessment:
|
|
78
|
+
"""Classify a page from whatever signals are available. The strongest negative verdict wins; if nothing at
|
|
79
|
+
all is recognised and no other signal is present, the page is UNREADABLE (a blank / failed scan)."""
|
|
80
|
+
whr = text_readability(text) if text is not None else None
|
|
81
|
+
|
|
82
|
+
def _a(q: ScanQuality, reason: str) -> ScanAssessment:
|
|
83
|
+
return ScanAssessment(quality=q, reason=reason, word_hit_rate=whr, laplacian_var=laplacian_var,
|
|
84
|
+
dark_frac=dark_frac, confidence=confidence)
|
|
85
|
+
|
|
86
|
+
# UNREADABLE: an empty/near-empty result with no positive evidence anywhere
|
|
87
|
+
if text is not None and not (text or "").strip() and laplacian_var is None and confidence is None:
|
|
88
|
+
return _a(ScanQuality.UNREADABLE, "no text recognised")
|
|
89
|
+
|
|
90
|
+
# DEGRADED signals (escalate to a VLM)
|
|
91
|
+
if whr is not None and (text or "").strip() and whr < _WORD_HIT_DEGRADED:
|
|
92
|
+
return _a(ScanQuality.DEGRADED, f"low word-hit-rate {whr:.3f} (likely garbage OCR)")
|
|
93
|
+
if laplacian_var is not None and laplacian_var < _LAPLACIAN_DEGRADED:
|
|
94
|
+
return _a(ScanQuality.DEGRADED, f"blurry image (laplacian_var {laplacian_var:.0f})")
|
|
95
|
+
if dark_frac is not None and dark_frac < _DARK_FRAC_FAINT:
|
|
96
|
+
return _a(ScanQuality.DEGRADED, f"faint image (dark_frac {dark_frac:.3f})")
|
|
97
|
+
if confidence is not None and confidence < _CONFIDENCE_DEGRADED:
|
|
98
|
+
return _a(ScanQuality.DEGRADED, f"low OCR confidence {confidence:.2f}")
|
|
99
|
+
|
|
100
|
+
return _a(ScanQuality.READABLE, "ok")
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def _page_texts(doc) -> dict[int, str]:
|
|
104
|
+
"""Group a parsed document's text by page number (from each item's `prov[0].page_no`). Duck-typed on
|
|
105
|
+
`iterate_items()` so it works on a real DoclingDocument or a hermetic fake."""
|
|
106
|
+
pages: dict[int, list[str]] = {}
|
|
107
|
+
for item, _level in doc.iterate_items():
|
|
108
|
+
text = (getattr(item, "text", "") or "").strip()
|
|
109
|
+
if not text:
|
|
110
|
+
continue
|
|
111
|
+
prov = getattr(item, "prov", None) or []
|
|
112
|
+
pg = prov[0].page_no if prov else 1
|
|
113
|
+
pages.setdefault(pg, []).append(text)
|
|
114
|
+
return {pg: "\n".join(v) for pg, v in pages.items()}
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def assess_document(doc, *, page_images: dict | None = None) -> dict[int, ScanAssessment]:
|
|
118
|
+
"""Assess each page of a parsed document -> {page_no: ScanAssessment}. When `page_images` (a {page_no:
|
|
119
|
+
grayscale numpy image} map) is given, the IMAGE metrics (blur/faintness) are folded in alongside the OCR
|
|
120
|
+
text -- the strong signal that separates a degraded scan whose OCR is garbled-but-common-word (which a
|
|
121
|
+
text-only gate misses) from a readable one. A page present only in the image map (no OCR text) is still
|
|
122
|
+
assessed. Pass NO images to re-check VLM output (the image stays blurry after the VLM recovers the text)."""
|
|
123
|
+
page_images = page_images or {}
|
|
124
|
+
texts = _page_texts(doc)
|
|
125
|
+
out: dict[int, ScanAssessment] = {}
|
|
126
|
+
for pg in sorted(set(texts) | set(page_images)):
|
|
127
|
+
img = page_images.get(pg)
|
|
128
|
+
metrics = image_quality(img) if img is not None else {}
|
|
129
|
+
out[pg] = assess_scan(texts.get(pg, ""), **metrics)
|
|
130
|
+
return out
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def register_scan_quality(registry) -> None:
|
|
134
|
+
"""0009-GATE: register `scan_quality` (function; page signals -> a readable/degraded/unreadable verdict)."""
|
|
135
|
+
registry.register("scan_quality", contract=ScanAssessment, kind="function",
|
|
136
|
+
display_name="Scan-quality gate (route degraded scans to VLM; flag unreadable PARTIAL)")
|
|
@@ -0,0 +1,191 @@
|
|
|
1
|
+
"""Engine issue 0023: a per-span RELEVANCE VERDICT on the corpus retrieval path.
|
|
2
|
+
|
|
3
|
+
`typed_property_retrieval` always returns the top-k nearest spans, so `not_found` was unreachable and a nonsense
|
|
4
|
+
query still returned a full page of clauses. No SCORE fixes this: RRF is a relabeling of the row number, cosine's
|
|
5
|
+
distribution moves with model/domain/chunking, and a cross-encoder is a better number but still a number to
|
|
6
|
+
threshold -- every threshold is a corpus-specific magic knob that fails silently. The generic answer is a VERDICT
|
|
7
|
+
(the shape the compliance judge and answer-abstention already use): the engine decides "does this span address
|
|
8
|
+
this condition?" and returns the FACT; the product keeps the matched/possible/not_found grouping (POLICY).
|
|
9
|
+
|
|
10
|
+
SKILL-SPLIT (mirrors `compliance_judgment`):
|
|
11
|
+
- **`span_relevance_judgment` (agent_skill)** -- the LLM relevance METHOD, authored as
|
|
12
|
+
`skills/span_relevance_judgment/SKILL.md`, applied through the model seam. Given one span's text + the
|
|
13
|
+
structured `Condition` (+ the typed properties detected on the span, as CONTEXT not proof), returns a raw
|
|
14
|
+
`RelevanceVerdict`. `build_arelevance_judge_fn` is its runtime; `structured_factory` is injected for tests.
|
|
15
|
+
- The applying capability owns the guarantees the skill does not: the closed verdict vocab and the CONSERVATIVE
|
|
16
|
+
DEFAULT -- an unreadable/failed judgement maps to `uncertain` (recall-safe: a judge failure never fabricates a
|
|
17
|
+
`not_found`; the span stays visible), exactly as the compliance judge defaults to `needs_review`.
|
|
18
|
+
|
|
19
|
+
`ajudge_spans` runs the judge over a retrieved set concurrently (async + semaphore + a wall-clock bound per span),
|
|
20
|
+
so a sweep's wall-clock is concurrency-bound, not count-bound (the parallel-LLM rule).
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
from __future__ import annotations
|
|
24
|
+
|
|
25
|
+
import asyncio
|
|
26
|
+
import os
|
|
27
|
+
from enum import Enum
|
|
28
|
+
from pathlib import Path
|
|
29
|
+
from typing import Awaitable, Callable, Optional
|
|
30
|
+
|
|
31
|
+
from pydantic import BaseModel
|
|
32
|
+
|
|
33
|
+
from rag_wright.models.seam import build_structured
|
|
34
|
+
|
|
35
|
+
_SKILL_PATH = Path(__file__).parents[1] / "skills" / "span_relevance_judgment" / "SKILL.md"
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
class Relevance(str, Enum):
|
|
39
|
+
"""The closed relevance vocab. `uncertain` is both a real judgement (ambiguous text) AND the conservative
|
|
40
|
+
default when the judge could not be read (recall-safe: never a fabricated not_found)."""
|
|
41
|
+
|
|
42
|
+
RELEVANT = "relevant"
|
|
43
|
+
NOT_RELEVANT = "not_relevant"
|
|
44
|
+
UNCERTAIN = "uncertain"
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
_VERDICTS = {v.value for v in Relevance}
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
class Condition(BaseModel):
|
|
51
|
+
"""The structured test a retrieved span is judged against (issue 0023). `clause_type` is primary (a category
|
|
52
|
+
to test membership of); `value_condition` is a narrower test within it (often absent or shared across a
|
|
53
|
+
multi-condition sweep); `question` is CONTEXT ONLY -- in a multi-condition sweep it belongs to all conditions
|
|
54
|
+
at once, so it must not by itself make a span relevant. (Per-condition question decomposition is a separate
|
|
55
|
+
engine gap, not owned here.)"""
|
|
56
|
+
|
|
57
|
+
clause_type: str
|
|
58
|
+
value_condition: Optional[str] = None
|
|
59
|
+
question: Optional[str] = None
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
class RelevanceVerdict(BaseModel):
|
|
63
|
+
"""The raw structured output of one relevance judgement (the `span_relevance_judgment` SKILL's typed output).
|
|
64
|
+
`verdict` is a loose string mapped to the closed `Relevance` vocab by the applying capability (unreadable ->
|
|
65
|
+
uncertain, the conservative default)."""
|
|
66
|
+
|
|
67
|
+
verdict: str
|
|
68
|
+
rationale: str = ""
|
|
69
|
+
confidence: float = 0.0
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
# ajudge_fn: (span_text, matched_constraints, condition) -> RelevanceVerdict. Primitives (not RankedSpan) so this
|
|
73
|
+
# capability stays independent of the retrieval contract -- the subgraph adapts its spans to these inputs.
|
|
74
|
+
AJudgeFn = Callable[[str, list[tuple[str, str]], Condition], Awaitable[RelevanceVerdict]]
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def _skill_body(path: Path) -> str:
|
|
78
|
+
text = path.read_text(encoding="utf-8")
|
|
79
|
+
if text.startswith("---"):
|
|
80
|
+
marker = text.find("\n---", 3)
|
|
81
|
+
if marker != -1:
|
|
82
|
+
text = text[marker + 4 :]
|
|
83
|
+
return text.strip()
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def relevance_method() -> str:
|
|
87
|
+
"""The authored relevance-judgment method (skills/span_relevance_judgment/SKILL.md)."""
|
|
88
|
+
return _skill_body(_SKILL_PATH)
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
_PROMPT_TAIL = (
|
|
92
|
+
"\n\nCONDITION being searched for:\n"
|
|
93
|
+
"- clause type: {clause_type}\n"
|
|
94
|
+
"- specific condition: {value_condition}\n"
|
|
95
|
+
"- user's question (CONTEXT ONLY -- may be shared across several conditions): {question}\n\n"
|
|
96
|
+
"TYPED PROPERTIES already detected on this span (CONTEXT -- extracted from the question and reused across "
|
|
97
|
+
"conditions, so evidence, NOT proof of relevance; judge the span TEXT):\n{matched}\n\n"
|
|
98
|
+
"RETRIEVED SPAN (the only text you may judge):\n{span}"
|
|
99
|
+
)
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def _tail(span_text: str, matched: list[tuple[str, str]], condition: Condition) -> str:
|
|
103
|
+
matched_str = ", ".join(f"{d} = {v}" for d, v in matched) if matched else "(none)"
|
|
104
|
+
return _PROMPT_TAIL.format(
|
|
105
|
+
clause_type=condition.clause_type,
|
|
106
|
+
value_condition=condition.value_condition or "(none -- judge against the clause type)",
|
|
107
|
+
question=condition.question or "(none)",
|
|
108
|
+
matched=matched_str,
|
|
109
|
+
span=span_text,
|
|
110
|
+
)
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def build_arelevance_judge_fn(model_id: str, *, structured_factory=build_structured) -> AJudgeFn:
|
|
114
|
+
"""The `span_relevance_judgment` SKILL runtime: an async relevance judge through the model seam. Given a span's
|
|
115
|
+
text + the typed properties detected on it (context) + the structured condition, returns a raw
|
|
116
|
+
`RelevanceVerdict`. `structured_factory` is injected for hermetic tests."""
|
|
117
|
+
method = relevance_method()
|
|
118
|
+
|
|
119
|
+
async def judge(span_text: str, matched: list[tuple[str, str]], condition: Condition) -> RelevanceVerdict:
|
|
120
|
+
# label names the generation (ADR-0058 / issue 0025) so a reader tells `span-relevance` from
|
|
121
|
+
# `query-constraints` in the cost report; ignored by the hermetic stub factory.
|
|
122
|
+
return await structured_factory(model_id, RelevanceVerdict, label="span-relevance").ainvoke(
|
|
123
|
+
method + _tail(span_text, matched, condition))
|
|
124
|
+
|
|
125
|
+
return judge
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def to_relevance(raw: str) -> Relevance:
|
|
129
|
+
"""Map the LLM's loose verdict string to the closed vocab; unreadable -> uncertain (conservative default:
|
|
130
|
+
recall-safe, so a mis-read never fabricates a not_found)."""
|
|
131
|
+
value = (raw or "").strip().lower().replace("-", "_").replace(" ", "_")
|
|
132
|
+
return Relevance(value) if value in _VERDICTS else Relevance.UNCERTAIN
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def finalize_verdict(raw: Optional[RelevanceVerdict]) -> RelevanceVerdict:
|
|
136
|
+
"""The deterministic guarantees the SKILL does not own (the applying-capability step): map the raw verdict to
|
|
137
|
+
the closed vocab (unreadable/None -> `uncertain`, the conservative recall-safe default) and clamp confidence to
|
|
138
|
+
[0, 1]. Mirrors `compliance_judgment.assemble_finding`'s conservative mapping."""
|
|
139
|
+
if raw is None:
|
|
140
|
+
return RelevanceVerdict(verdict=Relevance.UNCERTAIN.value, rationale="relevance judge did not rule",
|
|
141
|
+
confidence=0.0)
|
|
142
|
+
return RelevanceVerdict(verdict=to_relevance(raw.verdict).value, rationale=raw.rationale,
|
|
143
|
+
confidence=min(1.0, max(0.0, raw.confidence)))
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
_JUDGE_TIMEOUT_S = float(os.environ.get("RAG_RELEVANCE_TIMEOUT_S", "90")) # per-span wall-clock bound
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
async def ajudge_spans(
|
|
150
|
+
spans: list[tuple[str, list[tuple[str, str]]]], condition: Condition, *, ajudge_fn: AJudgeFn,
|
|
151
|
+
max_concurrency: int = 8, timeout_s: float | None = _JUDGE_TIMEOUT_S, timeout_retries: int = 1,
|
|
152
|
+
) -> list[RelevanceVerdict]:
|
|
153
|
+
"""Judge many retrieved spans against one condition CONCURRENTLY (asyncio.gather + Semaphore, the parallel-LLM
|
|
154
|
+
rule), order preserved. Each `spans` item is `(span_text, matched)`. Each judgement is bounded by `timeout_s`
|
|
155
|
+
(a hard wall-clock deadline); on the deadline it is retried up to `timeout_retries` times, then falls back to a
|
|
156
|
+
conservative `uncertain` verdict -- a stalled provider never hangs the sweep. A NON-timeout judge error
|
|
157
|
+
propagates (a genuine bug is never masked). Every returned span gets a verdict: the list is 1:1 with `spans`."""
|
|
158
|
+
semaphore = asyncio.Semaphore(max_concurrency)
|
|
159
|
+
|
|
160
|
+
async def _rule(span_text: str, matched: list[tuple[str, str]]) -> RelevanceVerdict:
|
|
161
|
+
if timeout_s is None:
|
|
162
|
+
return await ajudge_fn(span_text, matched, condition)
|
|
163
|
+
for attempt in range(timeout_retries + 1):
|
|
164
|
+
try:
|
|
165
|
+
async with asyncio.timeout(timeout_s):
|
|
166
|
+
return await ajudge_fn(span_text, matched, condition)
|
|
167
|
+
except (asyncio.TimeoutError, TimeoutError):
|
|
168
|
+
if attempt >= timeout_retries:
|
|
169
|
+
return RelevanceVerdict(verdict=Relevance.UNCERTAIN.value,
|
|
170
|
+
rationale="relevance judge timed out", confidence=0.0)
|
|
171
|
+
return RelevanceVerdict(verdict=Relevance.UNCERTAIN.value, rationale="relevance judge timed out")
|
|
172
|
+
|
|
173
|
+
async def _one(item: tuple[str, list[tuple[str, str]]]) -> RelevanceVerdict:
|
|
174
|
+
span_text, matched = item
|
|
175
|
+
async with semaphore:
|
|
176
|
+
return await _rule(span_text, matched)
|
|
177
|
+
|
|
178
|
+
return list(await asyncio.gather(*(_one(item) for item in spans)))
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
def register_span_relevance_judgment(registry) -> None:
|
|
182
|
+
"""Register `span_relevance_judgment` as an AGENT_SKILL (issue 0023, ADR-0088): a single grounded LLM relevance
|
|
183
|
+
judgement, authored as `skills/span_relevance_judgment/SKILL.md` and applied via the seam. Output =
|
|
184
|
+
`RelevanceVerdict`. Promoted to a canonical FR-C slug (`CANONICAL_CAPABILITY_SLUGS`), the same way every
|
|
185
|
+
post-v0.1 capability (compliance module, KG-primary retrieval core) was added -- registry mirror + ADR."""
|
|
186
|
+
registry.register(
|
|
187
|
+
"span_relevance_judgment",
|
|
188
|
+
contract=RelevanceVerdict,
|
|
189
|
+
kind="agent_skill",
|
|
190
|
+
display_name="Span relevance judgment (span x condition -> verdict; authored skill)",
|
|
191
|
+
)
|