rag-wright 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- rag_wright/__init__.py +13 -0
- rag_wright/api/__init__.py +33 -0
- rag_wright/api/config.py +59 -0
- rag_wright/api/discover.py +70 -0
- rag_wright/api/documents.py +39 -0
- rag_wright/api/ids.py +31 -0
- rag_wright/api/invoke.py +99 -0
- rag_wright/api/kg.py +61 -0
- rag_wright/api/mcp.py +94 -0
- rag_wright/api/usage.py +30 -0
- rag_wright/api/workspace.py +85 -0
- rag_wright/capabilities/__init__.py +8 -0
- rag_wright/capabilities/answer_generator.py +427 -0
- rag_wright/capabilities/ard.py +286 -0
- rag_wright/capabilities/assertion_extraction.py +79 -0
- rag_wright/capabilities/chunk_read.py +58 -0
- rag_wright/capabilities/chunk_write.py +163 -0
- rag_wright/capabilities/claim_extraction.py +153 -0
- rag_wright/capabilities/clause_exception_linking.py +117 -0
- rag_wright/capabilities/compliance_judgment.py +322 -0
- rag_wright/capabilities/compliance_store.py +87 -0
- rag_wright/capabilities/contract_kg_serve.py +156 -0
- rag_wright/capabilities/contract_kg_store.py +251 -0
- rag_wright/capabilities/dg_extraction.py +585 -0
- rag_wright/capabilities/disambiguation.py +163 -0
- rag_wright/capabilities/document_parse.py +87 -0
- rag_wright/capabilities/document_scope.py +49 -0
- rag_wright/capabilities/embedding.py +164 -0
- rag_wright/capabilities/embedding_profiles.py +43 -0
- rag_wright/capabilities/entity_resolution.py +154 -0
- rag_wright/capabilities/fusion.py +64 -0
- rag_wright/capabilities/graph_extraction.py +243 -0
- rag_wright/capabilities/graph_query.py +73 -0
- rag_wright/capabilities/graph_storage.py +111 -0
- rag_wright/capabilities/highlight_serve.py +142 -0
- rag_wright/capabilities/hybrid_search.py +65 -0
- rag_wright/capabilities/invoke.py +31 -0
- rag_wright/capabilities/jev_decision.py +38 -0
- rag_wright/capabilities/manifests.py +872 -0
- rag_wright/capabilities/okf_navigate.py +456 -0
- rag_wright/capabilities/parsing.py +286 -0
- rag_wright/capabilities/property_boosted_retrieval.py +125 -0
- rag_wright/capabilities/query_function_classifier.py +94 -0
- rag_wright/capabilities/query_understanding.py +109 -0
- rag_wright/capabilities/registry.py +262 -0
- rag_wright/capabilities/remote_encoders.py +94 -0
- rag_wright/capabilities/requirement_extraction.py +247 -0
- rag_wright/capabilities/reranking.py +123 -0
- rag_wright/capabilities/retrieval_core.py +126 -0
- rag_wright/capabilities/rlm_chunking.py +808 -0
- rag_wright/capabilities/rlm_synthesis.py +316 -0
- rag_wright/capabilities/scan_quality.py +136 -0
- rag_wright/capabilities/span_relevance_judgment.py +191 -0
- rag_wright/capabilities/vision_to_text.py +85 -0
- rag_wright/capabilities/vlm_ocr.py +85 -0
- rag_wright/contracts/__init__.py +6 -0
- rag_wright/contracts/chunk.py +79 -0
- rag_wright/contracts/compliance.py +303 -0
- rag_wright/contracts/contract_meta.py +27 -0
- rag_wright/contracts/extraction.py +130 -0
- rag_wright/contracts/function.py +167 -0
- rag_wright/contracts/function_routing.py +91 -0
- rag_wright/contracts/highlight.py +74 -0
- rag_wright/contracts/identifiers.py +153 -0
- rag_wright/contracts/jurisdiction.py +96 -0
- rag_wright/contracts/ontology.py +142 -0
- rag_wright/contracts/property.py +201 -0
- rag_wright/contracts/provenance.py +78 -0
- rag_wright/contracts/query_intent.py +53 -0
- rag_wright/contracts/span.py +76 -0
- rag_wright/contracts/value_match.py +84 -0
- rag_wright/corpus/__init__.py +0 -0
- rag_wright/corpus/canonicalize.py +116 -0
- rag_wright/corpus/cuad.py +153 -0
- rag_wright/corpus/cuad_ingestion.py +72 -0
- rag_wright/corpus/document_parser.py +299 -0
- rag_wright/corpus/edgar.py +231 -0
- rag_wright/corpus/gcs_ingestion.py +120 -0
- rag_wright/corpus/http.py +110 -0
- rag_wright/corpus/selection.py +152 -0
- rag_wright/mcp/__init__.py +11 -0
- rag_wright/mcp/compliance_server.py +299 -0
- rag_wright/mcp/intra_document_qa_server.py +170 -0
- rag_wright/mcp/relational_qa_server.py +171 -0
- rag_wright/mcp/session_store.py +64 -0
- rag_wright/mcp/typed_property_retrieval_server.py +191 -0
- rag_wright/models/__init__.py +8 -0
- rag_wright/models/profiles.py +331 -0
- rag_wright/models/seam.py +497 -0
- rag_wright/models/tag_structured.py +285 -0
- rag_wright/models/tracing.py +179 -0
- rag_wright/models/usage.py +102 -0
- rag_wright/okf/__init__.py +11 -0
- rag_wright/okf/compile.py +292 -0
- rag_wright/okf/document.py +47 -0
- rag_wright/okf/enrich.py +176 -0
- rag_wright/okf/links.py +190 -0
- rag_wright/okf/lint.py +105 -0
- rag_wright/ontology/__init__.py +6 -0
- rag_wright/ontology/_generated_template_meta.py +60 -0
- rag_wright/ontology/_generated_vocab.py +52 -0
- rag_wright/ontology/clause_template.py +964 -0
- rag_wright/ontology/codegen.py +84 -0
- rag_wright/ontology/compliance_bridge.ttl +186 -0
- rag_wright/ontology/contract_bridge.ttl +2685 -0
- rag_wright/ontology/contract_taxonomy.py +24 -0
- rag_wright/ontology/derive.py +58 -0
- rag_wright/ontology/loader.py +435 -0
- rag_wright/ontology/packs/ftc_16cfr255.ttl +29 -0
- rag_wright/ontology/registry.py +87 -0
- rag_wright/ontology/template_introspect.py +100 -0
- rag_wright/py.typed +0 -0
- rag_wright/reference/__init__.py +2 -0
- rag_wright/reference/compliance.py +41 -0
- rag_wright/reference/contract_seam.py +123 -0
- rag_wright/skills/__init__.py +7 -0
- rag_wright/skills/claim_extraction/SKILL.md +47 -0
- rag_wright/skills/claim_extraction/__init__.py +1 -0
- rag_wright/skills/claim_extraction/template.py +50 -0
- rag_wright/skills/compliance_judgment/SKILL.md +59 -0
- rag_wright/skills/corpus_ingest/SKILL.md +106 -0
- rag_wright/skills/extraction_semantic_judge/SKILL.md +51 -0
- rag_wright/skills/extraction_semantic_judge/__init__.py +1 -0
- rag_wright/skills/generation/SKILL.md +64 -0
- rag_wright/skills/generation/__init__.py +1 -0
- rag_wright/skills/generic_compliance_judgment/SKILL.md +58 -0
- rag_wright/skills/okf_navigate/SKILL.md +137 -0
- rag_wright/skills/requirement_extraction/SKILL.md +47 -0
- rag_wright/skills/requirement_extraction/__init__.py +1 -0
- rag_wright/skills/requirement_extraction/template.py +50 -0
- rag_wright/skills/rlm/SKILL.md +186 -0
- rag_wright/skills/rlm/__init__.py +31 -0
- rag_wright/skills/rlm/agent.py +292 -0
- rag_wright/skills/span_relevance_judgment/SKILL.md +67 -0
- rag_wright/skills/vision_to_text/SKILL.md +36 -0
- rag_wright/skills/vision_to_text/__init__.py +1 -0
- rag_wright/spans/__init__.py +1 -0
- rag_wright/spans/boundary.py +78 -0
- rag_wright/spans/clause_function_classifier.py +490 -0
- rag_wright/spans/clause_kg_extractor.py +337 -0
- rag_wright/spans/cuad_labels.py +81 -0
- rag_wright/spans/dim_classifier.py +158 -0
- rag_wright/spans/dim_fleet.json +411 -0
- rag_wright/spans/function_classifier.py +77 -0
- rag_wright/spans/function_families.py +62 -0
- rag_wright/spans/hybrid_classifier.py +103 -0
- rag_wright/spans/legalbert_classifier.py +83 -0
- rag_wright/spans/model_capabilities.py +107 -0
- rag_wright/spans/new_function_labels.py +111 -0
- rag_wright/spans/page_map.py +68 -0
- rag_wright/spans/property_extractor.py +365 -0
- rag_wright/spans/property_grounding.py +182 -0
- rag_wright/spans/reclassify.py +77 -0
- rag_wright/spans/scarce_function_labels.py +105 -0
- rag_wright/spans/segment.py +341 -0
- rag_wright/spans/semantic_judge.py +197 -0
- rag_wright/spans/symbolic_validation.py +131 -0
- rag_wright/spans/tag_clause_extractor.py +182 -0
- rag_wright/store/__init__.py +6 -0
- rag_wright/store/arcadedb.py +1135 -0
- rag_wright/store/chunk_text.py +66 -0
- rag_wright/store/seam.py +213 -0
- rag_wright/subgraphs/__init__.py +0 -0
- rag_wright/subgraphs/async_ingestion.py +204 -0
- rag_wright/subgraphs/compliance_check.py +1042 -0
- rag_wright/subgraphs/compliance_ingestion.py +306 -0
- rag_wright/subgraphs/contract_ingestion_pipeline.py +999 -0
- rag_wright/subgraphs/graph_extraction.py +102 -0
- rag_wright/subgraphs/intra_document_qa.py +328 -0
- rag_wright/subgraphs/observability.py +140 -0
- rag_wright/subgraphs/query_constraint_extraction.py +73 -0
- rag_wright/subgraphs/relational_qa.py +165 -0
- rag_wright/subgraphs/requirement_extraction.py +137 -0
- rag_wright/subgraphs/scaffold.py +65 -0
- rag_wright/subgraphs/semantic_chunking.py +183 -0
- rag_wright/subgraphs/typed_clause_extraction.py +172 -0
- rag_wright/subgraphs/typed_property_retrieval.py +278 -0
- rag_wright/util/__init__.py +1 -0
- rag_wright/util/concurrent.py +153 -0
- rag_wright/util/spacy_model.py +45 -0
- rag_wright-0.1.0.dist-info/METADATA +168 -0
- rag_wright-0.1.0.dist-info/RECORD +184 -0
- rag_wright-0.1.0.dist-info/WHEEL +4 -0
- rag_wright-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,170 @@
|
|
|
1
|
+
"""MCP-PROTO Phase B (B1): the `intra_document_qa` MCP-tool server (FastMCP).
|
|
2
|
+
|
|
3
|
+
Wraps the `intra_document_qa` SUBGRAPH (A1, Leg A) as a single MCP tool
|
|
4
|
+
`answer_contract_question(contract_id, question)` -> a cited `GeneratedAnswer` (as JSON). The store (the contract
|
|
5
|
+
clause KG) + the models (function classifier + Gemma generation via the seam) bind SERVER-SIDE from env, so the
|
|
6
|
+
tool call is just `{contract_id, question}` -- the token/coordination win. An external agent discovers this via
|
|
7
|
+
ARD search and calls it. Second Tier-1 leg wrapped after `compliance_check` (`compliance_server.py`, the
|
|
8
|
+
reference pattern this mirrors).
|
|
9
|
+
|
|
10
|
+
Grounded (library rule): FastMCP `server.py:L278` (`FastMCP(name, instructions, version=...)`), `@mcp.tool`,
|
|
11
|
+
`run(transport=...)` -- confirmed against the cloned-repo AST graph (`graphify-out/framework/graph.json`) + the
|
|
12
|
+
installed 3.4.6 signatures, same surface the reference server uses.
|
|
13
|
+
|
|
14
|
+
`build_intra_document_qa_mcp(qa_fn)` injects the answerer so the server is hermetically testable (a stub
|
|
15
|
+
`qa_fn`, no ArcadeDB/LLM). `main()` picks the production answerer (real subgraph, env-wired) or a deterministic
|
|
16
|
+
demo answerer (`RAG_MCP_DEMO=1`, no infra) and serves over stdio (so a Deep Agent can spawn it).
|
|
17
|
+
|
|
18
|
+
# real (needs the contract KG + a model backend via env, like scripts/phase_a_leg_validate.py):
|
|
19
|
+
uv run --no-sync python -m rag_wright.mcp.intra_document_qa_server
|
|
20
|
+
# demo (no infra -- deterministic answer; for the Deep-Agent prototype):
|
|
21
|
+
RAG_MCP_DEMO=1 uv run --no-sync python -m rag_wright.mcp.intra_document_qa_server
|
|
22
|
+
"""
|
|
23
|
+
|
|
24
|
+
from __future__ import annotations
|
|
25
|
+
|
|
26
|
+
import os
|
|
27
|
+
from typing import Any, Awaitable, Callable, Optional
|
|
28
|
+
|
|
29
|
+
from fastmcp import FastMCP
|
|
30
|
+
|
|
31
|
+
from rag_wright.capabilities.answer_generator import GeneratedAnswer
|
|
32
|
+
|
|
33
|
+
# qa_fn: (contract_id, question) -> GeneratedAnswer. Injected so the server is testable without infra.
|
|
34
|
+
# ASYNC-C1 (ADR-0057): async -- the tool handler awaits it, and it awaits the async intra_document_qa subgraph.
|
|
35
|
+
from rag_wright.mcp.session_store import StoreResolver, resolve_request_store
|
|
36
|
+
from rag_wright.store.seam import Store
|
|
37
|
+
|
|
38
|
+
# issue 0035: store-PARAMETRIC -- the runner takes the per-request store (resolved out-of-band), never a
|
|
39
|
+
# model-supplied one. The demo/stub runner ignores it.
|
|
40
|
+
QAFn = Callable[[Optional[Store], str, str], Awaitable[GeneratedAnswer]]
|
|
41
|
+
|
|
42
|
+
_TOOL_DESCRIPTION = (
|
|
43
|
+
"Answer a natural-language question about ONE known contract from its clause knowledge graph, returning a "
|
|
44
|
+
"grounded, cited answer. Classifies the question to its clause function(s), serves those clauses (with any "
|
|
45
|
+
"cap carve-outs), and generates the answer from that evidence ALONE: every claim cites the chunk_id(s) it "
|
|
46
|
+
"rests on, and a question the contract does not support yields an abstention (abstained=true), never a "
|
|
47
|
+
"fabrication. Use when the contract is already identified and the user asks about its terms."
|
|
48
|
+
)
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def _answer_to_dict(answer: GeneratedAnswer) -> dict[str, Any]:
|
|
52
|
+
"""The tool's JSON payload: the cited answer. This is the structured content the calling agent receives."""
|
|
53
|
+
return answer.model_dump(mode="json")
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def build_intra_document_qa_mcp(
|
|
57
|
+
qa_fn: QAFn, *, store_resolver: Optional[StoreResolver] = None, env_store: Any = None,
|
|
58
|
+
name: str = "rag-wright-intra-document-qa"
|
|
59
|
+
) -> FastMCP:
|
|
60
|
+
"""Build the FastMCP server exposing `intra_document_qa` as one tool. `qa_fn` is injected (real subgraph in
|
|
61
|
+
production; a stub in tests) so the MCP surface is testable with no ArcadeDB / LLM.
|
|
62
|
+
|
|
63
|
+
issue 0035: the tool takes NO tenant/database/scope argument (`contract_id` is a document key WITHIN the
|
|
64
|
+
resolved store, not a tenant selector). `store_resolver` (caller-supplied) binds the store PER SESSION from
|
|
65
|
+
the out-of-band MCP request context; `env_store` is the single-tenant fallback. The resolved store is passed
|
|
66
|
+
to `qa_fn` -- never a model-supplied value."""
|
|
67
|
+
mcp: FastMCP = FastMCP(
|
|
68
|
+
name=name,
|
|
69
|
+
instructions=(
|
|
70
|
+
"Contract question-answering over a per-contract clause knowledge graph. Use "
|
|
71
|
+
"answer_contract_question to answer a question about a known contract with cited evidence."
|
|
72
|
+
),
|
|
73
|
+
)
|
|
74
|
+
|
|
75
|
+
@mcp.tool(name="answer_contract_question", description=_TOOL_DESCRIPTION)
|
|
76
|
+
async def answer_contract_question(contract_id: str, question: str) -> dict[str, Any]:
|
|
77
|
+
"""Answer one question about a known contract from its clause KG.
|
|
78
|
+
|
|
79
|
+
Args:
|
|
80
|
+
contract_id: The identifier of the contract to query (its clauses must be in the KG).
|
|
81
|
+
question: The natural-language question about that contract's terms.
|
|
82
|
+
|
|
83
|
+
Returns:
|
|
84
|
+
A cited answer: {answer, citations[], abstained}. `citations` are chunk_ids present in the evidence;
|
|
85
|
+
`abstained` is true when the contract does not support an answer.
|
|
86
|
+
"""
|
|
87
|
+
store = await resolve_request_store(store_resolver, env_store) # per-request, out-of-band (issue 0035)
|
|
88
|
+
return _answer_to_dict(await qa_fn(store, contract_id, question))
|
|
89
|
+
|
|
90
|
+
return mcp
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
# --- production answerer: the real intra_document_qa subgraph, env-wired (like scripts/phase_a_leg_validate.py) -
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def production_qa_fn(*, function_model_id: str | None = None, answer_model_id: str | None = None) -> QAFn:
|
|
97
|
+
"""Wire the real `intra_document_qa` over the env-selected store + models: ArcadeDB contract KG (`ARCADEDB_*`,
|
|
98
|
+
`QA_DB`), the GENERAL model for generation (via the seam; `RAG_SERVING`). ADR-0047: there is no function
|
|
99
|
+
classifier anymore -- the leg serves the whole contract; `function_model_id` is kept as a back-compat alias
|
|
100
|
+
for the generation-model default. Heavy imports are lazy so `RAG_MCP_DEMO` never pays for them."""
|
|
101
|
+
from dotenv import load_dotenv
|
|
102
|
+
|
|
103
|
+
load_dotenv()
|
|
104
|
+
from rag_wright.models.profiles import ModelRole, model_for
|
|
105
|
+
from rag_wright.subgraphs.intra_document_qa import production_intra_document_qa
|
|
106
|
+
|
|
107
|
+
default_model = answer_model_id or function_model_id or model_for(ModelRole.GENERAL) # tenant-independent
|
|
108
|
+
|
|
109
|
+
async def _qa(store: Optional[Store], contract_id: str, question: str) -> GeneratedAnswer:
|
|
110
|
+
leg = production_intra_document_qa(store=store, answer_model_id=default_model) # per-request store (0035)
|
|
111
|
+
out = await leg.ainvoke({"contract_id": contract_id, "question": question})
|
|
112
|
+
ans = out.get("answer")
|
|
113
|
+
if ans is None: # a pipeline dead-letter (e.g. orphan span) -> an honest abstention, never a fabrication
|
|
114
|
+
return GeneratedAnswer(
|
|
115
|
+
answer="The contract knowledge graph could not be queried for this question.",
|
|
116
|
+
citations=[], abstained=True)
|
|
117
|
+
return ans
|
|
118
|
+
|
|
119
|
+
return _qa
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def production_env_store() -> Store:
|
|
123
|
+
"""The single-tenant fallback store from `QA_DB` (demo / eval / single-tenant). Built lazily so `RAG_MCP_DEMO`
|
|
124
|
+
never touches ArcadeDB; a multi-tenant caller passes a `store_resolver` instead (issue 0035)."""
|
|
125
|
+
from rag_wright.store.arcadedb import ArcadeDBStore
|
|
126
|
+
|
|
127
|
+
return ArcadeDBStore.from_env(database=os.environ.get("QA_DB", "ragwright_cuad_full"))
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
# --- demo answerer: a deterministic, real-shaped cited answer (no ArcadeDB / LLM) for the Deep-Agent prototype -
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def demo_qa_fn() -> QAFn:
|
|
134
|
+
"""A deterministic stub with the REAL contract shape -- a cited, grounded cap answer. Lets the Deep-Agent
|
|
135
|
+
prototype (and the hermetic test) exercise the full MCP path with no ArcadeDB / LLM."""
|
|
136
|
+
_cid = "AcmeMSA:12:deadbeef01"
|
|
137
|
+
|
|
138
|
+
async def _qa(store: Optional[Store], contract_id: str, question: str) -> GeneratedAnswer: # store ignored
|
|
139
|
+
return GeneratedAnswer(
|
|
140
|
+
answer=f"Seller's aggregate liability is capped at two times (2x) the fees paid in the "
|
|
141
|
+
f"preceding 12 months [{_cid}].",
|
|
142
|
+
citations=[_cid], abstained=False)
|
|
143
|
+
|
|
144
|
+
return _qa
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def register_intra_document_qa_mcp(registry) -> None:
|
|
148
|
+
"""Register `intra_document_qa_mcp` (MCP-PROTO Phase B): the ARD `mcp_tool` surface of the
|
|
149
|
+
`intra_document_qa` subgraph -- the same capability exposed as a discoverable, cross-agent MCP tool
|
|
150
|
+
(`answer_contract_question`, served by `rag_wright.mcp.intra_document_qa_server`) so an agent can call it as
|
|
151
|
+
ONE tool via ARD search instead of embedding the subgraph. Distinct ARD identity from the in-process
|
|
152
|
+
`intra_document_qa` subgraph; same output contract `GeneratedAnswer`."""
|
|
153
|
+
registry.register(
|
|
154
|
+
"intra_document_qa_mcp",
|
|
155
|
+
contract=GeneratedAnswer,
|
|
156
|
+
kind="mcp_tool",
|
|
157
|
+
display_name="Intra-document contract QA (MCP tool)",
|
|
158
|
+
)
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
def main() -> None:
|
|
162
|
+
"""Serve the intra_document_qa MCP tool over stdio. `RAG_MCP_DEMO=1` uses the no-infra demo answerer."""
|
|
163
|
+
if os.environ.get("RAG_MCP_DEMO") == "1":
|
|
164
|
+
build_intra_document_qa_mcp(demo_qa_fn()).run(transport="stdio")
|
|
165
|
+
else: # single-tenant CLI serving: env-store fallback (issue 0035; a multi-tenant caller passes a resolver)
|
|
166
|
+
build_intra_document_qa_mcp(production_qa_fn(), env_store=production_env_store).run(transport="stdio")
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
if __name__ == "__main__":
|
|
170
|
+
main()
|
|
@@ -0,0 +1,171 @@
|
|
|
1
|
+
"""MCP-PROTO Phase B (B2): the `relational_qa` MCP-tool server (FastMCP).
|
|
2
|
+
|
|
3
|
+
Wraps the `relational_qa` SUBGRAPH (A2) as a single MCP tool
|
|
4
|
+
`answer_relational_question(query, start_entity_id, max_hops)` -> a cited `GeneratedAnswer` (as JSON). The store
|
|
5
|
+
(the entity/relationship graph) + the generation model (via the seam) bind SERVER-SIDE from env, so the tool
|
|
6
|
+
call is just `{query, start_entity_id}` -- the token/coordination win. Evidence is GRAPH-STRUCTURAL (each reached
|
|
7
|
+
entity cited by its source contract; no chunk text), so a relational answer is verifiable by contract id. Third
|
|
8
|
+
Tier-1 leg wrapped, mirroring the `compliance_server.py` / `intra_document_qa_server.py` reference pattern.
|
|
9
|
+
|
|
10
|
+
Grounded (library rule): FastMCP `server.py:L278` (`FastMCP(name, instructions, version=...)`), `@mcp.tool`,
|
|
11
|
+
`run(transport=...)` -- confirmed against the cloned-repo AST graph (`graphify-out/framework/graph.json`) + the
|
|
12
|
+
installed 3.4.6 signatures, same surface the reference servers use.
|
|
13
|
+
|
|
14
|
+
`build_relational_qa_mcp(qa_fn)` injects the answerer so the server is hermetically testable (a stub `qa_fn`, no
|
|
15
|
+
ArcadeDB/LLM). `main()` picks the production answerer (real subgraph, env-wired) or a deterministic demo answerer
|
|
16
|
+
(`RAG_MCP_DEMO=1`, no infra) and serves over stdio (so a Deep Agent can spawn it).
|
|
17
|
+
|
|
18
|
+
# real (needs the entity graph + a model backend via env, like scripts/phase_a_leg_validate.py):
|
|
19
|
+
uv run --no-sync python -m rag_wright.mcp.relational_qa_server
|
|
20
|
+
# demo (no infra -- deterministic answer; for the Deep-Agent prototype):
|
|
21
|
+
RAG_MCP_DEMO=1 uv run --no-sync python -m rag_wright.mcp.relational_qa_server
|
|
22
|
+
"""
|
|
23
|
+
|
|
24
|
+
from __future__ import annotations
|
|
25
|
+
|
|
26
|
+
import os
|
|
27
|
+
from typing import Any, Awaitable, Callable, Optional
|
|
28
|
+
|
|
29
|
+
from fastmcp import FastMCP
|
|
30
|
+
|
|
31
|
+
from rag_wright.capabilities.answer_generator import GeneratedAnswer
|
|
32
|
+
|
|
33
|
+
# qa_fn: (query, start_entity_id, max_hops) -> GeneratedAnswer. Injected so the server is testable without infra.
|
|
34
|
+
# ASYNC-C1 (ADR-0057): async -- the tool handler awaits it, and it awaits the async relational_qa subgraph.
|
|
35
|
+
from rag_wright.mcp.session_store import StoreResolver, resolve_request_store
|
|
36
|
+
from rag_wright.store.seam import Store
|
|
37
|
+
|
|
38
|
+
# issue 0035: store-PARAMETRIC -- the runner takes the per-request store (resolved out-of-band), never a
|
|
39
|
+
# model-supplied one. The demo/stub runner ignores it.
|
|
40
|
+
RelationalQAFn = Callable[[Optional[Store], str, str, int], Awaitable[GeneratedAnswer]]
|
|
41
|
+
|
|
42
|
+
_TOOL_DESCRIPTION = (
|
|
43
|
+
"Answer a relational question about a known entity by traversing the contract entity graph, returning a "
|
|
44
|
+
"grounded, cited answer. From a start entity, follows relationship edges (default: contracting parties) up "
|
|
45
|
+
"to `max_hops`, and generates the answer from the reached graph structure ALONE: each fact is cited by the "
|
|
46
|
+
"SOURCE CONTRACT it came from (no chunk text), and a question the graph does not support yields an "
|
|
47
|
+
"abstention (abstained=true), never a fabrication. Use for 'which parties does X contract with' style "
|
|
48
|
+
"questions when the start entity is known."
|
|
49
|
+
)
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def _answer_to_dict(answer: GeneratedAnswer) -> dict[str, Any]:
|
|
53
|
+
"""The tool's JSON payload: the cited answer. This is the structured content the calling agent receives."""
|
|
54
|
+
return answer.model_dump(mode="json")
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def build_relational_qa_mcp(
|
|
58
|
+
qa_fn: RelationalQAFn, *, store_resolver: Optional[StoreResolver] = None, env_store: Any = None,
|
|
59
|
+
name: str = "rag-wright-relational-qa"
|
|
60
|
+
) -> FastMCP:
|
|
61
|
+
"""Build the FastMCP server exposing `relational_qa` as one tool. `qa_fn` is injected (real subgraph in
|
|
62
|
+
production; a stub in tests) so the MCP surface is testable with no ArcadeDB / LLM.
|
|
63
|
+
|
|
64
|
+
issue 0035: the tool takes NO tenant/database/scope argument (`start_entity_id` is an entity key WITHIN
|
|
65
|
+
the resolved store). `store_resolver` (caller-supplied) binds the store PER SESSION from the out-of-band
|
|
66
|
+
MCP request context; `env_store` is the single-tenant fallback. The resolved store is passed to `qa_fn`."""
|
|
67
|
+
mcp: FastMCP = FastMCP(
|
|
68
|
+
name=name,
|
|
69
|
+
instructions=(
|
|
70
|
+
"Relational question-answering over the contract entity graph. Use answer_relational_question to "
|
|
71
|
+
"answer a question about a known entity's relationships with cited, graph-structural evidence."
|
|
72
|
+
),
|
|
73
|
+
)
|
|
74
|
+
|
|
75
|
+
@mcp.tool(name="answer_relational_question", description=_TOOL_DESCRIPTION)
|
|
76
|
+
async def answer_relational_question(query: str, start_entity_id: str, max_hops: int = 1) -> dict[str, Any]:
|
|
77
|
+
"""Answer one relational question by traversing the entity graph from a start entity.
|
|
78
|
+
|
|
79
|
+
Args:
|
|
80
|
+
query: The natural-language relational question (e.g. "which parties does X contract with?").
|
|
81
|
+
start_entity_id: The entity_id to traverse from (must be in the graph).
|
|
82
|
+
max_hops: Traversal depth. Defaults to 1 (direct relationships).
|
|
83
|
+
|
|
84
|
+
Returns:
|
|
85
|
+
A cited answer: {answer, citations[], abstained}. `citations` are the SOURCE CONTRACT ids the facts
|
|
86
|
+
came from; `abstained` is true when the graph does not support an answer.
|
|
87
|
+
"""
|
|
88
|
+
store = await resolve_request_store(store_resolver, env_store) # per-request, out-of-band (issue 0035)
|
|
89
|
+
return _answer_to_dict(await qa_fn(store, query, start_entity_id, max_hops))
|
|
90
|
+
|
|
91
|
+
return mcp
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
# --- production answerer: the real relational_qa subgraph, env-wired (like scripts/phase_a_leg_validate.py) -----
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def production_qa_fn(*, answer_model_id: str | None = None) -> RelationalQAFn:
|
|
98
|
+
"""Wire the real `relational_qa` over the env-selected store + model: ArcadeDB entity graph (`ARCADEDB_*`,
|
|
99
|
+
`QA_DB`), the GENERAL model for generation (via the seam; `RAG_SERVING`). Generation goes through
|
|
100
|
+
`answer_model_for` so it takes the right path per ADR-0045 (client-side tag-parse). Heavy imports are lazy so
|
|
101
|
+
`RAG_MCP_DEMO` never pays for them."""
|
|
102
|
+
from dotenv import load_dotenv
|
|
103
|
+
|
|
104
|
+
load_dotenv()
|
|
105
|
+
from rag_wright.capabilities.answer_generator import answer_model_for
|
|
106
|
+
from rag_wright.models.profiles import ModelRole, model_for
|
|
107
|
+
from rag_wright.subgraphs.relational_qa import production_relational_qa
|
|
108
|
+
|
|
109
|
+
answer_model = answer_model_for(answer_model_id or model_for(ModelRole.GENERAL)) # tenant-independent
|
|
110
|
+
|
|
111
|
+
async def _qa(store: Optional[Store], query: str, start_entity_id: str, max_hops: int) -> GeneratedAnswer:
|
|
112
|
+
leg = production_relational_qa(store=store, answer_model=answer_model) # per-request store (0035)
|
|
113
|
+
out = await leg.ainvoke({"query": query, "start_entity_id": start_entity_id, "max_hops": max_hops})
|
|
114
|
+
ans = out.get("answer")
|
|
115
|
+
if ans is None: # defensive: no answer produced -> an honest abstention, never a fabrication
|
|
116
|
+
return GeneratedAnswer(
|
|
117
|
+
answer="The entity graph could not be traversed for this question.",
|
|
118
|
+
citations=[], abstained=True)
|
|
119
|
+
return ans
|
|
120
|
+
|
|
121
|
+
return _qa
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def production_env_store() -> Store:
|
|
125
|
+
"""The single-tenant fallback store from `QA_DB` (demo / eval / single-tenant). Built lazily so
|
|
126
|
+
`RAG_MCP_DEMO` never touches ArcadeDB; a multi-tenant caller passes a `store_resolver` (issue 0035)."""
|
|
127
|
+
from rag_wright.store.arcadedb import ArcadeDBStore
|
|
128
|
+
|
|
129
|
+
return ArcadeDBStore.from_env(database=os.environ.get("QA_DB", "ragwright_cuad_full"))
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
# --- demo answerer: a deterministic, real-shaped cited answer (no ArcadeDB / LLM) for the Deep-Agent prototype -
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def demo_qa_fn() -> RelationalQAFn:
|
|
136
|
+
"""A deterministic stub with the REAL contract shape -- a cited, graph-structural relational answer. Lets the
|
|
137
|
+
Deep-Agent prototype (and the hermetic test) exercise the full MCP path with no ArcadeDB / LLM."""
|
|
138
|
+
_cid = "AcmeBetaMSA:3:beef0002"
|
|
139
|
+
|
|
140
|
+
async def _qa(store: Optional[Store], query: str, start_entity_id: str, max_hops: int) -> GeneratedAnswer:
|
|
141
|
+
return GeneratedAnswer(
|
|
142
|
+
answer=f"AcmeCorp contracts with BetaLLC (per AcmeBetaMSA) [{_cid}].",
|
|
143
|
+
citations=[_cid], abstained=False)
|
|
144
|
+
|
|
145
|
+
return _qa
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
def register_relational_qa_mcp(registry) -> None:
|
|
149
|
+
"""Register `relational_qa_mcp` (MCP-PROTO Phase B): the ARD `mcp_tool` surface of the `relational_qa`
|
|
150
|
+
subgraph -- the same capability exposed as a discoverable, cross-agent MCP tool (`answer_relational_question`,
|
|
151
|
+
served by `rag_wright.mcp.relational_qa_server`) so an agent can call it as ONE tool via ARD search instead of
|
|
152
|
+
embedding the subgraph. Distinct ARD identity from the in-process `relational_qa` subgraph; same output
|
|
153
|
+
contract `GeneratedAnswer`."""
|
|
154
|
+
registry.register(
|
|
155
|
+
"relational_qa_mcp",
|
|
156
|
+
contract=GeneratedAnswer,
|
|
157
|
+
kind="mcp_tool",
|
|
158
|
+
display_name="Relational contract QA (MCP tool)",
|
|
159
|
+
)
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
def main() -> None:
|
|
163
|
+
"""Serve the relational_qa MCP tool over stdio. `RAG_MCP_DEMO=1` uses the no-infra demo answerer."""
|
|
164
|
+
if os.environ.get("RAG_MCP_DEMO") == "1":
|
|
165
|
+
build_relational_qa_mcp(demo_qa_fn()).run(transport="stdio")
|
|
166
|
+
else: # single-tenant CLI: env-store fallback (issue 0035; a multi-tenant caller passes a resolver)
|
|
167
|
+
build_relational_qa_mcp(production_qa_fn(), env_store=production_env_store).run(transport="stdio")
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
if __name__ == "__main__":
|
|
171
|
+
main()
|
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
"""issue 0035: the seam that lets an MCP tool be multi-tenant-safe.
|
|
2
|
+
|
|
3
|
+
THE PRINCIPLE: an MCP tool NEVER takes a tenant, database, or scope as a model-supplied argument. A tool's
|
|
4
|
+
arguments are filled by a language model, so a tenant argument puts a cross-tenant read one token-prediction
|
|
5
|
+
away. Instead the store is resolved from **session context the CALLER establishes out-of-band** (MCP
|
|
6
|
+
initialization params / transport headers), never from a model argument and never from a per-process env var
|
|
7
|
+
in a multi-tenant deployment.
|
|
8
|
+
|
|
9
|
+
The engine does NOT implement tenancy (that is product policy). It provides this seam: a `build_*_mcp` factory
|
|
10
|
+
accepts an optional `store_resolver`, and per request resolves the store from the caller-populated MCP request
|
|
11
|
+
context. When no resolver is wired, it falls back to a single process-wide store -- correct for a demo, an
|
|
12
|
+
eval, or a single-tenant deployment, and still with no model-supplied tenant.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import inspect
|
|
18
|
+
from typing import Any, Awaitable, Callable, Optional, Union
|
|
19
|
+
|
|
20
|
+
from rag_wright.store.seam import Store
|
|
21
|
+
|
|
22
|
+
# The caller binds a store per SESSION: given the opaque per-request MCP context (whatever tenancy the caller
|
|
23
|
+
# set out-of-band -- init params / headers), return that tenant's store. Sync or async. It is handed the MCP
|
|
24
|
+
# request context, NEVER a model-supplied value.
|
|
25
|
+
StoreResolver = Callable[[Any], Union[Store, Awaitable[Store]]]
|
|
26
|
+
|
|
27
|
+
# argument names an MCP tool must never expose to the model (the guard test asserts none appear in any schema).
|
|
28
|
+
FORBIDDEN_TENANT_ARGS = frozenset({"database", "db", "tenant", "scope", "corpus", "workspace", "customer"})
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
async def resolve_request_store(
|
|
32
|
+
store_resolver: Optional[StoreResolver], env_store: Union[Store, Callable[[], Store], None]
|
|
33
|
+
) -> Optional[Store]:
|
|
34
|
+
"""The per-request store, resolved WITHOUT any model-supplied argument.
|
|
35
|
+
|
|
36
|
+
- `store_resolver` given (multi-tenant): resolve against the CURRENT MCP request context (`get_context()`),
|
|
37
|
+
so the caller binds tenancy per session. The context is opaque to the engine -- the caller reads its own
|
|
38
|
+
tenancy from it. Awaited if the resolver is async.
|
|
39
|
+
- `store_resolver` None (single-tenant / demo / eval): return `env_store` (a store, or a factory returning
|
|
40
|
+
one), the process-wide fallback. Still never a model argument.
|
|
41
|
+
"""
|
|
42
|
+
if store_resolver is not None:
|
|
43
|
+
from fastmcp.server.dependencies import get_context # the current request's context (caller-populated)
|
|
44
|
+
|
|
45
|
+
ctx = get_context()
|
|
46
|
+
out = store_resolver(ctx)
|
|
47
|
+
return await out if inspect.isawaitable(out) else out
|
|
48
|
+
return env_store() if callable(env_store) else env_store
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def assert_no_tenant_arguments(mcp: Any) -> list[str]:
|
|
52
|
+
"""Return the offending `tool.arg` names if any MCP tool exposes a model-supplied tenant/database/scope
|
|
53
|
+
argument (issue 0035); an empty list means the server is safe. The guard test fails the build on any hit;
|
|
54
|
+
a caller may also call it on its own servers. Model-facing arguments are the tool's input-schema
|
|
55
|
+
`properties`; a `get_context()`-injected context is never in that schema, so it is correctly ignored."""
|
|
56
|
+
import asyncio
|
|
57
|
+
|
|
58
|
+
tools = asyncio.run(mcp.list_tools())
|
|
59
|
+
offenders: list[str] = []
|
|
60
|
+
for tool in tools:
|
|
61
|
+
props = set((getattr(tool, "parameters", {}) or {}).get("properties", {}))
|
|
62
|
+
for bad in sorted(props & FORBIDDEN_TENANT_ARGS):
|
|
63
|
+
offenders.append(f"{tool.name}.{bad}")
|
|
64
|
+
return offenders
|
|
@@ -0,0 +1,191 @@
|
|
|
1
|
+
"""MCP-PROTO Phase B (B3): the `typed_property_retrieval` MCP-tool server (FastMCP).
|
|
2
|
+
|
|
3
|
+
Wraps the `typed_property_retrieval` SUBGRAPH (Leg B) as a single MCP tool `retrieve_typed_property_spans(query)`
|
|
4
|
+
-> a `TypedPropertyRetrieval` (the query + its property-boosted, cited ranked spans) as JSON. The store (the
|
|
5
|
+
clause KG) + the encoders/models (BGE + LegalBERT + granite constraint-extraction + the function classifier via
|
|
6
|
+
the seam) bind SERVER-SIDE from env, so the tool call is just `{query}` -- the token/coordination win. An
|
|
7
|
+
external agent discovers this via ARD search and calls it.
|
|
8
|
+
|
|
9
|
+
Unlike B1/B2 (which return a `GeneratedAnswer`), this leg's contract is `TypedPropertyRetrieval` -- corpus-wide
|
|
10
|
+
RETRIEVAL, not single-answer generation -- so the tool returns ranked cited spans, not a written answer. Fourth
|
|
11
|
+
and last Tier-1 leg wrapped, mirroring the `compliance_server.py` reference pattern.
|
|
12
|
+
|
|
13
|
+
Grounded (library rule): FastMCP `server.py:L278` (`FastMCP(name, instructions, version=...)`), `@mcp.tool`,
|
|
14
|
+
`run(transport=...)` -- confirmed against the cloned-repo AST graph (`graphify-out/framework/graph.json`) + the
|
|
15
|
+
installed 3.4.6 signatures, same surface the reference servers use.
|
|
16
|
+
|
|
17
|
+
`build_typed_property_retrieval_mcp(retrieval_fn)` injects the retriever so the server is hermetically testable
|
|
18
|
+
(a stub, no ArcadeDB/encoders/LLM). `main()` picks the production retriever (real subgraph, env-wired) or a
|
|
19
|
+
deterministic demo retriever (`RAG_MCP_DEMO=1`, no infra) and serves over stdio (so a Deep Agent can spawn it).
|
|
20
|
+
|
|
21
|
+
# real (needs the clause KG + the encoders/models via env, like scripts/phase_a_leg_validate.py):
|
|
22
|
+
uv run --no-sync python -m rag_wright.mcp.typed_property_retrieval_server
|
|
23
|
+
# demo (no infra -- deterministic ranked spans; for the Deep-Agent prototype):
|
|
24
|
+
RAG_MCP_DEMO=1 uv run --no-sync python -m rag_wright.mcp.typed_property_retrieval_server
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
from __future__ import annotations
|
|
28
|
+
|
|
29
|
+
import os
|
|
30
|
+
from typing import Any, Awaitable, Callable, Optional
|
|
31
|
+
|
|
32
|
+
from fastmcp import FastMCP
|
|
33
|
+
|
|
34
|
+
from rag_wright.capabilities.property_boosted_retrieval import RankedSpan
|
|
35
|
+
from rag_wright.capabilities.span_relevance_judgment import RelevanceVerdict
|
|
36
|
+
from rag_wright.subgraphs.typed_property_retrieval import JudgedSpan, TypedPropertyRetrieval
|
|
37
|
+
|
|
38
|
+
# retrieval_fn: (query) -> TypedPropertyRetrieval. Injected so the server is testable without infra.
|
|
39
|
+
# ASYNC-C1 (ADR-0057): async -- the tool handler awaits it, and it awaits the async typed_property_retrieval subgraph.
|
|
40
|
+
from rag_wright.mcp.session_store import StoreResolver, resolve_request_store
|
|
41
|
+
from rag_wright.store.seam import Store
|
|
42
|
+
|
|
43
|
+
# issue 0035: store-PARAMETRIC -- the runner takes the per-request store (resolved out-of-band), never a
|
|
44
|
+
# model-supplied one. The demo/stub runner ignores the store.
|
|
45
|
+
RetrievalFn = Callable[[Optional[Store], str], Awaitable[TypedPropertyRetrieval]]
|
|
46
|
+
|
|
47
|
+
_TOOL_DESCRIPTION = (
|
|
48
|
+
"Retrieve the most relevant contract clauses for a query from across the corpus, property-boosted and "
|
|
49
|
+
"cited. Extracts the query's typed (dimension, value) constraints and routes its clause function(s), then "
|
|
50
|
+
"ranks a bounded semantic pool by constraint match, returning ranked spans each with its span_id citation, "
|
|
51
|
+
"text, clause function, match score, and the constraints it satisfied. This is corpus-wide RETRIEVAL (ranked "
|
|
52
|
+
"evidence), not a written answer. Use to find clauses matching a typed condition (e.g. 'cap on liability set "
|
|
53
|
+
"at a multiple of the fees paid') across many contracts."
|
|
54
|
+
)
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def _retrieval_to_dict(retrieval: TypedPropertyRetrieval) -> dict[str, Any]:
|
|
58
|
+
"""The tool's JSON payload: the query + its ranked cited spans. This is the structured content the calling
|
|
59
|
+
agent receives."""
|
|
60
|
+
return retrieval.model_dump(mode="json")
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def build_typed_property_retrieval_mcp(
|
|
64
|
+
retrieval_fn: RetrievalFn, *, store_resolver: Optional[StoreResolver] = None,
|
|
65
|
+
env_store: Any = None, name: str = "rag-wright-typed-property-retrieval"
|
|
66
|
+
) -> FastMCP:
|
|
67
|
+
"""Build the FastMCP server exposing `typed_property_retrieval` as one tool. `retrieval_fn` is injected (real
|
|
68
|
+
subgraph in production; a stub in tests) so the MCP surface is testable with no ArcadeDB / encoders / LLM.
|
|
69
|
+
|
|
70
|
+
issue 0035: the tool takes NO tenant/database/scope argument. `store_resolver` (caller-supplied) binds the
|
|
71
|
+
store PER SESSION from the out-of-band MCP request context; `env_store` is the single-tenant fallback used
|
|
72
|
+
when no resolver is wired. The resolved store is passed to `retrieval_fn` -- never a model-supplied value."""
|
|
73
|
+
mcp: FastMCP = FastMCP(
|
|
74
|
+
name=name,
|
|
75
|
+
instructions=(
|
|
76
|
+
"Corpus-wide, property-boosted clause retrieval over the contract KG. Use "
|
|
77
|
+
"retrieve_typed_property_spans to find the clauses matching a typed condition, ranked and cited."
|
|
78
|
+
),
|
|
79
|
+
)
|
|
80
|
+
|
|
81
|
+
@mcp.tool(name="retrieve_typed_property_spans", description=_TOOL_DESCRIPTION)
|
|
82
|
+
async def retrieve_typed_property_spans(query: str) -> dict[str, Any]:
|
|
83
|
+
"""Retrieve ranked, cited clause spans matching a query from across the corpus.
|
|
84
|
+
|
|
85
|
+
Args:
|
|
86
|
+
query: The retrieval query, ideally expressing a typed condition (a clause function and/or a
|
|
87
|
+
property value, e.g. "cap on liability at a multiple of fees").
|
|
88
|
+
|
|
89
|
+
Returns:
|
|
90
|
+
{query, results[]} where each result is a ranked cited span WITH its relevance verdict (issue 0023):
|
|
91
|
+
{span: {span_id, text, function, match_score, matched[], rank}, relevance: {verdict, rationale,
|
|
92
|
+
confidence} | null}. `verdict` is relevant | not_relevant | uncertain; `relevance` is null only when no
|
|
93
|
+
relevance judge is wired. Empty results when nothing matches.
|
|
94
|
+
"""
|
|
95
|
+
store = await resolve_request_store(store_resolver, env_store) # per-request, out-of-band (issue 0035)
|
|
96
|
+
return _retrieval_to_dict(await retrieval_fn(store, query))
|
|
97
|
+
|
|
98
|
+
return mcp
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
# --- production retriever: the real Leg B subgraph, env-wired (like scripts/phase_a_leg_validate.py validate_leg_b) -
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def production_retrieval_fn(*, k: int = 8) -> RetrievalFn:
|
|
105
|
+
"""Wire the real `typed_property_retrieval` over the encoders + models (BGE embedder local or the A100
|
|
106
|
+
`STACK_URL` adapter; granite constraint-extraction via the seam, `RAG_SERVING`). ADR-0047: no function
|
|
107
|
+
classifier -- whole-index pool. Heavy imports are lazy so `RAG_MCP_DEMO` never pays for them.
|
|
108
|
+
|
|
109
|
+
issue 0035: store-PARAMETRIC. The tenant-independent pieces (embedder, model) are built ONCE here; the
|
|
110
|
+
STORE arrives per request (resolved out-of-band by the caller) and the leg is wired against it per call --
|
|
111
|
+
so one server process serves many tenants without a per-process database."""
|
|
112
|
+
from dotenv import load_dotenv
|
|
113
|
+
|
|
114
|
+
load_dotenv()
|
|
115
|
+
from rag_wright.capabilities.dg_extraction import default_extraction_model
|
|
116
|
+
from rag_wright.capabilities.remote_encoders import query_embedder
|
|
117
|
+
from rag_wright.subgraphs.typed_property_retrieval import production_typed_property_retrieval
|
|
118
|
+
|
|
119
|
+
embedder = query_embedder() # tenant-independent, built once
|
|
120
|
+
extract_model = default_extraction_model("query-constraints")
|
|
121
|
+
|
|
122
|
+
async def _retrieve(store: Optional[Store], query: str) -> TypedPropertyRetrieval:
|
|
123
|
+
leg = production_typed_property_retrieval(store=store, embedder=embedder, extract_model=extract_model, k=k)
|
|
124
|
+
out = await leg.ainvoke({"query": query})
|
|
125
|
+
return out["retrieval"]
|
|
126
|
+
|
|
127
|
+
return _retrieve
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def production_env_store() -> Store:
|
|
131
|
+
"""The single-tenant fallback store from `QA_DB` (a demo / eval / single-tenant deployment). Built lazily so
|
|
132
|
+
`RAG_MCP_DEMO` never touches ArcadeDB; a multi-tenant caller passes a `store_resolver` instead (issue 0035)."""
|
|
133
|
+
from rag_wright.store.arcadedb import ArcadeDBStore
|
|
134
|
+
|
|
135
|
+
return ArcadeDBStore.from_env(database=os.environ.get("QA_DB", "ragwright_cuad_full"))
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
# --- demo retriever: deterministic, real-shaped ranked cited spans (no infra) for the Deep-Agent prototype -----
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
def demo_retrieval_fn() -> RetrievalFn:
|
|
142
|
+
"""A deterministic stub with the REAL contract shape -- two ranked, cited, property-matched cap spans. Lets
|
|
143
|
+
the Deep-Agent prototype (and the hermetic test) exercise the full MCP path with no ArcadeDB / encoders."""
|
|
144
|
+
|
|
145
|
+
async def _retrieve(store: Optional[Store], query: str) -> TypedPropertyRetrieval: # store ignored (demo)
|
|
146
|
+
results = [
|
|
147
|
+
JudgedSpan(
|
|
148
|
+
span=RankedSpan(
|
|
149
|
+
span_id="AcmeMSA:12:deadbeef01",
|
|
150
|
+
text="Seller's aggregate liability shall not exceed two times (2x) the fees paid.",
|
|
151
|
+
function="Cap On Liability", match_score=0.94, matched=[("cap_multiple", "2x")], rank=1),
|
|
152
|
+
relevance=RelevanceVerdict(verdict="relevant", rationale="an express liability cap", confidence=0.95)),
|
|
153
|
+
JudgedSpan(
|
|
154
|
+
span=RankedSpan(
|
|
155
|
+
span_id="BetaSaaS:7:cafe0042",
|
|
156
|
+
text="In no event shall liability exceed the total fees paid in the prior 12 months.",
|
|
157
|
+
function="Cap On Liability", match_score=0.71, matched=[("cap_basis", "fees paid")], rank=2),
|
|
158
|
+
relevance=RelevanceVerdict(verdict="relevant", rationale="a fees-paid liability cap", confidence=0.9)),
|
|
159
|
+
]
|
|
160
|
+
return TypedPropertyRetrieval(query=query, results=results)
|
|
161
|
+
|
|
162
|
+
return _retrieve
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
def register_typed_property_retrieval_mcp(registry) -> None:
|
|
166
|
+
"""Register `typed_property_retrieval_mcp` (MCP-PROTO Phase B): the ARD `mcp_tool` surface of the
|
|
167
|
+
`typed_property_retrieval` subgraph -- the same capability exposed as a discoverable, cross-agent MCP tool
|
|
168
|
+
(`retrieve_typed_property_spans`, served by `rag_wright.mcp.typed_property_retrieval_server`) so an agent can
|
|
169
|
+
call it as ONE tool via ARD search instead of embedding the subgraph. Distinct ARD identity from the
|
|
170
|
+
in-process `typed_property_retrieval` subgraph; same output contract `TypedPropertyRetrieval`."""
|
|
171
|
+
registry.register(
|
|
172
|
+
"typed_property_retrieval_mcp",
|
|
173
|
+
contract=TypedPropertyRetrieval,
|
|
174
|
+
kind="mcp_tool",
|
|
175
|
+
display_name="Typed property-boosted retrieval (MCP tool)",
|
|
176
|
+
)
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
def main() -> None:
|
|
180
|
+
"""Serve the typed_property_retrieval MCP tool over stdio. `RAG_MCP_DEMO=1` uses the no-infra demo retriever."""
|
|
181
|
+
# single-tenant CLI serving: env-store fallback (no resolver). A multi-tenant caller builds the server in
|
|
182
|
+
# process with a `store_resolver` instead (issue 0035).
|
|
183
|
+
if os.environ.get("RAG_MCP_DEMO") == "1":
|
|
184
|
+
build_typed_property_retrieval_mcp(demo_retrieval_fn()).run(transport="stdio")
|
|
185
|
+
else:
|
|
186
|
+
build_typed_property_retrieval_mcp(production_retrieval_fn(), env_store=production_env_store).run(
|
|
187
|
+
transport="stdio")
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
if __name__ == "__main__":
|
|
191
|
+
main()
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
"""The model-profile seam (tech stack, assumption 2).
|
|
2
|
+
|
|
3
|
+
Model-neutral through OpenRouter by default (Gemma 4 class for reasoning, generation,
|
|
4
|
+
vision-to-text, and RLM; a smaller model for chunking and summarization; a larger model for
|
|
5
|
+
quality-sensitive extraction), with local open-model deployment supported. Structured-output
|
|
6
|
+
calls go through this seam, keyed by model id, applied at the single point where a model is
|
|
7
|
+
constructed. Provider flags live in profile config and a dated ADR, never in node or agent code.
|
|
8
|
+
"""
|