rag-wright 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- rag_wright/__init__.py +13 -0
- rag_wright/api/__init__.py +33 -0
- rag_wright/api/config.py +59 -0
- rag_wright/api/discover.py +70 -0
- rag_wright/api/documents.py +39 -0
- rag_wright/api/ids.py +31 -0
- rag_wright/api/invoke.py +99 -0
- rag_wright/api/kg.py +61 -0
- rag_wright/api/mcp.py +94 -0
- rag_wright/api/usage.py +30 -0
- rag_wright/api/workspace.py +85 -0
- rag_wright/capabilities/__init__.py +8 -0
- rag_wright/capabilities/answer_generator.py +427 -0
- rag_wright/capabilities/ard.py +286 -0
- rag_wright/capabilities/assertion_extraction.py +79 -0
- rag_wright/capabilities/chunk_read.py +58 -0
- rag_wright/capabilities/chunk_write.py +163 -0
- rag_wright/capabilities/claim_extraction.py +153 -0
- rag_wright/capabilities/clause_exception_linking.py +117 -0
- rag_wright/capabilities/compliance_judgment.py +322 -0
- rag_wright/capabilities/compliance_store.py +87 -0
- rag_wright/capabilities/contract_kg_serve.py +156 -0
- rag_wright/capabilities/contract_kg_store.py +251 -0
- rag_wright/capabilities/dg_extraction.py +585 -0
- rag_wright/capabilities/disambiguation.py +163 -0
- rag_wright/capabilities/document_parse.py +87 -0
- rag_wright/capabilities/document_scope.py +49 -0
- rag_wright/capabilities/embedding.py +164 -0
- rag_wright/capabilities/embedding_profiles.py +43 -0
- rag_wright/capabilities/entity_resolution.py +154 -0
- rag_wright/capabilities/fusion.py +64 -0
- rag_wright/capabilities/graph_extraction.py +243 -0
- rag_wright/capabilities/graph_query.py +73 -0
- rag_wright/capabilities/graph_storage.py +111 -0
- rag_wright/capabilities/highlight_serve.py +142 -0
- rag_wright/capabilities/hybrid_search.py +65 -0
- rag_wright/capabilities/invoke.py +31 -0
- rag_wright/capabilities/jev_decision.py +38 -0
- rag_wright/capabilities/manifests.py +872 -0
- rag_wright/capabilities/okf_navigate.py +456 -0
- rag_wright/capabilities/parsing.py +286 -0
- rag_wright/capabilities/property_boosted_retrieval.py +125 -0
- rag_wright/capabilities/query_function_classifier.py +94 -0
- rag_wright/capabilities/query_understanding.py +109 -0
- rag_wright/capabilities/registry.py +262 -0
- rag_wright/capabilities/remote_encoders.py +94 -0
- rag_wright/capabilities/requirement_extraction.py +247 -0
- rag_wright/capabilities/reranking.py +123 -0
- rag_wright/capabilities/retrieval_core.py +126 -0
- rag_wright/capabilities/rlm_chunking.py +808 -0
- rag_wright/capabilities/rlm_synthesis.py +316 -0
- rag_wright/capabilities/scan_quality.py +136 -0
- rag_wright/capabilities/span_relevance_judgment.py +191 -0
- rag_wright/capabilities/vision_to_text.py +85 -0
- rag_wright/capabilities/vlm_ocr.py +85 -0
- rag_wright/contracts/__init__.py +6 -0
- rag_wright/contracts/chunk.py +79 -0
- rag_wright/contracts/compliance.py +303 -0
- rag_wright/contracts/contract_meta.py +27 -0
- rag_wright/contracts/extraction.py +130 -0
- rag_wright/contracts/function.py +167 -0
- rag_wright/contracts/function_routing.py +91 -0
- rag_wright/contracts/highlight.py +74 -0
- rag_wright/contracts/identifiers.py +153 -0
- rag_wright/contracts/jurisdiction.py +96 -0
- rag_wright/contracts/ontology.py +142 -0
- rag_wright/contracts/property.py +201 -0
- rag_wright/contracts/provenance.py +78 -0
- rag_wright/contracts/query_intent.py +53 -0
- rag_wright/contracts/span.py +76 -0
- rag_wright/contracts/value_match.py +84 -0
- rag_wright/corpus/__init__.py +0 -0
- rag_wright/corpus/canonicalize.py +116 -0
- rag_wright/corpus/cuad.py +153 -0
- rag_wright/corpus/cuad_ingestion.py +72 -0
- rag_wright/corpus/document_parser.py +299 -0
- rag_wright/corpus/edgar.py +231 -0
- rag_wright/corpus/gcs_ingestion.py +120 -0
- rag_wright/corpus/http.py +110 -0
- rag_wright/corpus/selection.py +152 -0
- rag_wright/mcp/__init__.py +11 -0
- rag_wright/mcp/compliance_server.py +299 -0
- rag_wright/mcp/intra_document_qa_server.py +170 -0
- rag_wright/mcp/relational_qa_server.py +171 -0
- rag_wright/mcp/session_store.py +64 -0
- rag_wright/mcp/typed_property_retrieval_server.py +191 -0
- rag_wright/models/__init__.py +8 -0
- rag_wright/models/profiles.py +331 -0
- rag_wright/models/seam.py +497 -0
- rag_wright/models/tag_structured.py +285 -0
- rag_wright/models/tracing.py +179 -0
- rag_wright/models/usage.py +102 -0
- rag_wright/okf/__init__.py +11 -0
- rag_wright/okf/compile.py +292 -0
- rag_wright/okf/document.py +47 -0
- rag_wright/okf/enrich.py +176 -0
- rag_wright/okf/links.py +190 -0
- rag_wright/okf/lint.py +105 -0
- rag_wright/ontology/__init__.py +6 -0
- rag_wright/ontology/_generated_template_meta.py +60 -0
- rag_wright/ontology/_generated_vocab.py +52 -0
- rag_wright/ontology/clause_template.py +964 -0
- rag_wright/ontology/codegen.py +84 -0
- rag_wright/ontology/compliance_bridge.ttl +186 -0
- rag_wright/ontology/contract_bridge.ttl +2685 -0
- rag_wright/ontology/contract_taxonomy.py +24 -0
- rag_wright/ontology/derive.py +58 -0
- rag_wright/ontology/loader.py +435 -0
- rag_wright/ontology/packs/ftc_16cfr255.ttl +29 -0
- rag_wright/ontology/registry.py +87 -0
- rag_wright/ontology/template_introspect.py +100 -0
- rag_wright/py.typed +0 -0
- rag_wright/reference/__init__.py +2 -0
- rag_wright/reference/compliance.py +41 -0
- rag_wright/reference/contract_seam.py +123 -0
- rag_wright/skills/__init__.py +7 -0
- rag_wright/skills/claim_extraction/SKILL.md +47 -0
- rag_wright/skills/claim_extraction/__init__.py +1 -0
- rag_wright/skills/claim_extraction/template.py +50 -0
- rag_wright/skills/compliance_judgment/SKILL.md +59 -0
- rag_wright/skills/corpus_ingest/SKILL.md +106 -0
- rag_wright/skills/extraction_semantic_judge/SKILL.md +51 -0
- rag_wright/skills/extraction_semantic_judge/__init__.py +1 -0
- rag_wright/skills/generation/SKILL.md +64 -0
- rag_wright/skills/generation/__init__.py +1 -0
- rag_wright/skills/generic_compliance_judgment/SKILL.md +58 -0
- rag_wright/skills/okf_navigate/SKILL.md +137 -0
- rag_wright/skills/requirement_extraction/SKILL.md +47 -0
- rag_wright/skills/requirement_extraction/__init__.py +1 -0
- rag_wright/skills/requirement_extraction/template.py +50 -0
- rag_wright/skills/rlm/SKILL.md +186 -0
- rag_wright/skills/rlm/__init__.py +31 -0
- rag_wright/skills/rlm/agent.py +292 -0
- rag_wright/skills/span_relevance_judgment/SKILL.md +67 -0
- rag_wright/skills/vision_to_text/SKILL.md +36 -0
- rag_wright/skills/vision_to_text/__init__.py +1 -0
- rag_wright/spans/__init__.py +1 -0
- rag_wright/spans/boundary.py +78 -0
- rag_wright/spans/clause_function_classifier.py +490 -0
- rag_wright/spans/clause_kg_extractor.py +337 -0
- rag_wright/spans/cuad_labels.py +81 -0
- rag_wright/spans/dim_classifier.py +158 -0
- rag_wright/spans/dim_fleet.json +411 -0
- rag_wright/spans/function_classifier.py +77 -0
- rag_wright/spans/function_families.py +62 -0
- rag_wright/spans/hybrid_classifier.py +103 -0
- rag_wright/spans/legalbert_classifier.py +83 -0
- rag_wright/spans/model_capabilities.py +107 -0
- rag_wright/spans/new_function_labels.py +111 -0
- rag_wright/spans/page_map.py +68 -0
- rag_wright/spans/property_extractor.py +365 -0
- rag_wright/spans/property_grounding.py +182 -0
- rag_wright/spans/reclassify.py +77 -0
- rag_wright/spans/scarce_function_labels.py +105 -0
- rag_wright/spans/segment.py +341 -0
- rag_wright/spans/semantic_judge.py +197 -0
- rag_wright/spans/symbolic_validation.py +131 -0
- rag_wright/spans/tag_clause_extractor.py +182 -0
- rag_wright/store/__init__.py +6 -0
- rag_wright/store/arcadedb.py +1135 -0
- rag_wright/store/chunk_text.py +66 -0
- rag_wright/store/seam.py +213 -0
- rag_wright/subgraphs/__init__.py +0 -0
- rag_wright/subgraphs/async_ingestion.py +204 -0
- rag_wright/subgraphs/compliance_check.py +1042 -0
- rag_wright/subgraphs/compliance_ingestion.py +306 -0
- rag_wright/subgraphs/contract_ingestion_pipeline.py +999 -0
- rag_wright/subgraphs/graph_extraction.py +102 -0
- rag_wright/subgraphs/intra_document_qa.py +328 -0
- rag_wright/subgraphs/observability.py +140 -0
- rag_wright/subgraphs/query_constraint_extraction.py +73 -0
- rag_wright/subgraphs/relational_qa.py +165 -0
- rag_wright/subgraphs/requirement_extraction.py +137 -0
- rag_wright/subgraphs/scaffold.py +65 -0
- rag_wright/subgraphs/semantic_chunking.py +183 -0
- rag_wright/subgraphs/typed_clause_extraction.py +172 -0
- rag_wright/subgraphs/typed_property_retrieval.py +278 -0
- rag_wright/util/__init__.py +1 -0
- rag_wright/util/concurrent.py +153 -0
- rag_wright/util/spacy_model.py +45 -0
- rag_wright-0.1.0.dist-info/METADATA +168 -0
- rag_wright-0.1.0.dist-info/RECORD +184 -0
- rag_wright-0.1.0.dist-info/WHEEL +4 -0
- rag_wright-0.1.0.dist-info/licenses/LICENSE +21 -0
rag_wright/__init__.py
ADDED
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
"""RAG_Wright: the open-core RAG ENGINE (RAG_Capability_Spec.md v0.1; engine/product split, ADR-0052).
|
|
2
|
+
|
|
3
|
+
This package builds and registers the FR-C capabilities (parsing, chunking, embedding, hybrid search,
|
|
4
|
+
reranking, graph extraction, entity resolution, ontology and registry derivation, reasoning and generation,
|
|
5
|
+
and the RLM skill) as ordinary tested software, AND the ingestion and query pipelines that compose them --
|
|
6
|
+
ordinary hardened LangGraph subgraphs, not a compiler output (GraphWright is PARKED, ADR-0052). Each capability
|
|
7
|
+
is registered under its FR-C name for Agentic Resource Discovery (ARD, `urn:air`).
|
|
8
|
+
|
|
9
|
+
The engine is ASYNC end to end (ADR-0057): every model call runs under a true wall-clock deadline. The public
|
|
10
|
+
ingest / query / compliance entrypoints are `async` (or return LangGraph graphs called via `.ainvoke`); see
|
|
11
|
+
`docs/product/engine_async_api.md` for the interface contract the product (RuleWright) depends on. Dependency
|
|
12
|
+
direction is strict and one-way: Product -> Engine, never Engine -> Product.
|
|
13
|
+
"""
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
"""The engine API layer (ADR-0117): the stable, domain-agnostic surface a product builds on.
|
|
2
|
+
|
|
3
|
+
EP-API-1 ships the typed config + the opaque workspace handle. Future increments add the per-kind invokers
|
|
4
|
+
(EP-API-2), generic `kg_read`/`kg_write` + id/format accessors (EP-API-3), and the options catalog + pluggable
|
|
5
|
+
embedders (EP-API-4). The product imports from here; it never reaches the store/embedder/model implementations."""
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
from rag_wright.api.config import EngineConfig, EngineOptions, IngestOptions, StoreConfig
|
|
9
|
+
# The capability-registration surface lives in capabilities.manifests; re-export it here (PREP-1.5) so the public
|
|
10
|
+
# story is uniformly "everything is rag_wright.api". The original import path keeps working — these are the same
|
|
11
|
+
# objects, not a fork.
|
|
12
|
+
from rag_wright.capabilities.manifests import (
|
|
13
|
+
load_reference_pack,
|
|
14
|
+
reference_pack,
|
|
15
|
+
register_capability,
|
|
16
|
+
)
|
|
17
|
+
from rag_wright.api.documents import aparse_document, parse_document, source_document
|
|
18
|
+
from rag_wright.api.ids import decode_bbox, document_of, id_source
|
|
19
|
+
from rag_wright.api.discover import Discovered, discover
|
|
20
|
+
from rag_wright.api.invoke import ainvoke_model, ainvoke_subgraph, capability_index, invoke_model
|
|
21
|
+
from rag_wright.api.kg import entities_by_name, kg_edges, kg_read, kg_write, span_positions
|
|
22
|
+
from rag_wright.api.usage import ModelUsage, UsageTotals, measure_usage
|
|
23
|
+
from rag_wright.api.workspace import WorkspaceHandle, open_workspace
|
|
24
|
+
|
|
25
|
+
__all__ = [
|
|
26
|
+
"EngineConfig", "StoreConfig", "EngineOptions", "IngestOptions", "WorkspaceHandle", "open_workspace",
|
|
27
|
+
"ainvoke_subgraph", "invoke_model", "ainvoke_model", "capability_index", "discover", "Discovered",
|
|
28
|
+
"kg_read", "kg_write", "kg_edges", "entities_by_name", "span_positions",
|
|
29
|
+
"document_of", "id_source", "decode_bbox",
|
|
30
|
+
"source_document", "parse_document", "aparse_document",
|
|
31
|
+
"measure_usage", "UsageTotals", "ModelUsage",
|
|
32
|
+
"register_capability", "load_reference_pack", "reference_pack",
|
|
33
|
+
]
|
rag_wright/api/config.py
ADDED
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
"""EP-API-1/4 (ADR-0117): the engine's typed configuration. The product constructs an `EngineConfig` and passes it
|
|
2
|
+
to `open_workspace`; the engine resolves backends behind an opaque handle. This is the ONLY place a store backend is
|
|
3
|
+
named -- everything else goes through the handle. EP-API-4a adds the `options` catalog (ingest knobs today;
|
|
4
|
+
retrieval/rerank groups land as needed), so a product tunes the engine through config, never environment variables."""
|
|
5
|
+
from __future__ import annotations
|
|
6
|
+
|
|
7
|
+
from dataclasses import dataclass, field
|
|
8
|
+
from typing import Optional
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
@dataclass(frozen=True)
|
|
12
|
+
class StoreConfig:
|
|
13
|
+
"""How to reach the KG/retrieval store. `backend` selects the implementation (only `arcadedb` today; a new
|
|
14
|
+
backend -- e.g. Neo4j -- is added here, invisibly to the product)."""
|
|
15
|
+
|
|
16
|
+
host: str
|
|
17
|
+
port: str
|
|
18
|
+
user: str
|
|
19
|
+
password: str
|
|
20
|
+
backend: str = "arcadedb"
|
|
21
|
+
protocol: str = "http"
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
@dataclass(frozen=True)
|
|
25
|
+
class IngestOptions:
|
|
26
|
+
"""Ingest-time knobs, settable through config instead of environment variables (EP-API-4a). Every field defaults
|
|
27
|
+
to `None` = "use the engine default", so the engine's existing env fallback is preserved (a non-API caller is
|
|
28
|
+
unaffected) and an API caller that leaves these unset gets today's behavior exactly. Set a field to override."""
|
|
29
|
+
|
|
30
|
+
classify_concurrency: Optional[int] = None # function-classify parallelism (was CLASSIFY_CONCURRENCY)
|
|
31
|
+
clause_concurrency: Optional[int] = None # clause-extraction parallelism (was CLAUSE_CONCURRENCY)
|
|
32
|
+
affiliations: Optional[bool] = None # run affiliation extraction (was RAG_INGEST_AFFILIATIONS)
|
|
33
|
+
function_classifier: Optional[str] = None # "setfit" | "llm" (was RAG_FUNCTION_CLASSIFIER)
|
|
34
|
+
list_model: Optional[str] = None # secondary list-union model, "off" to disable (was RAG_INGEST_LIST_MODEL)
|
|
35
|
+
clause_samples: Optional[int] = None # multi-sample count for the list union (was RAG_INGEST_CLAUSE_SAMPLES)
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
@dataclass(frozen=True)
|
|
39
|
+
class EngineOptions:
|
|
40
|
+
"""The engine's options catalog. Ingest knobs today; retrieval / reranking / chunking groups are added here as
|
|
41
|
+
they are promoted off environment variables."""
|
|
42
|
+
|
|
43
|
+
ingest: IngestOptions = field(default_factory=IngestOptions)
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
@dataclass(frozen=True)
|
|
47
|
+
class EngineConfig:
|
|
48
|
+
"""The product's view of the engine: the store connection, chosen models (by role alias), the embedding profile,
|
|
49
|
+
and the `options` catalog. Defaults just work; override only to trade quality/cost/latency. Implementation
|
|
50
|
+
details (ArcadeDB, BGE) never cross this boundary."""
|
|
51
|
+
|
|
52
|
+
store: StoreConfig
|
|
53
|
+
models: dict[str, str] = field(default_factory=dict) # ModelRole value -> engine-supported model alias (override)
|
|
54
|
+
embeddings: dict[str, str] = field(default_factory=lambda: {"text": "bge-m3"}) # profile -> supported embedder
|
|
55
|
+
options: EngineOptions = field(default_factory=EngineOptions) # EP-API-4a: ingest (+ future) knobs via config
|
|
56
|
+
# AC-journey: the DOMAIN pack `.ttl` declaring this domain's KG vertex/edge types (open_workspace creates them).
|
|
57
|
+
# None = the engine's reference CONTRACT pack. A new domain points this at its own `.ttl` -- "config + .ttl",
|
|
58
|
+
# no engine edit -- and `ensure_schema` creates that domain's schema on top of the always-on engine types.
|
|
59
|
+
pack: Optional[str] = None
|
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
"""Capability discovery: rank the live ARD catalog by semantic match to a task (embedding-based).
|
|
2
|
+
|
|
3
|
+
A product-side agent that needs to PLAN over the engine — when invoking a known capability by name through its seam
|
|
4
|
+
is not enough and it must find what's available for a task — calls `discover(task, resources=ws)`, then orchestrates
|
|
5
|
+
the `ainvoke_*` calls over the top matches. Ranking is by BGE-M3 embedding similarity of the task against each
|
|
6
|
+
capability's `representative_queries` + description (the signal manifests carry for exactly this), using the
|
|
7
|
+
workspace's query embedder — the same embedder and vector space retrieval uses. Domain-free: it ranks whatever is in
|
|
8
|
+
the LIVE catalog (`MANIFEST_SPECS`), so it covers the product's OWN registered capabilities, not just the reference
|
|
9
|
+
pack. Complements `capability_index` (the flat listing) with task-ranked selection.
|
|
10
|
+
"""
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import math
|
|
14
|
+
from dataclasses import dataclass
|
|
15
|
+
from typing import Optional
|
|
16
|
+
|
|
17
|
+
from rag_wright.api.workspace import WorkspaceHandle
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
@dataclass(frozen=True)
|
|
21
|
+
class Discovered:
|
|
22
|
+
"""One ranked capability match from `discover` — enough for an agent to pick and invoke it by `slug`."""
|
|
23
|
+
|
|
24
|
+
slug: str
|
|
25
|
+
kind: str
|
|
26
|
+
description: str
|
|
27
|
+
representative_queries: tuple[str, ...]
|
|
28
|
+
score: float # cosine similarity in [-1, 1]; higher is a better task match
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def _cosine(a: list[float], b: list[float]) -> float:
|
|
32
|
+
dot = sum(x * y for x, y in zip(a, b))
|
|
33
|
+
na = math.sqrt(sum(x * x for x in a))
|
|
34
|
+
nb = math.sqrt(sum(y * y for y in b))
|
|
35
|
+
return 0.0 if na == 0.0 or nb == 0.0 else dot / (na * nb)
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def _match_text(spec) -> str:
|
|
39
|
+
rq = " ".join(spec.representative_queries or ())
|
|
40
|
+
return f"{spec.display_name}. {spec.description} {rq}".strip()
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def discover(query: str, *, resources: WorkspaceHandle, kind: Optional[str] = None, k: int = 8) -> list[Discovered]:
|
|
44
|
+
"""Rank the live ARD catalog by semantic match to `query`; return the top `k` (optionally filtered to one
|
|
45
|
+
`kind`: `subgraph` / `model` / `function` / `agent_skill` / `mcp_tool`). Embedding-based, via the workspace's
|
|
46
|
+
query embedder (BGE-M3, the same space retrieval uses). Returns `[]` when the (filtered) catalog is empty; raises
|
|
47
|
+
`RuntimeError` if the workspace has no query embedder available (discovery needs one)."""
|
|
48
|
+
from rag_wright.capabilities.manifests import MANIFEST_SPECS
|
|
49
|
+
|
|
50
|
+
specs = [s for s in MANIFEST_SPECS.values() if kind is None or s.kind == kind]
|
|
51
|
+
if not specs:
|
|
52
|
+
return []
|
|
53
|
+
embedder = getattr(resources, "_embedder", None)
|
|
54
|
+
if embedder is None:
|
|
55
|
+
raise RuntimeError("discover() needs a query embedder; none is available on this workspace")
|
|
56
|
+
dense, _sparse = embedder.encode_batch([query] + [_match_text(s) for s in specs])
|
|
57
|
+
query_vec, cand_vecs = dense[0], dense[1:]
|
|
58
|
+
ranked = sorted(
|
|
59
|
+
zip(specs, cand_vecs), key=lambda sv: _cosine(query_vec, sv[1]), reverse=True
|
|
60
|
+
)
|
|
61
|
+
return [
|
|
62
|
+
Discovered(
|
|
63
|
+
slug=s.slug,
|
|
64
|
+
kind=s.kind,
|
|
65
|
+
description=s.description,
|
|
66
|
+
representative_queries=tuple(s.representative_queries or ()),
|
|
67
|
+
score=round(_cosine(query_vec, v), 4),
|
|
68
|
+
)
|
|
69
|
+
for s, v in ranked[:k]
|
|
70
|
+
]
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
"""EP-API-2b / EP-API-6 (ADR-0117): building a document to feed the ingestion capability.
|
|
2
|
+
|
|
3
|
+
`source_document(document_id, text=...)` makes an engine `SourceDocument` from plain text (the simplest path, no
|
|
4
|
+
parser needed). `parse_document` / `aparse_document` make a STRUCTURE-BEARING `SourceDocument` from a file PATH via
|
|
5
|
+
the engine's real docling parse (`.parsed` set), so the full ingestion pipeline incl docling runs through the API --
|
|
6
|
+
not only the text fallback. Heavy deps imported lazily so `import rag_wright.api` stays light."""
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
from typing import Any
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def source_document(document_id: str, *, text: str) -> Any:
|
|
14
|
+
"""A text-only `SourceDocument` (`source_doc_id`, `text`) to pass as the `document` input of the
|
|
15
|
+
`contract_ingestion_pipeline` capability. The id should be a canonical, delimiter-safe source-doc id."""
|
|
16
|
+
from rag_wright.capabilities.document_parse import SourceDocument
|
|
17
|
+
|
|
18
|
+
return SourceDocument(source_doc_id=document_id, text=text)
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def parse_document(document_id: str, path: Any, *, cache_dir: Any, metadata: dict | None = None) -> Any:
|
|
22
|
+
"""Docling-parse the file at `path` ONCE (content-hash gated + cached under `cache_dir`) into a
|
|
23
|
+
structure-bearing `SourceDocument` -- `.parsed` carries the `DoclingDocument` so the chunker's structural pass
|
|
24
|
+
fires on real headings, and `.text` holds the flattened text. This is the PDF/DOCX/HTML/MD ingest entry point
|
|
25
|
+
of the engine API; pass the result as the `document` input of `contract_ingestion_pipeline`. The docling parse
|
|
26
|
+
blocks; use `aparse_document` on an event loop."""
|
|
27
|
+
from rag_wright.capabilities.document_parse import parsed_source_document
|
|
28
|
+
|
|
29
|
+
p = Path(path)
|
|
30
|
+
return parsed_source_document(document_id, p.name, p.read_bytes(), cache_dir=cache_dir, metadata=metadata)
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
async def aparse_document(document_id: str, path: Any, *, cache_dir: Any, metadata: dict | None = None) -> Any:
|
|
34
|
+
"""The async, deadline-bounded twin of `parse_document` (ADR-0057): runs the docling parse off the event loop
|
|
35
|
+
so a hand-built async ingest can parse a document into a structure-bearing `SourceDocument` without blocking."""
|
|
36
|
+
from rag_wright.capabilities.document_parse import aparsed_source_document
|
|
37
|
+
|
|
38
|
+
p = Path(path)
|
|
39
|
+
return await aparsed_source_document(document_id, p.name, p.read_bytes(), cache_dir=cache_dir, metadata=metadata)
|
rag_wright/api/ids.py
ADDED
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
"""EP-API-3 (ADR-0117): engine id/format accessors.
|
|
2
|
+
|
|
3
|
+
So a product never parses engine id strings or storage formats itself (today the seam reimplements these, drifting
|
|
4
|
+
from the engine). Heavy deps are imported lazily so `import rag_wright.api` stays light (progressive loading)."""
|
|
5
|
+
from __future__ import annotations
|
|
6
|
+
|
|
7
|
+
from typing import Optional
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def document_of(entity_id: str) -> str:
|
|
11
|
+
"""The source-document id embedded in a span/chunk/clause id (`<source_doc_id>:<idx>:<hash>` -> the first,
|
|
12
|
+
delimiter-safe segment). Empty in -> empty out."""
|
|
13
|
+
from rag_wright.store.arcadedb import _doc_id_of
|
|
14
|
+
|
|
15
|
+
return _doc_id_of(entity_id)
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def id_source(requirement_id: str) -> str:
|
|
19
|
+
"""The source/policy of a Requirement id (`<source>:<section>:<hash>` -> the first segment; `source` is
|
|
20
|
+
delimiter-safe via `canonical_source_doc_id`, so this is the same first-segment rule as `document_of`)."""
|
|
21
|
+
from rag_wright.store.arcadedb import _doc_id_of
|
|
22
|
+
|
|
23
|
+
return _doc_id_of(requirement_id)
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def decode_bbox(raw: Optional[str]) -> Optional[tuple]:
|
|
27
|
+
"""Decode the engine's best-effort bounding box (stored as a JSON `[l,t,r,b]` string) to a `(l, t, r, b)` tuple,
|
|
28
|
+
or None. The single canonical decoder (retires the product seam's copy)."""
|
|
29
|
+
from rag_wright.capabilities.highlight_serve import _decode_bbox
|
|
30
|
+
|
|
31
|
+
return _decode_bbox(raw)
|
rag_wright/api/invoke.py
ADDED
|
@@ -0,0 +1,99 @@
|
|
|
1
|
+
"""EP-API-2 / EP-CORE-2 (ADR-0117, ADR-0118): the engine's capability invoker -- an adapter-free ARD *client*.
|
|
2
|
+
|
|
3
|
+
Progressive loading, like an agent holding skill name+metadata and loading the full thing on demand:
|
|
4
|
+
* a LIGHT index (`slug -> manifest spec`) is built ONCE from the ARD manifest specs (no capability IMPLEMENTATION
|
|
5
|
+
is imported) -- the "cards" layer, for discovery + knowing what exists;
|
|
6
|
+
* on invoke, the capability's `impl_ref` (a "module:attr" vendor-extension pointer on its manifest) is imported
|
|
7
|
+
LAZILY via `capability_impl` and called -- so there is NO central engine-owned adapter dict. A developer who
|
|
8
|
+
registers a capability with an `impl_ref` makes it invocable with zero engine edits (ADR-0118).
|
|
9
|
+
The ARD registry stays metadata (ADR-0003: a string pointer, never a callable). Each per-kind invoker takes an
|
|
10
|
+
opaque `WorkspaceHandle` as resources and wraps the call in a trace span; model usage/cost is captured by the
|
|
11
|
+
CALLER's `api.measure_usage()` (EP-API-5). Subgraph hardening (retry/dead-letter) comes from the LangGraph scaffold."""
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import asyncio
|
|
15
|
+
from typing import Any
|
|
16
|
+
|
|
17
|
+
from rag_wright.capabilities.invoke import capability_impl
|
|
18
|
+
from rag_wright.api.workspace import WorkspaceHandle
|
|
19
|
+
from rag_wright.models.tracing import traced_step
|
|
20
|
+
|
|
21
|
+
def _index() -> dict[str, Any]:
|
|
22
|
+
"""The light capability index (`slug -> manifest spec`): the LIVE runtime ARD catalog the developer populates
|
|
23
|
+
(EP-CORE-3). Read fresh each call so a just-registered capability is seen; carries only metadata (kind/
|
|
24
|
+
description/impl_ref/representative_queries), never an implementation."""
|
|
25
|
+
from rag_wright.capabilities.manifests import MANIFEST_SPECS
|
|
26
|
+
|
|
27
|
+
return MANIFEST_SPECS
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def capability_index() -> dict[str, dict]:
|
|
31
|
+
"""Public discovery index: `{slug: {kind, description}}` for every catalogued capability (the 'cards')."""
|
|
32
|
+
return {slug: {"kind": s.kind, "description": s.description} for slug, s in _index().items()}
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _validate(name: str, kind: str) -> None:
|
|
36
|
+
"""Validate a capability name against the ARD catalog (name known, kind matches). Raises KeyError/ValueError."""
|
|
37
|
+
idx = _index()
|
|
38
|
+
if name not in idx:
|
|
39
|
+
raise KeyError(f"unknown capability {name!r} (not in the ARD catalog)")
|
|
40
|
+
if idx[name].kind != kind:
|
|
41
|
+
raise ValueError(f"capability {name!r} is kind {idx[name].kind!r}, not {kind!r}")
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
async def ainvoke_subgraph(name: str, inputs: dict, *, resources: WorkspaceHandle) -> Any:
|
|
45
|
+
"""Invoke a subgraph-kind capability by name over the workspace, inside a trace span. The implementation is
|
|
46
|
+
resolved lazily from the manifest `impl_ref` (no central adapter dict). Retry/dead-letter comes from the
|
|
47
|
+
LangGraph scaffold the subgraph is built on; usage is captured by the caller's `measure_usage()` (EP-API-5)."""
|
|
48
|
+
_validate(name, "subgraph")
|
|
49
|
+
factory = capability_impl(name) # resolves impl_ref -> the co-located `ainvoke(resources, inputs)`
|
|
50
|
+
with traced_step(f"invoke:{name}"):
|
|
51
|
+
return await factory(resources, inputs)
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def invoke_model(name: str, inputs: dict, *, resources: WorkspaceHandle) -> Any:
|
|
55
|
+
"""Invoke a model-kind capability by name (SYNCHRONOUSLY). Validated against the ARD catalog, then resolved via
|
|
56
|
+
`impl_ref` and dispatched -- the SAME path the ingestion pipeline routes through (one production path, no second
|
|
57
|
+
hand-built fleet). `resources` is accepted for API uniformity but model capabilities are store-independent.
|
|
58
|
+
Usage is the caller's `measure_usage()` scope (EP-API-5). A model impl may be async (I/O-bound, e.g. an
|
|
59
|
+
LLM-backed cap) -- those cannot be invoked here; call `ainvoke_model` instead (we refuse rather than silently
|
|
60
|
+
return an un-awaited coroutine)."""
|
|
61
|
+
_validate(name, "model")
|
|
62
|
+
factory = capability_impl(name)
|
|
63
|
+
if asyncio.iscoroutinefunction(factory):
|
|
64
|
+
raise TypeError(
|
|
65
|
+
f"capability {name!r} has an async impl; call ainvoke_model() instead of invoke_model()")
|
|
66
|
+
with traced_step(f"invoke:{name}"):
|
|
67
|
+
return factory(resources, inputs)
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
async def ainvoke_model(name: str, inputs: dict, *, resources: WorkspaceHandle,
|
|
71
|
+
sem: asyncio.Semaphore | None = None) -> Any:
|
|
72
|
+
"""Invoke a model-kind capability by name, ASYNCHRONOUSLY -- the async surface for model caps (the subgraph
|
|
73
|
+
legs already have `ainvoke_subgraph`). A model impl is one of two shapes, and this routes each honestly:
|
|
74
|
+
* SYNC (CPU-bound local inference -- a classifier/XGBoost fleet): run OFF the event loop in a worker thread
|
|
75
|
+
(`asyncio.to_thread`), so a big batch never blocks the loop;
|
|
76
|
+
* ASYNC (I/O-bound -- an LLM-backed cap calling OpenRouter or a local vLLM client): AWAITED directly, so the
|
|
77
|
+
I/O concurrency is real (not a thread wrapping a blocking call).
|
|
78
|
+
`sem` (an `asyncio.Semaphore`) bounds total in-flight work when a caller fans out a batch -- the same
|
|
79
|
+
backpressure the ingestion pipeline applies via `adispatch_model`. Usage is the caller's `measure_usage()`
|
|
80
|
+
scope (EP-API-5)."""
|
|
81
|
+
_validate(name, "model")
|
|
82
|
+
factory = capability_impl(name)
|
|
83
|
+
|
|
84
|
+
async def _run() -> Any:
|
|
85
|
+
with traced_step(f"invoke:{name}"):
|
|
86
|
+
if asyncio.iscoroutinefunction(factory):
|
|
87
|
+
return await factory(resources, inputs)
|
|
88
|
+
return await asyncio.to_thread(factory, resources, inputs)
|
|
89
|
+
|
|
90
|
+
if sem is None:
|
|
91
|
+
return await _run()
|
|
92
|
+
async with sem:
|
|
93
|
+
return await _run()
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def _invocable_names() -> dict[str, str]:
|
|
97
|
+
"""{name: kind} for every capability that declares an `impl_ref` -- the drift guard asserts each resolves to a
|
|
98
|
+
callable of the declared kind (replaces the old central-adapter-dict binding)."""
|
|
99
|
+
return {slug: s.kind for slug, s in _index().items() if getattr(s, "impl_ref", None)}
|
rag_wright/api/kg.py
ADDED
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
"""EP-API-3 (ADR-0117): scoped KG access over the opaque workspace handle.
|
|
2
|
+
|
|
3
|
+
Thin wrappers that delegate to the handle's resolved store, so a product does generic typed-KG reads/writes (and
|
|
4
|
+
span-position lookups for citations) WITHOUT importing `ArcadeDBStore` or touching `ws._store`. The generic
|
|
5
|
+
`kg_read`/`kg_write` primitives already live on the store (DD-1a/b); this exposes them on the API."""
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
from typing import Any, Optional
|
|
9
|
+
|
|
10
|
+
from rag_wright.api.workspace import WorkspaceHandle
|
|
11
|
+
|
|
12
|
+
# the span-position fields a citation needs (NOT the dense vector) -- offsets + page/bbox provenance
|
|
13
|
+
_SPAN_POSITION_FIELDS = [
|
|
14
|
+
"span_id", "parent_chunk_id", "span_index", "text", "function", "contract_id",
|
|
15
|
+
"doc_start", "doc_end", "pages", "bbox",
|
|
16
|
+
]
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def kg_read(ws: WorkspaceHandle, node_type: str, *, where: Optional[dict] = None, fields: Optional[list] = None,
|
|
20
|
+
distinct: Optional[str] = None, order_by: Optional[str] = None, limit: Optional[int] = None) -> list[dict]:
|
|
21
|
+
"""Read typed nodes of `node_type` from the workspace (see `Store.kg_read`). Equality/`IN` filters, projection,
|
|
22
|
+
distinct, order, limit; an empty list `where` value is scope-to-nothing -> `[]`."""
|
|
23
|
+
return ws._store.kg_read(node_type, where=where, fields=fields, distinct=distinct, order_by=order_by, limit=limit)
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def kg_write(ws: WorkspaceHandle, nodes: list, edges: Any = ()) -> None:
|
|
27
|
+
"""Upsert typed `nodes` + create typed `edges` in one transaction (see `Store.kg_write`). `nodes`/`edges` are
|
|
28
|
+
`KgNode`/`KgEdge` (from `rag_wright.store.seam`); the store encodes each field per its pack-declared type."""
|
|
29
|
+
ws._store.kg_write(nodes, edges)
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def kg_edges(ws: WorkspaceHandle, from_type: Optional[str] = None, *, where: Optional[dict] = None,
|
|
33
|
+
key_range: Optional[tuple] = None, direction: str = "out", edge_type: Optional[str] = None,
|
|
34
|
+
edge_where: Optional[dict] = None, target_where: Optional[dict] = None,
|
|
35
|
+
select: dict) -> list[dict]:
|
|
36
|
+
"""Generic edge TRAVERSAL over the workspace (see `Store.kg_edges`): node-start out/in MATCH (by `where`
|
|
37
|
+
equality/membership or a contract-scope `key_range`) or a direct edge scan; `select` projects `c.`/`e.`/`v.`
|
|
38
|
+
expressions. The engine's relational/graph primitive on the API, so a domain's graph query never touches
|
|
39
|
+
`ws._store`. (`NOT_NULL` for a presence filter is `rag_wright.store.seam.NOT_NULL`.)"""
|
|
40
|
+
return ws._store.kg_edges(from_type, where=where, key_range=key_range, direction=direction,
|
|
41
|
+
edge_type=edge_type, edge_where=edge_where, target_where=target_where, select=select)
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def entities_by_name(ws: WorkspaceHandle, name: str) -> list[dict]:
|
|
45
|
+
"""Resolve an entity NAME to every entity node it matches: `[{entity_id, name, entity_type}]` (the engine owns
|
|
46
|
+
the surface-form normalization, so variants collapse to one id). One name can match several nodes (a resolved
|
|
47
|
+
node + an unlinked ref sharing a clustering key) -- all are returned. `entity_id` is exactly the
|
|
48
|
+
`start_entity_id` a graph traversal takes. The engine's generic entity-lookup primitive on the API."""
|
|
49
|
+
return ws._store.entities_by_name(name)
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def span_positions(ws: WorkspaceHandle, document: str) -> list[dict]:
|
|
53
|
+
"""Every span of `document` with its position provenance (doc offsets, pages, DECODED bbox), ordered by document
|
|
54
|
+
position. The engine MECHANISM behind a product's citation/highlight types -- the product wraps these rows into
|
|
55
|
+
its own presentation type (e.g. `SpanLocation`)."""
|
|
56
|
+
from rag_wright.api.ids import decode_bbox
|
|
57
|
+
from rag_wright.store.arcadedb import SPAN_TYPE
|
|
58
|
+
|
|
59
|
+
rows = ws._store.kg_read(SPAN_TYPE, fields=_SPAN_POSITION_FIELDS, where={"contract_id": document},
|
|
60
|
+
order_by="doc_start")
|
|
61
|
+
return [{**r, "bbox": decode_bbox(r.get("bbox"))} for r in rows]
|
rag_wright/api/mcp.py
ADDED
|
@@ -0,0 +1,94 @@
|
|
|
1
|
+
"""EP-RT-2 (ADR-0117): the generic capability->MCP adapter.
|
|
2
|
+
|
|
3
|
+
One function, `build_capability_mcp(slug, *, resources)`, exposes ANY catalogued invokable capability as a FastMCP
|
|
4
|
+
server with NO bespoke per-capability server code -- the tool's name/title/description are read from the ARD
|
|
5
|
+
manifest, and its handler dispatches through the engine invoker (`ainvoke_subgraph`/`invoke_model`) over the bound
|
|
6
|
+
workspace. This is the zero-boilerplate path: a new-domain product (or a new engine capability) gets a discoverable
|
|
7
|
+
MCP tool for free, the moment the capability is in the ARD catalog.
|
|
8
|
+
|
|
9
|
+
The four hand-written servers in `rag_wright/mcp/` stay: they offer a CURATED, typed tool signature + description for
|
|
10
|
+
the Tier-1 legs. This generic adapter is the complement -- it takes a single opaque `inputs` dict (the capability's
|
|
11
|
+
own input contract, dispatched straight to the invoker) rather than a per-capability typed signature, because the
|
|
12
|
+
ARD manifest carries no JSON input schema to derive one from.
|
|
13
|
+
|
|
14
|
+
Only `subgraph` and `model` kinds are supported (the kinds the invoker can dispatch). An `mcp_tool` is already an
|
|
15
|
+
MCP tool; `function`/`agent_skill` have no invoker adapter yet (EP-API-2c) -- all rejected with a clear error.
|
|
16
|
+
|
|
17
|
+
Store binding is server-side via the `WorkspaceHandle` (issue 0035): the calling agent never supplies a tenant or
|
|
18
|
+
store; the workspace is bound when the server is built. Multi-tenant request routing is a product concern (a product
|
|
19
|
+
opens one workspace per corpus; `corpus` = the backend db name)."""
|
|
20
|
+
from __future__ import annotations
|
|
21
|
+
|
|
22
|
+
import asyncio
|
|
23
|
+
from typing import Any
|
|
24
|
+
|
|
25
|
+
from fastmcp import FastMCP
|
|
26
|
+
|
|
27
|
+
from rag_wright.api.invoke import _index, ainvoke_subgraph, invoke_model
|
|
28
|
+
from rag_wright.api.workspace import WorkspaceHandle
|
|
29
|
+
|
|
30
|
+
_INVOKABLE_KINDS = ("subgraph", "model")
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def _jsonable(obj: Any) -> Any:
|
|
34
|
+
"""Normalize an invoker result (a LangGraph state dict, a Pydantic model, or a list of them) to JSON-safe
|
|
35
|
+
data -- the structured content the calling agent receives."""
|
|
36
|
+
if hasattr(obj, "model_dump"):
|
|
37
|
+
return obj.model_dump(mode="json")
|
|
38
|
+
if isinstance(obj, dict):
|
|
39
|
+
return {k: _jsonable(v) for k, v in obj.items()}
|
|
40
|
+
if isinstance(obj, (list, tuple)):
|
|
41
|
+
return [_jsonable(v) for v in obj]
|
|
42
|
+
return obj
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def build_capability_mcp(slug: str, *, resources: WorkspaceHandle, name: str | None = None) -> FastMCP:
|
|
46
|
+
"""Build a FastMCP server exposing the catalogued capability `slug` as one MCP tool, driven by its ARD manifest.
|
|
47
|
+
|
|
48
|
+
The tool is named for the capability (its canonical slug -- stable + discoverable), titled with the manifest
|
|
49
|
+
display name, and described by the manifest description. It takes one `inputs` dict (the capability's input
|
|
50
|
+
contract) and dispatches through the engine invoker over `resources` (the bound workspace), returning the result
|
|
51
|
+
as JSON. Only `subgraph`/`model` kinds are supported.
|
|
52
|
+
|
|
53
|
+
Raises `KeyError` for an unknown slug; `ValueError` for a non-invokable kind."""
|
|
54
|
+
idx = _index()
|
|
55
|
+
if slug not in idx:
|
|
56
|
+
raise KeyError(f"unknown capability {slug!r} (not in the ARD catalog)")
|
|
57
|
+
spec = idx[slug]
|
|
58
|
+
if spec.kind not in _INVOKABLE_KINDS:
|
|
59
|
+
raise ValueError(
|
|
60
|
+
f"capability {slug!r} is kind {spec.kind!r}; only {_INVOKABLE_KINDS} are exposable via the generic "
|
|
61
|
+
f"MCP adapter (an mcp_tool is already an MCP tool; function/agent-skill have no invoker adapter yet)")
|
|
62
|
+
|
|
63
|
+
mcp: FastMCP = FastMCP(
|
|
64
|
+
name=name or f"rag-wright-{slug.replace('_', '-')}",
|
|
65
|
+
instructions=(
|
|
66
|
+
f"{spec.description}\n\nCall `{slug}` with an `inputs` object carrying the capability's inputs. "
|
|
67
|
+
f"Example queries: {'; '.join(spec.representative_queries)}."),
|
|
68
|
+
)
|
|
69
|
+
kind = spec.kind
|
|
70
|
+
|
|
71
|
+
@mcp.tool(name=slug, title=spec.display_name, description=spec.description)
|
|
72
|
+
async def _invoke_capability(inputs: dict) -> dict:
|
|
73
|
+
"""Invoke the capability over the bound workspace.
|
|
74
|
+
|
|
75
|
+
Args:
|
|
76
|
+
inputs: The capability's input contract (e.g. {"query": ...} for a retrieval leg, or
|
|
77
|
+
{"text": ..., "functions": [...]} for a classifier). Dispatched straight to the engine invoker.
|
|
78
|
+
|
|
79
|
+
Returns:
|
|
80
|
+
{"result": <the capability's output as JSON>} -- a uniform envelope (a subgraph's state dict or a
|
|
81
|
+
model's soft-tag list both land under `result`), since the output contract varies by capability.
|
|
82
|
+
"""
|
|
83
|
+
if kind == "subgraph":
|
|
84
|
+
out = await ainvoke_subgraph(slug, inputs, resources=resources)
|
|
85
|
+
else: # model -- sync invoker; offload so the event loop is not blocked by local inference
|
|
86
|
+
out = await asyncio.to_thread(invoke_model, slug, inputs, resources=resources)
|
|
87
|
+
return {"result": _jsonable(out)}
|
|
88
|
+
|
|
89
|
+
return mcp
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def serve_capability_mcp(slug: str, *, resources: WorkspaceHandle, transport: str = "stdio") -> None:
|
|
93
|
+
"""Serve one catalogued capability as an MCP tool over `transport` (default stdio, so an agent can spawn it)."""
|
|
94
|
+
build_capability_mcp(slug, resources=resources).run(transport=transport)
|
rag_wright/api/usage.py
ADDED
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
"""EP-API-5 (ADR-0117, ADR-0105): usage/cost on the engine API surface.
|
|
2
|
+
|
|
3
|
+
A product measures the model usage an engine call incurs -- calls, input/output tokens, known cost (plus the count
|
|
4
|
+
of calls the backend surfaced NO cost for), total latency, and a per-model breakdown -- by wrapping the call in
|
|
5
|
+
`measure_usage()`. It is the public face of the engine's in-band usage accounting (`models/usage.usage_scope`), so
|
|
6
|
+
the product reads stats without reaching into engine internals. Ambient (a `ContextVar`), additive across nesting
|
|
7
|
+
(an outer scope totals everything; inner scopes attribute their slice), and carried into worker threads/executors.
|
|
8
|
+
|
|
9
|
+
with engine.measure_usage() as usage:
|
|
10
|
+
out = await engine.ainvoke_subgraph("intra_document_qa", {...}, resources=ws)
|
|
11
|
+
usage.calls, usage.input_tokens, usage.output_tokens, usage.cost_usd, usage.calls_without_cost,
|
|
12
|
+
usage.latency_ms_total, usage.by_model # {model_id: ModelUsage(...)}
|
|
13
|
+
"""
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
from contextlib import contextmanager
|
|
17
|
+
from typing import Iterator
|
|
18
|
+
|
|
19
|
+
from rag_wright.models.usage import ModelUsage, UsageTotals, usage_scope
|
|
20
|
+
|
|
21
|
+
__all__ = ["measure_usage", "UsageTotals", "ModelUsage"]
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
@contextmanager
|
|
25
|
+
def measure_usage() -> Iterator[UsageTotals]:
|
|
26
|
+
"""Accumulate the model usage of every engine call made inside the block; read the returned `UsageTotals`
|
|
27
|
+
after it. Nesting is additive, so a task-level scope totals everything while an inner per-call scope attributes
|
|
28
|
+
its slice. Capturing is opt-in: with no active scope, the engine records usage nowhere (zero overhead)."""
|
|
29
|
+
with usage_scope() as totals:
|
|
30
|
+
yield totals
|
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
"""EP-API-1 (ADR-0117): `open_workspace` + the opaque `WorkspaceHandle`.
|
|
2
|
+
|
|
3
|
+
The product names a `corpus` (the logical KG = the backend database name; the product maps its own tenant -> db-name,
|
|
4
|
+
so tenancy stays product-owned) and gets back an opaque handle that resolves + caches the store (and lazily the
|
|
5
|
+
embedder) from the `EngineConfig`. The handle is what the engine's per-kind invokers (EP-API-2) take as resources.
|
|
6
|
+
The product never imports `ArcadeDBStore`/`query_embedder` and the handle exposes no public store accessor."""
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import logging
|
|
10
|
+
from typing import Any, Optional
|
|
11
|
+
|
|
12
|
+
from rag_wright.api.config import EngineConfig
|
|
13
|
+
from rag_wright.models.profiles import ModelRole, model_for
|
|
14
|
+
|
|
15
|
+
logger = logging.getLogger(__name__)
|
|
16
|
+
|
|
17
|
+
# Cached per (backend, host, port, corpus) -- one resolved workspace per customer DB (absorbs the product's
|
|
18
|
+
# per-customer graph/store cache). Process-local; the embedder is lazy + per-handle.
|
|
19
|
+
_WORKSPACES: dict[tuple, "WorkspaceHandle"] = {}
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
class WorkspaceHandle:
|
|
23
|
+
"""An opaque handle to a resolved engine workspace. Public surface: `model_id(role)`. The resolved store +
|
|
24
|
+
embedder are engine-internal (`_store` / `_embedder`), used by the invokers -- NOT a product accessor."""
|
|
25
|
+
|
|
26
|
+
def __init__(self, store: Any, config: EngineConfig, corpus: str) -> None:
|
|
27
|
+
self._store = store
|
|
28
|
+
self._config = config
|
|
29
|
+
self._corpus = corpus
|
|
30
|
+
self.__embedder: Any = None
|
|
31
|
+
self.__embedder_built = False
|
|
32
|
+
|
|
33
|
+
@property
|
|
34
|
+
def _embedder(self) -> Any:
|
|
35
|
+
"""The query embedder for this workspace, built lazily on first use (so an ingest-only workspace -- or a
|
|
36
|
+
test -- needs no embedder backend). Best-effort: unavailable -> None (the consumer surfaces it)."""
|
|
37
|
+
if not self.__embedder_built:
|
|
38
|
+
self.__embedder = _build_embedder(self._config)
|
|
39
|
+
self.__embedder_built = True
|
|
40
|
+
return self.__embedder
|
|
41
|
+
|
|
42
|
+
def model_id(self, role: ModelRole) -> str:
|
|
43
|
+
"""Resolve a model role to its id: the `EngineConfig.models` override wins, else the profile default."""
|
|
44
|
+
key = role.value if isinstance(role, ModelRole) else str(role)
|
|
45
|
+
return self._config.models.get(key) or model_for(role)
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def open_workspace(config: EngineConfig, *, corpus: str, reset: bool = False) -> WorkspaceHandle:
|
|
49
|
+
"""Resolve (and cache) the workspace for `corpus` (the backend database name) from `config`. Ensures the schema.
|
|
50
|
+
Returns an opaque `WorkspaceHandle`. `reset=True` drops + recreates the database (test/clean-slate) and bypasses
|
|
51
|
+
the cache."""
|
|
52
|
+
sc = config.store
|
|
53
|
+
cache_key = (sc.backend, sc.host, sc.port, corpus)
|
|
54
|
+
if not reset and cache_key in _WORKSPACES:
|
|
55
|
+
return _WORKSPACES[cache_key]
|
|
56
|
+
store = _build_store(config, corpus, reset=reset)
|
|
57
|
+
store.ensure_schema()
|
|
58
|
+
handle = WorkspaceHandle(store, config, corpus)
|
|
59
|
+
_WORKSPACES[cache_key] = handle
|
|
60
|
+
return handle
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def _build_store(config: EngineConfig, corpus: str, *, reset: bool) -> Any:
|
|
64
|
+
sc = config.store
|
|
65
|
+
if sc.backend != "arcadedb":
|
|
66
|
+
raise ValueError(f"unsupported store backend: {sc.backend!r}") # a new backend plugs in here (DD)
|
|
67
|
+
from rag_wright.store.arcadedb import ArcadeDBStore
|
|
68
|
+
|
|
69
|
+
return ArcadeDBStore.from_config(sc.host, sc.port, sc.user, sc.password, database=corpus,
|
|
70
|
+
protocol=sc.protocol, reset=reset, pack_ttl=config.pack)
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def _build_embedder(config: EngineConfig) -> Optional[Any]:
|
|
74
|
+
"""The workspace's QUERY embedder, selected by the `text` embedding profile (EP-API-4b; default bge-m3). An
|
|
75
|
+
unknown profile raises; an unavailable backend degrades to None (the retrieval consumer surfaces it)."""
|
|
76
|
+
from rag_wright.capabilities.embedding_profiles import build_query_embedder
|
|
77
|
+
|
|
78
|
+
profile = config.embeddings.get("text", "bge-m3")
|
|
79
|
+
try:
|
|
80
|
+
return build_query_embedder(profile)
|
|
81
|
+
except ValueError:
|
|
82
|
+
raise # an unknown profile is a config error, not a transient backend failure
|
|
83
|
+
except Exception: # noqa: BLE001 - embedder backend unavailable -> None; the retrieval consumer surfaces it
|
|
84
|
+
logger.warning("engine_embedder_unavailable", exc_info=True)
|
|
85
|
+
return None
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
"""The FR-C capability catalog, each built and registered under its FR-C name.
|
|
2
|
+
|
|
3
|
+
FR-C.1 Parsing (Docling), FR-C.2 Embedding (BGE-M3), FR-C.3 Hybrid search (ArcadeDB),
|
|
4
|
+
FR-C.4 Reranking (BGE-reranker), FR-C.5 Graph query (ArcadeDB), FR-C.6 Graph extraction,
|
|
5
|
+
FR-C.7 Entity resolution, FR-C.8 Ontology and registry derivation,
|
|
6
|
+
FR-C.9 Reasoning, generation, and vision-to-text (Gemma 4), FR-C.10 RLM skill.
|
|
7
|
+
Plus the ingestion-side (FR-I) and query-side (FR-Q) capability behavior.
|
|
8
|
+
"""
|