rag-wright 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- rag_wright/__init__.py +13 -0
- rag_wright/api/__init__.py +33 -0
- rag_wright/api/config.py +59 -0
- rag_wright/api/discover.py +70 -0
- rag_wright/api/documents.py +39 -0
- rag_wright/api/ids.py +31 -0
- rag_wright/api/invoke.py +99 -0
- rag_wright/api/kg.py +61 -0
- rag_wright/api/mcp.py +94 -0
- rag_wright/api/usage.py +30 -0
- rag_wright/api/workspace.py +85 -0
- rag_wright/capabilities/__init__.py +8 -0
- rag_wright/capabilities/answer_generator.py +427 -0
- rag_wright/capabilities/ard.py +286 -0
- rag_wright/capabilities/assertion_extraction.py +79 -0
- rag_wright/capabilities/chunk_read.py +58 -0
- rag_wright/capabilities/chunk_write.py +163 -0
- rag_wright/capabilities/claim_extraction.py +153 -0
- rag_wright/capabilities/clause_exception_linking.py +117 -0
- rag_wright/capabilities/compliance_judgment.py +322 -0
- rag_wright/capabilities/compliance_store.py +87 -0
- rag_wright/capabilities/contract_kg_serve.py +156 -0
- rag_wright/capabilities/contract_kg_store.py +251 -0
- rag_wright/capabilities/dg_extraction.py +585 -0
- rag_wright/capabilities/disambiguation.py +163 -0
- rag_wright/capabilities/document_parse.py +87 -0
- rag_wright/capabilities/document_scope.py +49 -0
- rag_wright/capabilities/embedding.py +164 -0
- rag_wright/capabilities/embedding_profiles.py +43 -0
- rag_wright/capabilities/entity_resolution.py +154 -0
- rag_wright/capabilities/fusion.py +64 -0
- rag_wright/capabilities/graph_extraction.py +243 -0
- rag_wright/capabilities/graph_query.py +73 -0
- rag_wright/capabilities/graph_storage.py +111 -0
- rag_wright/capabilities/highlight_serve.py +142 -0
- rag_wright/capabilities/hybrid_search.py +65 -0
- rag_wright/capabilities/invoke.py +31 -0
- rag_wright/capabilities/jev_decision.py +38 -0
- rag_wright/capabilities/manifests.py +872 -0
- rag_wright/capabilities/okf_navigate.py +456 -0
- rag_wright/capabilities/parsing.py +286 -0
- rag_wright/capabilities/property_boosted_retrieval.py +125 -0
- rag_wright/capabilities/query_function_classifier.py +94 -0
- rag_wright/capabilities/query_understanding.py +109 -0
- rag_wright/capabilities/registry.py +262 -0
- rag_wright/capabilities/remote_encoders.py +94 -0
- rag_wright/capabilities/requirement_extraction.py +247 -0
- rag_wright/capabilities/reranking.py +123 -0
- rag_wright/capabilities/retrieval_core.py +126 -0
- rag_wright/capabilities/rlm_chunking.py +808 -0
- rag_wright/capabilities/rlm_synthesis.py +316 -0
- rag_wright/capabilities/scan_quality.py +136 -0
- rag_wright/capabilities/span_relevance_judgment.py +191 -0
- rag_wright/capabilities/vision_to_text.py +85 -0
- rag_wright/capabilities/vlm_ocr.py +85 -0
- rag_wright/contracts/__init__.py +6 -0
- rag_wright/contracts/chunk.py +79 -0
- rag_wright/contracts/compliance.py +303 -0
- rag_wright/contracts/contract_meta.py +27 -0
- rag_wright/contracts/extraction.py +130 -0
- rag_wright/contracts/function.py +167 -0
- rag_wright/contracts/function_routing.py +91 -0
- rag_wright/contracts/highlight.py +74 -0
- rag_wright/contracts/identifiers.py +153 -0
- rag_wright/contracts/jurisdiction.py +96 -0
- rag_wright/contracts/ontology.py +142 -0
- rag_wright/contracts/property.py +201 -0
- rag_wright/contracts/provenance.py +78 -0
- rag_wright/contracts/query_intent.py +53 -0
- rag_wright/contracts/span.py +76 -0
- rag_wright/contracts/value_match.py +84 -0
- rag_wright/corpus/__init__.py +0 -0
- rag_wright/corpus/canonicalize.py +116 -0
- rag_wright/corpus/cuad.py +153 -0
- rag_wright/corpus/cuad_ingestion.py +72 -0
- rag_wright/corpus/document_parser.py +299 -0
- rag_wright/corpus/edgar.py +231 -0
- rag_wright/corpus/gcs_ingestion.py +120 -0
- rag_wright/corpus/http.py +110 -0
- rag_wright/corpus/selection.py +152 -0
- rag_wright/mcp/__init__.py +11 -0
- rag_wright/mcp/compliance_server.py +299 -0
- rag_wright/mcp/intra_document_qa_server.py +170 -0
- rag_wright/mcp/relational_qa_server.py +171 -0
- rag_wright/mcp/session_store.py +64 -0
- rag_wright/mcp/typed_property_retrieval_server.py +191 -0
- rag_wright/models/__init__.py +8 -0
- rag_wright/models/profiles.py +331 -0
- rag_wright/models/seam.py +497 -0
- rag_wright/models/tag_structured.py +285 -0
- rag_wright/models/tracing.py +179 -0
- rag_wright/models/usage.py +102 -0
- rag_wright/okf/__init__.py +11 -0
- rag_wright/okf/compile.py +292 -0
- rag_wright/okf/document.py +47 -0
- rag_wright/okf/enrich.py +176 -0
- rag_wright/okf/links.py +190 -0
- rag_wright/okf/lint.py +105 -0
- rag_wright/ontology/__init__.py +6 -0
- rag_wright/ontology/_generated_template_meta.py +60 -0
- rag_wright/ontology/_generated_vocab.py +52 -0
- rag_wright/ontology/clause_template.py +964 -0
- rag_wright/ontology/codegen.py +84 -0
- rag_wright/ontology/compliance_bridge.ttl +186 -0
- rag_wright/ontology/contract_bridge.ttl +2685 -0
- rag_wright/ontology/contract_taxonomy.py +24 -0
- rag_wright/ontology/derive.py +58 -0
- rag_wright/ontology/loader.py +435 -0
- rag_wright/ontology/packs/ftc_16cfr255.ttl +29 -0
- rag_wright/ontology/registry.py +87 -0
- rag_wright/ontology/template_introspect.py +100 -0
- rag_wright/py.typed +0 -0
- rag_wright/reference/__init__.py +2 -0
- rag_wright/reference/compliance.py +41 -0
- rag_wright/reference/contract_seam.py +123 -0
- rag_wright/skills/__init__.py +7 -0
- rag_wright/skills/claim_extraction/SKILL.md +47 -0
- rag_wright/skills/claim_extraction/__init__.py +1 -0
- rag_wright/skills/claim_extraction/template.py +50 -0
- rag_wright/skills/compliance_judgment/SKILL.md +59 -0
- rag_wright/skills/corpus_ingest/SKILL.md +106 -0
- rag_wright/skills/extraction_semantic_judge/SKILL.md +51 -0
- rag_wright/skills/extraction_semantic_judge/__init__.py +1 -0
- rag_wright/skills/generation/SKILL.md +64 -0
- rag_wright/skills/generation/__init__.py +1 -0
- rag_wright/skills/generic_compliance_judgment/SKILL.md +58 -0
- rag_wright/skills/okf_navigate/SKILL.md +137 -0
- rag_wright/skills/requirement_extraction/SKILL.md +47 -0
- rag_wright/skills/requirement_extraction/__init__.py +1 -0
- rag_wright/skills/requirement_extraction/template.py +50 -0
- rag_wright/skills/rlm/SKILL.md +186 -0
- rag_wright/skills/rlm/__init__.py +31 -0
- rag_wright/skills/rlm/agent.py +292 -0
- rag_wright/skills/span_relevance_judgment/SKILL.md +67 -0
- rag_wright/skills/vision_to_text/SKILL.md +36 -0
- rag_wright/skills/vision_to_text/__init__.py +1 -0
- rag_wright/spans/__init__.py +1 -0
- rag_wright/spans/boundary.py +78 -0
- rag_wright/spans/clause_function_classifier.py +490 -0
- rag_wright/spans/clause_kg_extractor.py +337 -0
- rag_wright/spans/cuad_labels.py +81 -0
- rag_wright/spans/dim_classifier.py +158 -0
- rag_wright/spans/dim_fleet.json +411 -0
- rag_wright/spans/function_classifier.py +77 -0
- rag_wright/spans/function_families.py +62 -0
- rag_wright/spans/hybrid_classifier.py +103 -0
- rag_wright/spans/legalbert_classifier.py +83 -0
- rag_wright/spans/model_capabilities.py +107 -0
- rag_wright/spans/new_function_labels.py +111 -0
- rag_wright/spans/page_map.py +68 -0
- rag_wright/spans/property_extractor.py +365 -0
- rag_wright/spans/property_grounding.py +182 -0
- rag_wright/spans/reclassify.py +77 -0
- rag_wright/spans/scarce_function_labels.py +105 -0
- rag_wright/spans/segment.py +341 -0
- rag_wright/spans/semantic_judge.py +197 -0
- rag_wright/spans/symbolic_validation.py +131 -0
- rag_wright/spans/tag_clause_extractor.py +182 -0
- rag_wright/store/__init__.py +6 -0
- rag_wright/store/arcadedb.py +1135 -0
- rag_wright/store/chunk_text.py +66 -0
- rag_wright/store/seam.py +213 -0
- rag_wright/subgraphs/__init__.py +0 -0
- rag_wright/subgraphs/async_ingestion.py +204 -0
- rag_wright/subgraphs/compliance_check.py +1042 -0
- rag_wright/subgraphs/compliance_ingestion.py +306 -0
- rag_wright/subgraphs/contract_ingestion_pipeline.py +999 -0
- rag_wright/subgraphs/graph_extraction.py +102 -0
- rag_wright/subgraphs/intra_document_qa.py +328 -0
- rag_wright/subgraphs/observability.py +140 -0
- rag_wright/subgraphs/query_constraint_extraction.py +73 -0
- rag_wright/subgraphs/relational_qa.py +165 -0
- rag_wright/subgraphs/requirement_extraction.py +137 -0
- rag_wright/subgraphs/scaffold.py +65 -0
- rag_wright/subgraphs/semantic_chunking.py +183 -0
- rag_wright/subgraphs/typed_clause_extraction.py +172 -0
- rag_wright/subgraphs/typed_property_retrieval.py +278 -0
- rag_wright/util/__init__.py +1 -0
- rag_wright/util/concurrent.py +153 -0
- rag_wright/util/spacy_model.py +45 -0
- rag_wright-0.1.0.dist-info/METADATA +168 -0
- rag_wright-0.1.0.dist-info/RECORD +184 -0
- rag_wright-0.1.0.dist-info/WHEEL +4 -0
- rag_wright-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
"""Hybrid search (FR-C.3, FR-Q.1): server-side RRF fusion of the dense and sparse legs.
|
|
2
|
+
|
|
3
|
+
Given a natural-language query, embed it once with the shared BGE-M3 embedder seam (T19) — dense and
|
|
4
|
+
sparse over the same query text — and hand both query vectors to the store's server-side hybrid
|
|
5
|
+
search (the T13 seam), which fuses the dense `vector.neighbors` leg and the sparse
|
|
6
|
+
`vector.sparseNeighbors` leg by Reciprocal Rank Fusion in ArcadeDB (proven end to end at T14) into one
|
|
7
|
+
ranked candidate list, honoring metadata filters. The fusion runs in the store, so there is no
|
|
8
|
+
cross-store join and no client-side re-ranking here; this is the query-side entry point the compiler
|
|
9
|
+
binds under FR-C.3. Reranking (the precision gate, FR-C.4/T22) consumes this list downstream.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
from typing import Optional
|
|
15
|
+
|
|
16
|
+
from pydantic import BaseModel
|
|
17
|
+
|
|
18
|
+
from rag_wright.capabilities.embedding import Embedder
|
|
19
|
+
from rag_wright.contracts.chunk import MetadataValue
|
|
20
|
+
from rag_wright.store.seam import Store
|
|
21
|
+
|
|
22
|
+
DEFAULT_K = 10 # candidate-list depth; the eval and the reranker (T22) can override per call
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
class Candidate(BaseModel):
|
|
26
|
+
"""One fused retrieval candidate: enough to cite it (`chunk_id`) and filter by source."""
|
|
27
|
+
|
|
28
|
+
model_config = {"frozen": True}
|
|
29
|
+
|
|
30
|
+
chunk_id: str
|
|
31
|
+
source_doc_id: str
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
class HybridSearchResult(BaseModel):
|
|
35
|
+
"""The hybrid-search capability's output: the RRF-ranked candidate list for a query (FR-C.3)."""
|
|
36
|
+
|
|
37
|
+
model_config = {"frozen": True}
|
|
38
|
+
|
|
39
|
+
query: str
|
|
40
|
+
candidates: list[Candidate] # RRF-ranked, best first
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def hybrid_search(
|
|
44
|
+
query: str,
|
|
45
|
+
*,
|
|
46
|
+
store: Store,
|
|
47
|
+
embedder: Embedder,
|
|
48
|
+
k: int = DEFAULT_K,
|
|
49
|
+
filters: Optional[dict[str, MetadataValue]] = None,
|
|
50
|
+
) -> HybridSearchResult:
|
|
51
|
+
"""Embed the query and return the RRF-fused, filter-honoring top-`k` candidate list.
|
|
52
|
+
|
|
53
|
+
The query is embedded once — dense and sparse over the same query text (a query has no
|
|
54
|
+
summary/full-text split, unlike a chunk) — through the injected `Embedder` seam, then fused
|
|
55
|
+
server-side by the store. The server's rank order is preserved; this capability does not re-rank.
|
|
56
|
+
"""
|
|
57
|
+
dense_query = embedder.encode_dense(query)
|
|
58
|
+
sparse_query = embedder.encode_sparse(query)
|
|
59
|
+
rows = store.hybrid_search(dense_query, sparse_query, k=k, filters=filters)
|
|
60
|
+
candidates = [
|
|
61
|
+
Candidate(chunk_id=row["chunk_id"], source_doc_id=row["source_doc_id"]) for row in rows
|
|
62
|
+
]
|
|
63
|
+
return HybridSearchResult(query=query, candidates=candidates)
|
|
64
|
+
|
|
65
|
+
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
"""EP-CORE-2 (ADR-0118): the capability implementation resolver -- the adapter-free heart of the ARD client.
|
|
2
|
+
|
|
3
|
+
A capability's manifest carries an `impl_ref` ("module:attr") pointing to its invoke factory, a
|
|
4
|
+
`(resources, inputs) -> result` callable (model factories ignore `resources`). `capability_impl(name)` reads the
|
|
5
|
+
light index, pulls the `impl_ref`, and imports it LAZILY -- so there is NO central engine-owned adapter dict, and a
|
|
6
|
+
developer registering a capability with an `impl_ref` makes it invocable with zero engine edits. ARD stays
|
|
7
|
+
metadata-only (ADR-0003): the registry holds a string pointer, never a callable; this module does the import.
|
|
8
|
+
|
|
9
|
+
Lives at the capabilities layer so BOTH the engine API invoker (`api/invoke.py`) and the ingestion pipeline's model
|
|
10
|
+
dispatch (`spans/model_capabilities.py`) resolve the same way."""
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
from importlib import import_module
|
|
14
|
+
from typing import Any, Callable
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def capability_impl(name: str) -> Callable[..., Any]:
|
|
18
|
+
"""Resolve a capability name to its invoke factory via the manifest `impl_ref`. Raises `KeyError` if the name is
|
|
19
|
+
unknown to the ARD catalog, `NotImplementedError` if it declares no `impl_ref` (not invokable by name)."""
|
|
20
|
+
from rag_wright.capabilities.manifests import MANIFEST_SPECS
|
|
21
|
+
|
|
22
|
+
spec = MANIFEST_SPECS.get(name)
|
|
23
|
+
if spec is None:
|
|
24
|
+
raise KeyError(f"unknown capability {name!r} (not in the ARD catalog)")
|
|
25
|
+
ref = spec.impl_ref
|
|
26
|
+
if not ref:
|
|
27
|
+
raise NotImplementedError(f"capability {name!r} declares no impl_ref (not invokable by name)")
|
|
28
|
+
module_path, _, attr = ref.partition(":")
|
|
29
|
+
if not module_path or not attr:
|
|
30
|
+
raise ValueError(f"capability {name!r} has a malformed impl_ref {ref!r} (expected 'module:attr')")
|
|
31
|
+
return getattr(import_module(module_path), attr)
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
"""ADR-0119: a generic Jev typed-decision capability over the OpenRouter Decisions API.
|
|
2
|
+
|
|
3
|
+
Jev (TypeSafe's "System-1" decision model) returns CALIBRATED typed answers -- a yes/no (`noul`), a `choice` from
|
|
4
|
+
a set, or a `score` -- for a `state` + typed `questions`, with no generated text, in ~70-500 ms. This capability
|
|
5
|
+
is DOMAIN-FREE mechanism: it just forwards a Decisions-API request and returns the typed answers. The compliance
|
|
6
|
+
reference domain uses it for the operative-rule gate + claim_types/actor (ADR-0119, where it reaches the LLM's
|
|
7
|
+
accuracy zero/few-shot, calibrated, at ~$0.00002/call); any domain can use it for routing / tagging / screening.
|
|
8
|
+
|
|
9
|
+
It is I/O-bound (a network call), so it is an ASYNC `model` capability -- invoke via `api.ainvoke_model` (the sync
|
|
10
|
+
`api.invoke_model` refuses an async impl). Laya (`laya` skill) is the open-weight / on-prem fallback.
|
|
11
|
+
"""
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import os
|
|
15
|
+
from typing import Any
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
async def jev_decision(resources: Any, inputs: dict) -> dict: # noqa: ARG001 - API model, store-independent
|
|
19
|
+
"""Invoke a typed-decision model (Jev). `inputs`: `state` (the text/object to judge), `questions` (dict keyed
|
|
20
|
+
by question id, each `{type: "noul"|"choice"|"score", instructions, criteria}`), optional `model` (a
|
|
21
|
+
DecisionModelProfile key, default `jev-1.13`). The endpoint, served id, key env and timeout come from the
|
|
22
|
+
decision-model PROFILE (`models.profiles.decision_profile`, ADR-0119) -- not hardcoded here -- so swapping
|
|
23
|
+
Jev versions or pointing at an on-prem Laya decisions server is config. Returns the Decisions-API body
|
|
24
|
+
`{answers: {<id>: {type, noul|choice|score, ...}}, usage: {...}}`. Raises if the key env is unset or the API errors."""
|
|
25
|
+
import httpx
|
|
26
|
+
|
|
27
|
+
from rag_wright.models.profiles import decision_profile
|
|
28
|
+
|
|
29
|
+
prof = decision_profile(inputs.get("model"))
|
|
30
|
+
api_key = os.environ.get(prof.api_key_env)
|
|
31
|
+
if not api_key:
|
|
32
|
+
raise RuntimeError(f"jev_decision requires {prof.api_key_env}")
|
|
33
|
+
body = {"model": prof.served, "state": inputs["state"], "questions": inputs["questions"]}
|
|
34
|
+
timeout = float(os.environ.get("RAG_JEV_TIMEOUT_S", str(prof.timeout_s)))
|
|
35
|
+
async with httpx.AsyncClient() as client:
|
|
36
|
+
r = await client.post(prof.endpoint, headers={"Authorization": f"Bearer {api_key}"}, json=body, timeout=timeout)
|
|
37
|
+
r.raise_for_status()
|
|
38
|
+
return r.json()
|