rag-wright 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (184) hide show
  1. rag_wright/__init__.py +13 -0
  2. rag_wright/api/__init__.py +33 -0
  3. rag_wright/api/config.py +59 -0
  4. rag_wright/api/discover.py +70 -0
  5. rag_wright/api/documents.py +39 -0
  6. rag_wright/api/ids.py +31 -0
  7. rag_wright/api/invoke.py +99 -0
  8. rag_wright/api/kg.py +61 -0
  9. rag_wright/api/mcp.py +94 -0
  10. rag_wright/api/usage.py +30 -0
  11. rag_wright/api/workspace.py +85 -0
  12. rag_wright/capabilities/__init__.py +8 -0
  13. rag_wright/capabilities/answer_generator.py +427 -0
  14. rag_wright/capabilities/ard.py +286 -0
  15. rag_wright/capabilities/assertion_extraction.py +79 -0
  16. rag_wright/capabilities/chunk_read.py +58 -0
  17. rag_wright/capabilities/chunk_write.py +163 -0
  18. rag_wright/capabilities/claim_extraction.py +153 -0
  19. rag_wright/capabilities/clause_exception_linking.py +117 -0
  20. rag_wright/capabilities/compliance_judgment.py +322 -0
  21. rag_wright/capabilities/compliance_store.py +87 -0
  22. rag_wright/capabilities/contract_kg_serve.py +156 -0
  23. rag_wright/capabilities/contract_kg_store.py +251 -0
  24. rag_wright/capabilities/dg_extraction.py +585 -0
  25. rag_wright/capabilities/disambiguation.py +163 -0
  26. rag_wright/capabilities/document_parse.py +87 -0
  27. rag_wright/capabilities/document_scope.py +49 -0
  28. rag_wright/capabilities/embedding.py +164 -0
  29. rag_wright/capabilities/embedding_profiles.py +43 -0
  30. rag_wright/capabilities/entity_resolution.py +154 -0
  31. rag_wright/capabilities/fusion.py +64 -0
  32. rag_wright/capabilities/graph_extraction.py +243 -0
  33. rag_wright/capabilities/graph_query.py +73 -0
  34. rag_wright/capabilities/graph_storage.py +111 -0
  35. rag_wright/capabilities/highlight_serve.py +142 -0
  36. rag_wright/capabilities/hybrid_search.py +65 -0
  37. rag_wright/capabilities/invoke.py +31 -0
  38. rag_wright/capabilities/jev_decision.py +38 -0
  39. rag_wright/capabilities/manifests.py +872 -0
  40. rag_wright/capabilities/okf_navigate.py +456 -0
  41. rag_wright/capabilities/parsing.py +286 -0
  42. rag_wright/capabilities/property_boosted_retrieval.py +125 -0
  43. rag_wright/capabilities/query_function_classifier.py +94 -0
  44. rag_wright/capabilities/query_understanding.py +109 -0
  45. rag_wright/capabilities/registry.py +262 -0
  46. rag_wright/capabilities/remote_encoders.py +94 -0
  47. rag_wright/capabilities/requirement_extraction.py +247 -0
  48. rag_wright/capabilities/reranking.py +123 -0
  49. rag_wright/capabilities/retrieval_core.py +126 -0
  50. rag_wright/capabilities/rlm_chunking.py +808 -0
  51. rag_wright/capabilities/rlm_synthesis.py +316 -0
  52. rag_wright/capabilities/scan_quality.py +136 -0
  53. rag_wright/capabilities/span_relevance_judgment.py +191 -0
  54. rag_wright/capabilities/vision_to_text.py +85 -0
  55. rag_wright/capabilities/vlm_ocr.py +85 -0
  56. rag_wright/contracts/__init__.py +6 -0
  57. rag_wright/contracts/chunk.py +79 -0
  58. rag_wright/contracts/compliance.py +303 -0
  59. rag_wright/contracts/contract_meta.py +27 -0
  60. rag_wright/contracts/extraction.py +130 -0
  61. rag_wright/contracts/function.py +167 -0
  62. rag_wright/contracts/function_routing.py +91 -0
  63. rag_wright/contracts/highlight.py +74 -0
  64. rag_wright/contracts/identifiers.py +153 -0
  65. rag_wright/contracts/jurisdiction.py +96 -0
  66. rag_wright/contracts/ontology.py +142 -0
  67. rag_wright/contracts/property.py +201 -0
  68. rag_wright/contracts/provenance.py +78 -0
  69. rag_wright/contracts/query_intent.py +53 -0
  70. rag_wright/contracts/span.py +76 -0
  71. rag_wright/contracts/value_match.py +84 -0
  72. rag_wright/corpus/__init__.py +0 -0
  73. rag_wright/corpus/canonicalize.py +116 -0
  74. rag_wright/corpus/cuad.py +153 -0
  75. rag_wright/corpus/cuad_ingestion.py +72 -0
  76. rag_wright/corpus/document_parser.py +299 -0
  77. rag_wright/corpus/edgar.py +231 -0
  78. rag_wright/corpus/gcs_ingestion.py +120 -0
  79. rag_wright/corpus/http.py +110 -0
  80. rag_wright/corpus/selection.py +152 -0
  81. rag_wright/mcp/__init__.py +11 -0
  82. rag_wright/mcp/compliance_server.py +299 -0
  83. rag_wright/mcp/intra_document_qa_server.py +170 -0
  84. rag_wright/mcp/relational_qa_server.py +171 -0
  85. rag_wright/mcp/session_store.py +64 -0
  86. rag_wright/mcp/typed_property_retrieval_server.py +191 -0
  87. rag_wright/models/__init__.py +8 -0
  88. rag_wright/models/profiles.py +331 -0
  89. rag_wright/models/seam.py +497 -0
  90. rag_wright/models/tag_structured.py +285 -0
  91. rag_wright/models/tracing.py +179 -0
  92. rag_wright/models/usage.py +102 -0
  93. rag_wright/okf/__init__.py +11 -0
  94. rag_wright/okf/compile.py +292 -0
  95. rag_wright/okf/document.py +47 -0
  96. rag_wright/okf/enrich.py +176 -0
  97. rag_wright/okf/links.py +190 -0
  98. rag_wright/okf/lint.py +105 -0
  99. rag_wright/ontology/__init__.py +6 -0
  100. rag_wright/ontology/_generated_template_meta.py +60 -0
  101. rag_wright/ontology/_generated_vocab.py +52 -0
  102. rag_wright/ontology/clause_template.py +964 -0
  103. rag_wright/ontology/codegen.py +84 -0
  104. rag_wright/ontology/compliance_bridge.ttl +186 -0
  105. rag_wright/ontology/contract_bridge.ttl +2685 -0
  106. rag_wright/ontology/contract_taxonomy.py +24 -0
  107. rag_wright/ontology/derive.py +58 -0
  108. rag_wright/ontology/loader.py +435 -0
  109. rag_wright/ontology/packs/ftc_16cfr255.ttl +29 -0
  110. rag_wright/ontology/registry.py +87 -0
  111. rag_wright/ontology/template_introspect.py +100 -0
  112. rag_wright/py.typed +0 -0
  113. rag_wright/reference/__init__.py +2 -0
  114. rag_wright/reference/compliance.py +41 -0
  115. rag_wright/reference/contract_seam.py +123 -0
  116. rag_wright/skills/__init__.py +7 -0
  117. rag_wright/skills/claim_extraction/SKILL.md +47 -0
  118. rag_wright/skills/claim_extraction/__init__.py +1 -0
  119. rag_wright/skills/claim_extraction/template.py +50 -0
  120. rag_wright/skills/compliance_judgment/SKILL.md +59 -0
  121. rag_wright/skills/corpus_ingest/SKILL.md +106 -0
  122. rag_wright/skills/extraction_semantic_judge/SKILL.md +51 -0
  123. rag_wright/skills/extraction_semantic_judge/__init__.py +1 -0
  124. rag_wright/skills/generation/SKILL.md +64 -0
  125. rag_wright/skills/generation/__init__.py +1 -0
  126. rag_wright/skills/generic_compliance_judgment/SKILL.md +58 -0
  127. rag_wright/skills/okf_navigate/SKILL.md +137 -0
  128. rag_wright/skills/requirement_extraction/SKILL.md +47 -0
  129. rag_wright/skills/requirement_extraction/__init__.py +1 -0
  130. rag_wright/skills/requirement_extraction/template.py +50 -0
  131. rag_wright/skills/rlm/SKILL.md +186 -0
  132. rag_wright/skills/rlm/__init__.py +31 -0
  133. rag_wright/skills/rlm/agent.py +292 -0
  134. rag_wright/skills/span_relevance_judgment/SKILL.md +67 -0
  135. rag_wright/skills/vision_to_text/SKILL.md +36 -0
  136. rag_wright/skills/vision_to_text/__init__.py +1 -0
  137. rag_wright/spans/__init__.py +1 -0
  138. rag_wright/spans/boundary.py +78 -0
  139. rag_wright/spans/clause_function_classifier.py +490 -0
  140. rag_wright/spans/clause_kg_extractor.py +337 -0
  141. rag_wright/spans/cuad_labels.py +81 -0
  142. rag_wright/spans/dim_classifier.py +158 -0
  143. rag_wright/spans/dim_fleet.json +411 -0
  144. rag_wright/spans/function_classifier.py +77 -0
  145. rag_wright/spans/function_families.py +62 -0
  146. rag_wright/spans/hybrid_classifier.py +103 -0
  147. rag_wright/spans/legalbert_classifier.py +83 -0
  148. rag_wright/spans/model_capabilities.py +107 -0
  149. rag_wright/spans/new_function_labels.py +111 -0
  150. rag_wright/spans/page_map.py +68 -0
  151. rag_wright/spans/property_extractor.py +365 -0
  152. rag_wright/spans/property_grounding.py +182 -0
  153. rag_wright/spans/reclassify.py +77 -0
  154. rag_wright/spans/scarce_function_labels.py +105 -0
  155. rag_wright/spans/segment.py +341 -0
  156. rag_wright/spans/semantic_judge.py +197 -0
  157. rag_wright/spans/symbolic_validation.py +131 -0
  158. rag_wright/spans/tag_clause_extractor.py +182 -0
  159. rag_wright/store/__init__.py +6 -0
  160. rag_wright/store/arcadedb.py +1135 -0
  161. rag_wright/store/chunk_text.py +66 -0
  162. rag_wright/store/seam.py +213 -0
  163. rag_wright/subgraphs/__init__.py +0 -0
  164. rag_wright/subgraphs/async_ingestion.py +204 -0
  165. rag_wright/subgraphs/compliance_check.py +1042 -0
  166. rag_wright/subgraphs/compliance_ingestion.py +306 -0
  167. rag_wright/subgraphs/contract_ingestion_pipeline.py +999 -0
  168. rag_wright/subgraphs/graph_extraction.py +102 -0
  169. rag_wright/subgraphs/intra_document_qa.py +328 -0
  170. rag_wright/subgraphs/observability.py +140 -0
  171. rag_wright/subgraphs/query_constraint_extraction.py +73 -0
  172. rag_wright/subgraphs/relational_qa.py +165 -0
  173. rag_wright/subgraphs/requirement_extraction.py +137 -0
  174. rag_wright/subgraphs/scaffold.py +65 -0
  175. rag_wright/subgraphs/semantic_chunking.py +183 -0
  176. rag_wright/subgraphs/typed_clause_extraction.py +172 -0
  177. rag_wright/subgraphs/typed_property_retrieval.py +278 -0
  178. rag_wright/util/__init__.py +1 -0
  179. rag_wright/util/concurrent.py +153 -0
  180. rag_wright/util/spacy_model.py +45 -0
  181. rag_wright-0.1.0.dist-info/METADATA +168 -0
  182. rag_wright-0.1.0.dist-info/RECORD +184 -0
  183. rag_wright-0.1.0.dist-info/WHEEL +4 -0
  184. rag_wright-0.1.0.dist-info/licenses/LICENSE +21 -0
rag_wright/__init__.py ADDED
@@ -0,0 +1,13 @@
1
+ """RAG_Wright: the open-core RAG ENGINE (RAG_Capability_Spec.md v0.1; engine/product split, ADR-0052).
2
+
3
+ This package builds and registers the FR-C capabilities (parsing, chunking, embedding, hybrid search,
4
+ reranking, graph extraction, entity resolution, ontology and registry derivation, reasoning and generation,
5
+ and the RLM skill) as ordinary tested software, AND the ingestion and query pipelines that compose them --
6
+ ordinary hardened LangGraph subgraphs, not a compiler output (GraphWright is PARKED, ADR-0052). Each capability
7
+ is registered under its FR-C name for Agentic Resource Discovery (ARD, `urn:air`).
8
+
9
+ The engine is ASYNC end to end (ADR-0057): every model call runs under a true wall-clock deadline. The public
10
+ ingest / query / compliance entrypoints are `async` (or return LangGraph graphs called via `.ainvoke`); see
11
+ `docs/product/engine_async_api.md` for the interface contract the product (RuleWright) depends on. Dependency
12
+ direction is strict and one-way: Product -> Engine, never Engine -> Product.
13
+ """
@@ -0,0 +1,33 @@
1
+ """The engine API layer (ADR-0117): the stable, domain-agnostic surface a product builds on.
2
+
3
+ EP-API-1 ships the typed config + the opaque workspace handle. Future increments add the per-kind invokers
4
+ (EP-API-2), generic `kg_read`/`kg_write` + id/format accessors (EP-API-3), and the options catalog + pluggable
5
+ embedders (EP-API-4). The product imports from here; it never reaches the store/embedder/model implementations."""
6
+ from __future__ import annotations
7
+
8
+ from rag_wright.api.config import EngineConfig, EngineOptions, IngestOptions, StoreConfig
9
+ # The capability-registration surface lives in capabilities.manifests; re-export it here (PREP-1.5) so the public
10
+ # story is uniformly "everything is rag_wright.api". The original import path keeps working — these are the same
11
+ # objects, not a fork.
12
+ from rag_wright.capabilities.manifests import (
13
+ load_reference_pack,
14
+ reference_pack,
15
+ register_capability,
16
+ )
17
+ from rag_wright.api.documents import aparse_document, parse_document, source_document
18
+ from rag_wright.api.ids import decode_bbox, document_of, id_source
19
+ from rag_wright.api.discover import Discovered, discover
20
+ from rag_wright.api.invoke import ainvoke_model, ainvoke_subgraph, capability_index, invoke_model
21
+ from rag_wright.api.kg import entities_by_name, kg_edges, kg_read, kg_write, span_positions
22
+ from rag_wright.api.usage import ModelUsage, UsageTotals, measure_usage
23
+ from rag_wright.api.workspace import WorkspaceHandle, open_workspace
24
+
25
+ __all__ = [
26
+ "EngineConfig", "StoreConfig", "EngineOptions", "IngestOptions", "WorkspaceHandle", "open_workspace",
27
+ "ainvoke_subgraph", "invoke_model", "ainvoke_model", "capability_index", "discover", "Discovered",
28
+ "kg_read", "kg_write", "kg_edges", "entities_by_name", "span_positions",
29
+ "document_of", "id_source", "decode_bbox",
30
+ "source_document", "parse_document", "aparse_document",
31
+ "measure_usage", "UsageTotals", "ModelUsage",
32
+ "register_capability", "load_reference_pack", "reference_pack",
33
+ ]
@@ -0,0 +1,59 @@
1
+ """EP-API-1/4 (ADR-0117): the engine's typed configuration. The product constructs an `EngineConfig` and passes it
2
+ to `open_workspace`; the engine resolves backends behind an opaque handle. This is the ONLY place a store backend is
3
+ named -- everything else goes through the handle. EP-API-4a adds the `options` catalog (ingest knobs today;
4
+ retrieval/rerank groups land as needed), so a product tunes the engine through config, never environment variables."""
5
+ from __future__ import annotations
6
+
7
+ from dataclasses import dataclass, field
8
+ from typing import Optional
9
+
10
+
11
+ @dataclass(frozen=True)
12
+ class StoreConfig:
13
+ """How to reach the KG/retrieval store. `backend` selects the implementation (only `arcadedb` today; a new
14
+ backend -- e.g. Neo4j -- is added here, invisibly to the product)."""
15
+
16
+ host: str
17
+ port: str
18
+ user: str
19
+ password: str
20
+ backend: str = "arcadedb"
21
+ protocol: str = "http"
22
+
23
+
24
+ @dataclass(frozen=True)
25
+ class IngestOptions:
26
+ """Ingest-time knobs, settable through config instead of environment variables (EP-API-4a). Every field defaults
27
+ to `None` = "use the engine default", so the engine's existing env fallback is preserved (a non-API caller is
28
+ unaffected) and an API caller that leaves these unset gets today's behavior exactly. Set a field to override."""
29
+
30
+ classify_concurrency: Optional[int] = None # function-classify parallelism (was CLASSIFY_CONCURRENCY)
31
+ clause_concurrency: Optional[int] = None # clause-extraction parallelism (was CLAUSE_CONCURRENCY)
32
+ affiliations: Optional[bool] = None # run affiliation extraction (was RAG_INGEST_AFFILIATIONS)
33
+ function_classifier: Optional[str] = None # "setfit" | "llm" (was RAG_FUNCTION_CLASSIFIER)
34
+ list_model: Optional[str] = None # secondary list-union model, "off" to disable (was RAG_INGEST_LIST_MODEL)
35
+ clause_samples: Optional[int] = None # multi-sample count for the list union (was RAG_INGEST_CLAUSE_SAMPLES)
36
+
37
+
38
+ @dataclass(frozen=True)
39
+ class EngineOptions:
40
+ """The engine's options catalog. Ingest knobs today; retrieval / reranking / chunking groups are added here as
41
+ they are promoted off environment variables."""
42
+
43
+ ingest: IngestOptions = field(default_factory=IngestOptions)
44
+
45
+
46
+ @dataclass(frozen=True)
47
+ class EngineConfig:
48
+ """The product's view of the engine: the store connection, chosen models (by role alias), the embedding profile,
49
+ and the `options` catalog. Defaults just work; override only to trade quality/cost/latency. Implementation
50
+ details (ArcadeDB, BGE) never cross this boundary."""
51
+
52
+ store: StoreConfig
53
+ models: dict[str, str] = field(default_factory=dict) # ModelRole value -> engine-supported model alias (override)
54
+ embeddings: dict[str, str] = field(default_factory=lambda: {"text": "bge-m3"}) # profile -> supported embedder
55
+ options: EngineOptions = field(default_factory=EngineOptions) # EP-API-4a: ingest (+ future) knobs via config
56
+ # AC-journey: the DOMAIN pack `.ttl` declaring this domain's KG vertex/edge types (open_workspace creates them).
57
+ # None = the engine's reference CONTRACT pack. A new domain points this at its own `.ttl` -- "config + .ttl",
58
+ # no engine edit -- and `ensure_schema` creates that domain's schema on top of the always-on engine types.
59
+ pack: Optional[str] = None
@@ -0,0 +1,70 @@
1
+ """Capability discovery: rank the live ARD catalog by semantic match to a task (embedding-based).
2
+
3
+ A product-side agent that needs to PLAN over the engine — when invoking a known capability by name through its seam
4
+ is not enough and it must find what's available for a task — calls `discover(task, resources=ws)`, then orchestrates
5
+ the `ainvoke_*` calls over the top matches. Ranking is by BGE-M3 embedding similarity of the task against each
6
+ capability's `representative_queries` + description (the signal manifests carry for exactly this), using the
7
+ workspace's query embedder — the same embedder and vector space retrieval uses. Domain-free: it ranks whatever is in
8
+ the LIVE catalog (`MANIFEST_SPECS`), so it covers the product's OWN registered capabilities, not just the reference
9
+ pack. Complements `capability_index` (the flat listing) with task-ranked selection.
10
+ """
11
+ from __future__ import annotations
12
+
13
+ import math
14
+ from dataclasses import dataclass
15
+ from typing import Optional
16
+
17
+ from rag_wright.api.workspace import WorkspaceHandle
18
+
19
+
20
+ @dataclass(frozen=True)
21
+ class Discovered:
22
+ """One ranked capability match from `discover` — enough for an agent to pick and invoke it by `slug`."""
23
+
24
+ slug: str
25
+ kind: str
26
+ description: str
27
+ representative_queries: tuple[str, ...]
28
+ score: float # cosine similarity in [-1, 1]; higher is a better task match
29
+
30
+
31
+ def _cosine(a: list[float], b: list[float]) -> float:
32
+ dot = sum(x * y for x, y in zip(a, b))
33
+ na = math.sqrt(sum(x * x for x in a))
34
+ nb = math.sqrt(sum(y * y for y in b))
35
+ return 0.0 if na == 0.0 or nb == 0.0 else dot / (na * nb)
36
+
37
+
38
+ def _match_text(spec) -> str:
39
+ rq = " ".join(spec.representative_queries or ())
40
+ return f"{spec.display_name}. {spec.description} {rq}".strip()
41
+
42
+
43
+ def discover(query: str, *, resources: WorkspaceHandle, kind: Optional[str] = None, k: int = 8) -> list[Discovered]:
44
+ """Rank the live ARD catalog by semantic match to `query`; return the top `k` (optionally filtered to one
45
+ `kind`: `subgraph` / `model` / `function` / `agent_skill` / `mcp_tool`). Embedding-based, via the workspace's
46
+ query embedder (BGE-M3, the same space retrieval uses). Returns `[]` when the (filtered) catalog is empty; raises
47
+ `RuntimeError` if the workspace has no query embedder available (discovery needs one)."""
48
+ from rag_wright.capabilities.manifests import MANIFEST_SPECS
49
+
50
+ specs = [s for s in MANIFEST_SPECS.values() if kind is None or s.kind == kind]
51
+ if not specs:
52
+ return []
53
+ embedder = getattr(resources, "_embedder", None)
54
+ if embedder is None:
55
+ raise RuntimeError("discover() needs a query embedder; none is available on this workspace")
56
+ dense, _sparse = embedder.encode_batch([query] + [_match_text(s) for s in specs])
57
+ query_vec, cand_vecs = dense[0], dense[1:]
58
+ ranked = sorted(
59
+ zip(specs, cand_vecs), key=lambda sv: _cosine(query_vec, sv[1]), reverse=True
60
+ )
61
+ return [
62
+ Discovered(
63
+ slug=s.slug,
64
+ kind=s.kind,
65
+ description=s.description,
66
+ representative_queries=tuple(s.representative_queries or ()),
67
+ score=round(_cosine(query_vec, v), 4),
68
+ )
69
+ for s, v in ranked[:k]
70
+ ]
@@ -0,0 +1,39 @@
1
+ """EP-API-2b / EP-API-6 (ADR-0117): building a document to feed the ingestion capability.
2
+
3
+ `source_document(document_id, text=...)` makes an engine `SourceDocument` from plain text (the simplest path, no
4
+ parser needed). `parse_document` / `aparse_document` make a STRUCTURE-BEARING `SourceDocument` from a file PATH via
5
+ the engine's real docling parse (`.parsed` set), so the full ingestion pipeline incl docling runs through the API --
6
+ not only the text fallback. Heavy deps imported lazily so `import rag_wright.api` stays light."""
7
+ from __future__ import annotations
8
+
9
+ from pathlib import Path
10
+ from typing import Any
11
+
12
+
13
+ def source_document(document_id: str, *, text: str) -> Any:
14
+ """A text-only `SourceDocument` (`source_doc_id`, `text`) to pass as the `document` input of the
15
+ `contract_ingestion_pipeline` capability. The id should be a canonical, delimiter-safe source-doc id."""
16
+ from rag_wright.capabilities.document_parse import SourceDocument
17
+
18
+ return SourceDocument(source_doc_id=document_id, text=text)
19
+
20
+
21
+ def parse_document(document_id: str, path: Any, *, cache_dir: Any, metadata: dict | None = None) -> Any:
22
+ """Docling-parse the file at `path` ONCE (content-hash gated + cached under `cache_dir`) into a
23
+ structure-bearing `SourceDocument` -- `.parsed` carries the `DoclingDocument` so the chunker's structural pass
24
+ fires on real headings, and `.text` holds the flattened text. This is the PDF/DOCX/HTML/MD ingest entry point
25
+ of the engine API; pass the result as the `document` input of `contract_ingestion_pipeline`. The docling parse
26
+ blocks; use `aparse_document` on an event loop."""
27
+ from rag_wright.capabilities.document_parse import parsed_source_document
28
+
29
+ p = Path(path)
30
+ return parsed_source_document(document_id, p.name, p.read_bytes(), cache_dir=cache_dir, metadata=metadata)
31
+
32
+
33
+ async def aparse_document(document_id: str, path: Any, *, cache_dir: Any, metadata: dict | None = None) -> Any:
34
+ """The async, deadline-bounded twin of `parse_document` (ADR-0057): runs the docling parse off the event loop
35
+ so a hand-built async ingest can parse a document into a structure-bearing `SourceDocument` without blocking."""
36
+ from rag_wright.capabilities.document_parse import aparsed_source_document
37
+
38
+ p = Path(path)
39
+ return await aparsed_source_document(document_id, p.name, p.read_bytes(), cache_dir=cache_dir, metadata=metadata)
rag_wright/api/ids.py ADDED
@@ -0,0 +1,31 @@
1
+ """EP-API-3 (ADR-0117): engine id/format accessors.
2
+
3
+ So a product never parses engine id strings or storage formats itself (today the seam reimplements these, drifting
4
+ from the engine). Heavy deps are imported lazily so `import rag_wright.api` stays light (progressive loading)."""
5
+ from __future__ import annotations
6
+
7
+ from typing import Optional
8
+
9
+
10
+ def document_of(entity_id: str) -> str:
11
+ """The source-document id embedded in a span/chunk/clause id (`<source_doc_id>:<idx>:<hash>` -> the first,
12
+ delimiter-safe segment). Empty in -> empty out."""
13
+ from rag_wright.store.arcadedb import _doc_id_of
14
+
15
+ return _doc_id_of(entity_id)
16
+
17
+
18
+ def id_source(requirement_id: str) -> str:
19
+ """The source/policy of a Requirement id (`<source>:<section>:<hash>` -> the first segment; `source` is
20
+ delimiter-safe via `canonical_source_doc_id`, so this is the same first-segment rule as `document_of`)."""
21
+ from rag_wright.store.arcadedb import _doc_id_of
22
+
23
+ return _doc_id_of(requirement_id)
24
+
25
+
26
+ def decode_bbox(raw: Optional[str]) -> Optional[tuple]:
27
+ """Decode the engine's best-effort bounding box (stored as a JSON `[l,t,r,b]` string) to a `(l, t, r, b)` tuple,
28
+ or None. The single canonical decoder (retires the product seam's copy)."""
29
+ from rag_wright.capabilities.highlight_serve import _decode_bbox
30
+
31
+ return _decode_bbox(raw)
@@ -0,0 +1,99 @@
1
+ """EP-API-2 / EP-CORE-2 (ADR-0117, ADR-0118): the engine's capability invoker -- an adapter-free ARD *client*.
2
+
3
+ Progressive loading, like an agent holding skill name+metadata and loading the full thing on demand:
4
+ * a LIGHT index (`slug -> manifest spec`) is built ONCE from the ARD manifest specs (no capability IMPLEMENTATION
5
+ is imported) -- the "cards" layer, for discovery + knowing what exists;
6
+ * on invoke, the capability's `impl_ref` (a "module:attr" vendor-extension pointer on its manifest) is imported
7
+ LAZILY via `capability_impl` and called -- so there is NO central engine-owned adapter dict. A developer who
8
+ registers a capability with an `impl_ref` makes it invocable with zero engine edits (ADR-0118).
9
+ The ARD registry stays metadata (ADR-0003: a string pointer, never a callable). Each per-kind invoker takes an
10
+ opaque `WorkspaceHandle` as resources and wraps the call in a trace span; model usage/cost is captured by the
11
+ CALLER's `api.measure_usage()` (EP-API-5). Subgraph hardening (retry/dead-letter) comes from the LangGraph scaffold."""
12
+ from __future__ import annotations
13
+
14
+ import asyncio
15
+ from typing import Any
16
+
17
+ from rag_wright.capabilities.invoke import capability_impl
18
+ from rag_wright.api.workspace import WorkspaceHandle
19
+ from rag_wright.models.tracing import traced_step
20
+
21
+ def _index() -> dict[str, Any]:
22
+ """The light capability index (`slug -> manifest spec`): the LIVE runtime ARD catalog the developer populates
23
+ (EP-CORE-3). Read fresh each call so a just-registered capability is seen; carries only metadata (kind/
24
+ description/impl_ref/representative_queries), never an implementation."""
25
+ from rag_wright.capabilities.manifests import MANIFEST_SPECS
26
+
27
+ return MANIFEST_SPECS
28
+
29
+
30
+ def capability_index() -> dict[str, dict]:
31
+ """Public discovery index: `{slug: {kind, description}}` for every catalogued capability (the 'cards')."""
32
+ return {slug: {"kind": s.kind, "description": s.description} for slug, s in _index().items()}
33
+
34
+
35
+ def _validate(name: str, kind: str) -> None:
36
+ """Validate a capability name against the ARD catalog (name known, kind matches). Raises KeyError/ValueError."""
37
+ idx = _index()
38
+ if name not in idx:
39
+ raise KeyError(f"unknown capability {name!r} (not in the ARD catalog)")
40
+ if idx[name].kind != kind:
41
+ raise ValueError(f"capability {name!r} is kind {idx[name].kind!r}, not {kind!r}")
42
+
43
+
44
+ async def ainvoke_subgraph(name: str, inputs: dict, *, resources: WorkspaceHandle) -> Any:
45
+ """Invoke a subgraph-kind capability by name over the workspace, inside a trace span. The implementation is
46
+ resolved lazily from the manifest `impl_ref` (no central adapter dict). Retry/dead-letter comes from the
47
+ LangGraph scaffold the subgraph is built on; usage is captured by the caller's `measure_usage()` (EP-API-5)."""
48
+ _validate(name, "subgraph")
49
+ factory = capability_impl(name) # resolves impl_ref -> the co-located `ainvoke(resources, inputs)`
50
+ with traced_step(f"invoke:{name}"):
51
+ return await factory(resources, inputs)
52
+
53
+
54
+ def invoke_model(name: str, inputs: dict, *, resources: WorkspaceHandle) -> Any:
55
+ """Invoke a model-kind capability by name (SYNCHRONOUSLY). Validated against the ARD catalog, then resolved via
56
+ `impl_ref` and dispatched -- the SAME path the ingestion pipeline routes through (one production path, no second
57
+ hand-built fleet). `resources` is accepted for API uniformity but model capabilities are store-independent.
58
+ Usage is the caller's `measure_usage()` scope (EP-API-5). A model impl may be async (I/O-bound, e.g. an
59
+ LLM-backed cap) -- those cannot be invoked here; call `ainvoke_model` instead (we refuse rather than silently
60
+ return an un-awaited coroutine)."""
61
+ _validate(name, "model")
62
+ factory = capability_impl(name)
63
+ if asyncio.iscoroutinefunction(factory):
64
+ raise TypeError(
65
+ f"capability {name!r} has an async impl; call ainvoke_model() instead of invoke_model()")
66
+ with traced_step(f"invoke:{name}"):
67
+ return factory(resources, inputs)
68
+
69
+
70
+ async def ainvoke_model(name: str, inputs: dict, *, resources: WorkspaceHandle,
71
+ sem: asyncio.Semaphore | None = None) -> Any:
72
+ """Invoke a model-kind capability by name, ASYNCHRONOUSLY -- the async surface for model caps (the subgraph
73
+ legs already have `ainvoke_subgraph`). A model impl is one of two shapes, and this routes each honestly:
74
+ * SYNC (CPU-bound local inference -- a classifier/XGBoost fleet): run OFF the event loop in a worker thread
75
+ (`asyncio.to_thread`), so a big batch never blocks the loop;
76
+ * ASYNC (I/O-bound -- an LLM-backed cap calling OpenRouter or a local vLLM client): AWAITED directly, so the
77
+ I/O concurrency is real (not a thread wrapping a blocking call).
78
+ `sem` (an `asyncio.Semaphore`) bounds total in-flight work when a caller fans out a batch -- the same
79
+ backpressure the ingestion pipeline applies via `adispatch_model`. Usage is the caller's `measure_usage()`
80
+ scope (EP-API-5)."""
81
+ _validate(name, "model")
82
+ factory = capability_impl(name)
83
+
84
+ async def _run() -> Any:
85
+ with traced_step(f"invoke:{name}"):
86
+ if asyncio.iscoroutinefunction(factory):
87
+ return await factory(resources, inputs)
88
+ return await asyncio.to_thread(factory, resources, inputs)
89
+
90
+ if sem is None:
91
+ return await _run()
92
+ async with sem:
93
+ return await _run()
94
+
95
+
96
+ def _invocable_names() -> dict[str, str]:
97
+ """{name: kind} for every capability that declares an `impl_ref` -- the drift guard asserts each resolves to a
98
+ callable of the declared kind (replaces the old central-adapter-dict binding)."""
99
+ return {slug: s.kind for slug, s in _index().items() if getattr(s, "impl_ref", None)}
rag_wright/api/kg.py ADDED
@@ -0,0 +1,61 @@
1
+ """EP-API-3 (ADR-0117): scoped KG access over the opaque workspace handle.
2
+
3
+ Thin wrappers that delegate to the handle's resolved store, so a product does generic typed-KG reads/writes (and
4
+ span-position lookups for citations) WITHOUT importing `ArcadeDBStore` or touching `ws._store`. The generic
5
+ `kg_read`/`kg_write` primitives already live on the store (DD-1a/b); this exposes them on the API."""
6
+ from __future__ import annotations
7
+
8
+ from typing import Any, Optional
9
+
10
+ from rag_wright.api.workspace import WorkspaceHandle
11
+
12
+ # the span-position fields a citation needs (NOT the dense vector) -- offsets + page/bbox provenance
13
+ _SPAN_POSITION_FIELDS = [
14
+ "span_id", "parent_chunk_id", "span_index", "text", "function", "contract_id",
15
+ "doc_start", "doc_end", "pages", "bbox",
16
+ ]
17
+
18
+
19
+ def kg_read(ws: WorkspaceHandle, node_type: str, *, where: Optional[dict] = None, fields: Optional[list] = None,
20
+ distinct: Optional[str] = None, order_by: Optional[str] = None, limit: Optional[int] = None) -> list[dict]:
21
+ """Read typed nodes of `node_type` from the workspace (see `Store.kg_read`). Equality/`IN` filters, projection,
22
+ distinct, order, limit; an empty list `where` value is scope-to-nothing -> `[]`."""
23
+ return ws._store.kg_read(node_type, where=where, fields=fields, distinct=distinct, order_by=order_by, limit=limit)
24
+
25
+
26
+ def kg_write(ws: WorkspaceHandle, nodes: list, edges: Any = ()) -> None:
27
+ """Upsert typed `nodes` + create typed `edges` in one transaction (see `Store.kg_write`). `nodes`/`edges` are
28
+ `KgNode`/`KgEdge` (from `rag_wright.store.seam`); the store encodes each field per its pack-declared type."""
29
+ ws._store.kg_write(nodes, edges)
30
+
31
+
32
+ def kg_edges(ws: WorkspaceHandle, from_type: Optional[str] = None, *, where: Optional[dict] = None,
33
+ key_range: Optional[tuple] = None, direction: str = "out", edge_type: Optional[str] = None,
34
+ edge_where: Optional[dict] = None, target_where: Optional[dict] = None,
35
+ select: dict) -> list[dict]:
36
+ """Generic edge TRAVERSAL over the workspace (see `Store.kg_edges`): node-start out/in MATCH (by `where`
37
+ equality/membership or a contract-scope `key_range`) or a direct edge scan; `select` projects `c.`/`e.`/`v.`
38
+ expressions. The engine's relational/graph primitive on the API, so a domain's graph query never touches
39
+ `ws._store`. (`NOT_NULL` for a presence filter is `rag_wright.store.seam.NOT_NULL`.)"""
40
+ return ws._store.kg_edges(from_type, where=where, key_range=key_range, direction=direction,
41
+ edge_type=edge_type, edge_where=edge_where, target_where=target_where, select=select)
42
+
43
+
44
+ def entities_by_name(ws: WorkspaceHandle, name: str) -> list[dict]:
45
+ """Resolve an entity NAME to every entity node it matches: `[{entity_id, name, entity_type}]` (the engine owns
46
+ the surface-form normalization, so variants collapse to one id). One name can match several nodes (a resolved
47
+ node + an unlinked ref sharing a clustering key) -- all are returned. `entity_id` is exactly the
48
+ `start_entity_id` a graph traversal takes. The engine's generic entity-lookup primitive on the API."""
49
+ return ws._store.entities_by_name(name)
50
+
51
+
52
+ def span_positions(ws: WorkspaceHandle, document: str) -> list[dict]:
53
+ """Every span of `document` with its position provenance (doc offsets, pages, DECODED bbox), ordered by document
54
+ position. The engine MECHANISM behind a product's citation/highlight types -- the product wraps these rows into
55
+ its own presentation type (e.g. `SpanLocation`)."""
56
+ from rag_wright.api.ids import decode_bbox
57
+ from rag_wright.store.arcadedb import SPAN_TYPE
58
+
59
+ rows = ws._store.kg_read(SPAN_TYPE, fields=_SPAN_POSITION_FIELDS, where={"contract_id": document},
60
+ order_by="doc_start")
61
+ return [{**r, "bbox": decode_bbox(r.get("bbox"))} for r in rows]
rag_wright/api/mcp.py ADDED
@@ -0,0 +1,94 @@
1
+ """EP-RT-2 (ADR-0117): the generic capability->MCP adapter.
2
+
3
+ One function, `build_capability_mcp(slug, *, resources)`, exposes ANY catalogued invokable capability as a FastMCP
4
+ server with NO bespoke per-capability server code -- the tool's name/title/description are read from the ARD
5
+ manifest, and its handler dispatches through the engine invoker (`ainvoke_subgraph`/`invoke_model`) over the bound
6
+ workspace. This is the zero-boilerplate path: a new-domain product (or a new engine capability) gets a discoverable
7
+ MCP tool for free, the moment the capability is in the ARD catalog.
8
+
9
+ The four hand-written servers in `rag_wright/mcp/` stay: they offer a CURATED, typed tool signature + description for
10
+ the Tier-1 legs. This generic adapter is the complement -- it takes a single opaque `inputs` dict (the capability's
11
+ own input contract, dispatched straight to the invoker) rather than a per-capability typed signature, because the
12
+ ARD manifest carries no JSON input schema to derive one from.
13
+
14
+ Only `subgraph` and `model` kinds are supported (the kinds the invoker can dispatch). An `mcp_tool` is already an
15
+ MCP tool; `function`/`agent_skill` have no invoker adapter yet (EP-API-2c) -- all rejected with a clear error.
16
+
17
+ Store binding is server-side via the `WorkspaceHandle` (issue 0035): the calling agent never supplies a tenant or
18
+ store; the workspace is bound when the server is built. Multi-tenant request routing is a product concern (a product
19
+ opens one workspace per corpus; `corpus` = the backend db name)."""
20
+ from __future__ import annotations
21
+
22
+ import asyncio
23
+ from typing import Any
24
+
25
+ from fastmcp import FastMCP
26
+
27
+ from rag_wright.api.invoke import _index, ainvoke_subgraph, invoke_model
28
+ from rag_wright.api.workspace import WorkspaceHandle
29
+
30
+ _INVOKABLE_KINDS = ("subgraph", "model")
31
+
32
+
33
+ def _jsonable(obj: Any) -> Any:
34
+ """Normalize an invoker result (a LangGraph state dict, a Pydantic model, or a list of them) to JSON-safe
35
+ data -- the structured content the calling agent receives."""
36
+ if hasattr(obj, "model_dump"):
37
+ return obj.model_dump(mode="json")
38
+ if isinstance(obj, dict):
39
+ return {k: _jsonable(v) for k, v in obj.items()}
40
+ if isinstance(obj, (list, tuple)):
41
+ return [_jsonable(v) for v in obj]
42
+ return obj
43
+
44
+
45
+ def build_capability_mcp(slug: str, *, resources: WorkspaceHandle, name: str | None = None) -> FastMCP:
46
+ """Build a FastMCP server exposing the catalogued capability `slug` as one MCP tool, driven by its ARD manifest.
47
+
48
+ The tool is named for the capability (its canonical slug -- stable + discoverable), titled with the manifest
49
+ display name, and described by the manifest description. It takes one `inputs` dict (the capability's input
50
+ contract) and dispatches through the engine invoker over `resources` (the bound workspace), returning the result
51
+ as JSON. Only `subgraph`/`model` kinds are supported.
52
+
53
+ Raises `KeyError` for an unknown slug; `ValueError` for a non-invokable kind."""
54
+ idx = _index()
55
+ if slug not in idx:
56
+ raise KeyError(f"unknown capability {slug!r} (not in the ARD catalog)")
57
+ spec = idx[slug]
58
+ if spec.kind not in _INVOKABLE_KINDS:
59
+ raise ValueError(
60
+ f"capability {slug!r} is kind {spec.kind!r}; only {_INVOKABLE_KINDS} are exposable via the generic "
61
+ f"MCP adapter (an mcp_tool is already an MCP tool; function/agent-skill have no invoker adapter yet)")
62
+
63
+ mcp: FastMCP = FastMCP(
64
+ name=name or f"rag-wright-{slug.replace('_', '-')}",
65
+ instructions=(
66
+ f"{spec.description}\n\nCall `{slug}` with an `inputs` object carrying the capability's inputs. "
67
+ f"Example queries: {'; '.join(spec.representative_queries)}."),
68
+ )
69
+ kind = spec.kind
70
+
71
+ @mcp.tool(name=slug, title=spec.display_name, description=spec.description)
72
+ async def _invoke_capability(inputs: dict) -> dict:
73
+ """Invoke the capability over the bound workspace.
74
+
75
+ Args:
76
+ inputs: The capability's input contract (e.g. {"query": ...} for a retrieval leg, or
77
+ {"text": ..., "functions": [...]} for a classifier). Dispatched straight to the engine invoker.
78
+
79
+ Returns:
80
+ {"result": <the capability's output as JSON>} -- a uniform envelope (a subgraph's state dict or a
81
+ model's soft-tag list both land under `result`), since the output contract varies by capability.
82
+ """
83
+ if kind == "subgraph":
84
+ out = await ainvoke_subgraph(slug, inputs, resources=resources)
85
+ else: # model -- sync invoker; offload so the event loop is not blocked by local inference
86
+ out = await asyncio.to_thread(invoke_model, slug, inputs, resources=resources)
87
+ return {"result": _jsonable(out)}
88
+
89
+ return mcp
90
+
91
+
92
+ def serve_capability_mcp(slug: str, *, resources: WorkspaceHandle, transport: str = "stdio") -> None:
93
+ """Serve one catalogued capability as an MCP tool over `transport` (default stdio, so an agent can spawn it)."""
94
+ build_capability_mcp(slug, resources=resources).run(transport=transport)
@@ -0,0 +1,30 @@
1
+ """EP-API-5 (ADR-0117, ADR-0105): usage/cost on the engine API surface.
2
+
3
+ A product measures the model usage an engine call incurs -- calls, input/output tokens, known cost (plus the count
4
+ of calls the backend surfaced NO cost for), total latency, and a per-model breakdown -- by wrapping the call in
5
+ `measure_usage()`. It is the public face of the engine's in-band usage accounting (`models/usage.usage_scope`), so
6
+ the product reads stats without reaching into engine internals. Ambient (a `ContextVar`), additive across nesting
7
+ (an outer scope totals everything; inner scopes attribute their slice), and carried into worker threads/executors.
8
+
9
+ with engine.measure_usage() as usage:
10
+ out = await engine.ainvoke_subgraph("intra_document_qa", {...}, resources=ws)
11
+ usage.calls, usage.input_tokens, usage.output_tokens, usage.cost_usd, usage.calls_without_cost,
12
+ usage.latency_ms_total, usage.by_model # {model_id: ModelUsage(...)}
13
+ """
14
+ from __future__ import annotations
15
+
16
+ from contextlib import contextmanager
17
+ from typing import Iterator
18
+
19
+ from rag_wright.models.usage import ModelUsage, UsageTotals, usage_scope
20
+
21
+ __all__ = ["measure_usage", "UsageTotals", "ModelUsage"]
22
+
23
+
24
+ @contextmanager
25
+ def measure_usage() -> Iterator[UsageTotals]:
26
+ """Accumulate the model usage of every engine call made inside the block; read the returned `UsageTotals`
27
+ after it. Nesting is additive, so a task-level scope totals everything while an inner per-call scope attributes
28
+ its slice. Capturing is opt-in: with no active scope, the engine records usage nowhere (zero overhead)."""
29
+ with usage_scope() as totals:
30
+ yield totals
@@ -0,0 +1,85 @@
1
+ """EP-API-1 (ADR-0117): `open_workspace` + the opaque `WorkspaceHandle`.
2
+
3
+ The product names a `corpus` (the logical KG = the backend database name; the product maps its own tenant -> db-name,
4
+ so tenancy stays product-owned) and gets back an opaque handle that resolves + caches the store (and lazily the
5
+ embedder) from the `EngineConfig`. The handle is what the engine's per-kind invokers (EP-API-2) take as resources.
6
+ The product never imports `ArcadeDBStore`/`query_embedder` and the handle exposes no public store accessor."""
7
+ from __future__ import annotations
8
+
9
+ import logging
10
+ from typing import Any, Optional
11
+
12
+ from rag_wright.api.config import EngineConfig
13
+ from rag_wright.models.profiles import ModelRole, model_for
14
+
15
+ logger = logging.getLogger(__name__)
16
+
17
+ # Cached per (backend, host, port, corpus) -- one resolved workspace per customer DB (absorbs the product's
18
+ # per-customer graph/store cache). Process-local; the embedder is lazy + per-handle.
19
+ _WORKSPACES: dict[tuple, "WorkspaceHandle"] = {}
20
+
21
+
22
+ class WorkspaceHandle:
23
+ """An opaque handle to a resolved engine workspace. Public surface: `model_id(role)`. The resolved store +
24
+ embedder are engine-internal (`_store` / `_embedder`), used by the invokers -- NOT a product accessor."""
25
+
26
+ def __init__(self, store: Any, config: EngineConfig, corpus: str) -> None:
27
+ self._store = store
28
+ self._config = config
29
+ self._corpus = corpus
30
+ self.__embedder: Any = None
31
+ self.__embedder_built = False
32
+
33
+ @property
34
+ def _embedder(self) -> Any:
35
+ """The query embedder for this workspace, built lazily on first use (so an ingest-only workspace -- or a
36
+ test -- needs no embedder backend). Best-effort: unavailable -> None (the consumer surfaces it)."""
37
+ if not self.__embedder_built:
38
+ self.__embedder = _build_embedder(self._config)
39
+ self.__embedder_built = True
40
+ return self.__embedder
41
+
42
+ def model_id(self, role: ModelRole) -> str:
43
+ """Resolve a model role to its id: the `EngineConfig.models` override wins, else the profile default."""
44
+ key = role.value if isinstance(role, ModelRole) else str(role)
45
+ return self._config.models.get(key) or model_for(role)
46
+
47
+
48
+ def open_workspace(config: EngineConfig, *, corpus: str, reset: bool = False) -> WorkspaceHandle:
49
+ """Resolve (and cache) the workspace for `corpus` (the backend database name) from `config`. Ensures the schema.
50
+ Returns an opaque `WorkspaceHandle`. `reset=True` drops + recreates the database (test/clean-slate) and bypasses
51
+ the cache."""
52
+ sc = config.store
53
+ cache_key = (sc.backend, sc.host, sc.port, corpus)
54
+ if not reset and cache_key in _WORKSPACES:
55
+ return _WORKSPACES[cache_key]
56
+ store = _build_store(config, corpus, reset=reset)
57
+ store.ensure_schema()
58
+ handle = WorkspaceHandle(store, config, corpus)
59
+ _WORKSPACES[cache_key] = handle
60
+ return handle
61
+
62
+
63
+ def _build_store(config: EngineConfig, corpus: str, *, reset: bool) -> Any:
64
+ sc = config.store
65
+ if sc.backend != "arcadedb":
66
+ raise ValueError(f"unsupported store backend: {sc.backend!r}") # a new backend plugs in here (DD)
67
+ from rag_wright.store.arcadedb import ArcadeDBStore
68
+
69
+ return ArcadeDBStore.from_config(sc.host, sc.port, sc.user, sc.password, database=corpus,
70
+ protocol=sc.protocol, reset=reset, pack_ttl=config.pack)
71
+
72
+
73
+ def _build_embedder(config: EngineConfig) -> Optional[Any]:
74
+ """The workspace's QUERY embedder, selected by the `text` embedding profile (EP-API-4b; default bge-m3). An
75
+ unknown profile raises; an unavailable backend degrades to None (the retrieval consumer surfaces it)."""
76
+ from rag_wright.capabilities.embedding_profiles import build_query_embedder
77
+
78
+ profile = config.embeddings.get("text", "bge-m3")
79
+ try:
80
+ return build_query_embedder(profile)
81
+ except ValueError:
82
+ raise # an unknown profile is a config error, not a transient backend failure
83
+ except Exception: # noqa: BLE001 - embedder backend unavailable -> None; the retrieval consumer surfaces it
84
+ logger.warning("engine_embedder_unavailable", exc_info=True)
85
+ return None
@@ -0,0 +1,8 @@
1
+ """The FR-C capability catalog, each built and registered under its FR-C name.
2
+
3
+ FR-C.1 Parsing (Docling), FR-C.2 Embedding (BGE-M3), FR-C.3 Hybrid search (ArcadeDB),
4
+ FR-C.4 Reranking (BGE-reranker), FR-C.5 Graph query (ArcadeDB), FR-C.6 Graph extraction,
5
+ FR-C.7 Entity resolution, FR-C.8 Ontology and registry derivation,
6
+ FR-C.9 Reasoning, generation, and vision-to-text (Gemma 4), FR-C.10 RLM skill.
7
+ Plus the ingestion-side (FR-I) and query-side (FR-Q) capability behavior.
8
+ """