rag-wright 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (184) hide show
  1. rag_wright/__init__.py +13 -0
  2. rag_wright/api/__init__.py +33 -0
  3. rag_wright/api/config.py +59 -0
  4. rag_wright/api/discover.py +70 -0
  5. rag_wright/api/documents.py +39 -0
  6. rag_wright/api/ids.py +31 -0
  7. rag_wright/api/invoke.py +99 -0
  8. rag_wright/api/kg.py +61 -0
  9. rag_wright/api/mcp.py +94 -0
  10. rag_wright/api/usage.py +30 -0
  11. rag_wright/api/workspace.py +85 -0
  12. rag_wright/capabilities/__init__.py +8 -0
  13. rag_wright/capabilities/answer_generator.py +427 -0
  14. rag_wright/capabilities/ard.py +286 -0
  15. rag_wright/capabilities/assertion_extraction.py +79 -0
  16. rag_wright/capabilities/chunk_read.py +58 -0
  17. rag_wright/capabilities/chunk_write.py +163 -0
  18. rag_wright/capabilities/claim_extraction.py +153 -0
  19. rag_wright/capabilities/clause_exception_linking.py +117 -0
  20. rag_wright/capabilities/compliance_judgment.py +322 -0
  21. rag_wright/capabilities/compliance_store.py +87 -0
  22. rag_wright/capabilities/contract_kg_serve.py +156 -0
  23. rag_wright/capabilities/contract_kg_store.py +251 -0
  24. rag_wright/capabilities/dg_extraction.py +585 -0
  25. rag_wright/capabilities/disambiguation.py +163 -0
  26. rag_wright/capabilities/document_parse.py +87 -0
  27. rag_wright/capabilities/document_scope.py +49 -0
  28. rag_wright/capabilities/embedding.py +164 -0
  29. rag_wright/capabilities/embedding_profiles.py +43 -0
  30. rag_wright/capabilities/entity_resolution.py +154 -0
  31. rag_wright/capabilities/fusion.py +64 -0
  32. rag_wright/capabilities/graph_extraction.py +243 -0
  33. rag_wright/capabilities/graph_query.py +73 -0
  34. rag_wright/capabilities/graph_storage.py +111 -0
  35. rag_wright/capabilities/highlight_serve.py +142 -0
  36. rag_wright/capabilities/hybrid_search.py +65 -0
  37. rag_wright/capabilities/invoke.py +31 -0
  38. rag_wright/capabilities/jev_decision.py +38 -0
  39. rag_wright/capabilities/manifests.py +872 -0
  40. rag_wright/capabilities/okf_navigate.py +456 -0
  41. rag_wright/capabilities/parsing.py +286 -0
  42. rag_wright/capabilities/property_boosted_retrieval.py +125 -0
  43. rag_wright/capabilities/query_function_classifier.py +94 -0
  44. rag_wright/capabilities/query_understanding.py +109 -0
  45. rag_wright/capabilities/registry.py +262 -0
  46. rag_wright/capabilities/remote_encoders.py +94 -0
  47. rag_wright/capabilities/requirement_extraction.py +247 -0
  48. rag_wright/capabilities/reranking.py +123 -0
  49. rag_wright/capabilities/retrieval_core.py +126 -0
  50. rag_wright/capabilities/rlm_chunking.py +808 -0
  51. rag_wright/capabilities/rlm_synthesis.py +316 -0
  52. rag_wright/capabilities/scan_quality.py +136 -0
  53. rag_wright/capabilities/span_relevance_judgment.py +191 -0
  54. rag_wright/capabilities/vision_to_text.py +85 -0
  55. rag_wright/capabilities/vlm_ocr.py +85 -0
  56. rag_wright/contracts/__init__.py +6 -0
  57. rag_wright/contracts/chunk.py +79 -0
  58. rag_wright/contracts/compliance.py +303 -0
  59. rag_wright/contracts/contract_meta.py +27 -0
  60. rag_wright/contracts/extraction.py +130 -0
  61. rag_wright/contracts/function.py +167 -0
  62. rag_wright/contracts/function_routing.py +91 -0
  63. rag_wright/contracts/highlight.py +74 -0
  64. rag_wright/contracts/identifiers.py +153 -0
  65. rag_wright/contracts/jurisdiction.py +96 -0
  66. rag_wright/contracts/ontology.py +142 -0
  67. rag_wright/contracts/property.py +201 -0
  68. rag_wright/contracts/provenance.py +78 -0
  69. rag_wright/contracts/query_intent.py +53 -0
  70. rag_wright/contracts/span.py +76 -0
  71. rag_wright/contracts/value_match.py +84 -0
  72. rag_wright/corpus/__init__.py +0 -0
  73. rag_wright/corpus/canonicalize.py +116 -0
  74. rag_wright/corpus/cuad.py +153 -0
  75. rag_wright/corpus/cuad_ingestion.py +72 -0
  76. rag_wright/corpus/document_parser.py +299 -0
  77. rag_wright/corpus/edgar.py +231 -0
  78. rag_wright/corpus/gcs_ingestion.py +120 -0
  79. rag_wright/corpus/http.py +110 -0
  80. rag_wright/corpus/selection.py +152 -0
  81. rag_wright/mcp/__init__.py +11 -0
  82. rag_wright/mcp/compliance_server.py +299 -0
  83. rag_wright/mcp/intra_document_qa_server.py +170 -0
  84. rag_wright/mcp/relational_qa_server.py +171 -0
  85. rag_wright/mcp/session_store.py +64 -0
  86. rag_wright/mcp/typed_property_retrieval_server.py +191 -0
  87. rag_wright/models/__init__.py +8 -0
  88. rag_wright/models/profiles.py +331 -0
  89. rag_wright/models/seam.py +497 -0
  90. rag_wright/models/tag_structured.py +285 -0
  91. rag_wright/models/tracing.py +179 -0
  92. rag_wright/models/usage.py +102 -0
  93. rag_wright/okf/__init__.py +11 -0
  94. rag_wright/okf/compile.py +292 -0
  95. rag_wright/okf/document.py +47 -0
  96. rag_wright/okf/enrich.py +176 -0
  97. rag_wright/okf/links.py +190 -0
  98. rag_wright/okf/lint.py +105 -0
  99. rag_wright/ontology/__init__.py +6 -0
  100. rag_wright/ontology/_generated_template_meta.py +60 -0
  101. rag_wright/ontology/_generated_vocab.py +52 -0
  102. rag_wright/ontology/clause_template.py +964 -0
  103. rag_wright/ontology/codegen.py +84 -0
  104. rag_wright/ontology/compliance_bridge.ttl +186 -0
  105. rag_wright/ontology/contract_bridge.ttl +2685 -0
  106. rag_wright/ontology/contract_taxonomy.py +24 -0
  107. rag_wright/ontology/derive.py +58 -0
  108. rag_wright/ontology/loader.py +435 -0
  109. rag_wright/ontology/packs/ftc_16cfr255.ttl +29 -0
  110. rag_wright/ontology/registry.py +87 -0
  111. rag_wright/ontology/template_introspect.py +100 -0
  112. rag_wright/py.typed +0 -0
  113. rag_wright/reference/__init__.py +2 -0
  114. rag_wright/reference/compliance.py +41 -0
  115. rag_wright/reference/contract_seam.py +123 -0
  116. rag_wright/skills/__init__.py +7 -0
  117. rag_wright/skills/claim_extraction/SKILL.md +47 -0
  118. rag_wright/skills/claim_extraction/__init__.py +1 -0
  119. rag_wright/skills/claim_extraction/template.py +50 -0
  120. rag_wright/skills/compliance_judgment/SKILL.md +59 -0
  121. rag_wright/skills/corpus_ingest/SKILL.md +106 -0
  122. rag_wright/skills/extraction_semantic_judge/SKILL.md +51 -0
  123. rag_wright/skills/extraction_semantic_judge/__init__.py +1 -0
  124. rag_wright/skills/generation/SKILL.md +64 -0
  125. rag_wright/skills/generation/__init__.py +1 -0
  126. rag_wright/skills/generic_compliance_judgment/SKILL.md +58 -0
  127. rag_wright/skills/okf_navigate/SKILL.md +137 -0
  128. rag_wright/skills/requirement_extraction/SKILL.md +47 -0
  129. rag_wright/skills/requirement_extraction/__init__.py +1 -0
  130. rag_wright/skills/requirement_extraction/template.py +50 -0
  131. rag_wright/skills/rlm/SKILL.md +186 -0
  132. rag_wright/skills/rlm/__init__.py +31 -0
  133. rag_wright/skills/rlm/agent.py +292 -0
  134. rag_wright/skills/span_relevance_judgment/SKILL.md +67 -0
  135. rag_wright/skills/vision_to_text/SKILL.md +36 -0
  136. rag_wright/skills/vision_to_text/__init__.py +1 -0
  137. rag_wright/spans/__init__.py +1 -0
  138. rag_wright/spans/boundary.py +78 -0
  139. rag_wright/spans/clause_function_classifier.py +490 -0
  140. rag_wright/spans/clause_kg_extractor.py +337 -0
  141. rag_wright/spans/cuad_labels.py +81 -0
  142. rag_wright/spans/dim_classifier.py +158 -0
  143. rag_wright/spans/dim_fleet.json +411 -0
  144. rag_wright/spans/function_classifier.py +77 -0
  145. rag_wright/spans/function_families.py +62 -0
  146. rag_wright/spans/hybrid_classifier.py +103 -0
  147. rag_wright/spans/legalbert_classifier.py +83 -0
  148. rag_wright/spans/model_capabilities.py +107 -0
  149. rag_wright/spans/new_function_labels.py +111 -0
  150. rag_wright/spans/page_map.py +68 -0
  151. rag_wright/spans/property_extractor.py +365 -0
  152. rag_wright/spans/property_grounding.py +182 -0
  153. rag_wright/spans/reclassify.py +77 -0
  154. rag_wright/spans/scarce_function_labels.py +105 -0
  155. rag_wright/spans/segment.py +341 -0
  156. rag_wright/spans/semantic_judge.py +197 -0
  157. rag_wright/spans/symbolic_validation.py +131 -0
  158. rag_wright/spans/tag_clause_extractor.py +182 -0
  159. rag_wright/store/__init__.py +6 -0
  160. rag_wright/store/arcadedb.py +1135 -0
  161. rag_wright/store/chunk_text.py +66 -0
  162. rag_wright/store/seam.py +213 -0
  163. rag_wright/subgraphs/__init__.py +0 -0
  164. rag_wright/subgraphs/async_ingestion.py +204 -0
  165. rag_wright/subgraphs/compliance_check.py +1042 -0
  166. rag_wright/subgraphs/compliance_ingestion.py +306 -0
  167. rag_wright/subgraphs/contract_ingestion_pipeline.py +999 -0
  168. rag_wright/subgraphs/graph_extraction.py +102 -0
  169. rag_wright/subgraphs/intra_document_qa.py +328 -0
  170. rag_wright/subgraphs/observability.py +140 -0
  171. rag_wright/subgraphs/query_constraint_extraction.py +73 -0
  172. rag_wright/subgraphs/relational_qa.py +165 -0
  173. rag_wright/subgraphs/requirement_extraction.py +137 -0
  174. rag_wright/subgraphs/scaffold.py +65 -0
  175. rag_wright/subgraphs/semantic_chunking.py +183 -0
  176. rag_wright/subgraphs/typed_clause_extraction.py +172 -0
  177. rag_wright/subgraphs/typed_property_retrieval.py +278 -0
  178. rag_wright/util/__init__.py +1 -0
  179. rag_wright/util/concurrent.py +153 -0
  180. rag_wright/util/spacy_model.py +45 -0
  181. rag_wright-0.1.0.dist-info/METADATA +168 -0
  182. rag_wright-0.1.0.dist-info/RECORD +184 -0
  183. rag_wright-0.1.0.dist-info/WHEEL +4 -0
  184. rag_wright-0.1.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,331 @@
1
+ """Per-model profiles and role resolution for the model seam (T11).
2
+
3
+ A profile is keyed by model id and carries only what the seam needs to make a *structured-output*
4
+ call safely: the structured-output method (default `function_calling`, more broadly supported across
5
+ open models than `json_schema`) and an optional structured-only `extra_body`. The `extra_body` is
6
+ applied by the seam only to the forced structured call (for example to disable thinking on the
7
+ structured emit), leaving free-text and reasoning calls unaffected. Provider/model flags live here
8
+ in config and in a dated ADR, never in a capability or node call site (CLAUDE.md standing rule).
9
+
10
+ Model priority for the structured-output-under-reasoning call class is a config decision, not
11
+ capability code: DeepSeek V4 Pro is the primary, Qwen 3.7 Plus the selectable secondary, and the
12
+ Gemma 4 class is the general / local-deployment default (SPEC section 4, tech stack; Phase 2 ledger).
13
+ The exact OpenRouter slug and the empirical structured profile (method + `extra_body`) for the
14
+ structured-reasoning model are confirmed against a live call at T12 and recorded in its ADR; each id
15
+ below is a documented default, overridable by env so the empirical slug needs no code change.
16
+ """
17
+
18
+ from __future__ import annotations
19
+
20
+ import os
21
+ from enum import Enum
22
+ from typing import Any, Literal, Optional
23
+
24
+ from pydantic import BaseModel, ConfigDict, Field
25
+
26
+ StructuredMethod = Literal["function_calling", "json_mode", "json_schema"]
27
+
28
+
29
+ class ModelProfile(BaseModel):
30
+ """How a given model id must be driven for a forced structured-output call."""
31
+
32
+ # `model_id` sits in Pydantic's protected `model_` namespace; opt out (it is a plain field).
33
+ model_config = ConfigDict(frozen=True, protected_namespaces=())
34
+
35
+ model_id: str
36
+ structured_method: StructuredMethod = "function_calling"
37
+ # Some serving stacks cannot do SERVER-SIDE grammar-constrained structured output at all (self-hosted
38
+ # Gemma 4 on vLLM 0.26: json_schema/json_mode run away to max_model_len; function_calling needs a buggy
39
+ # tool-parser -- while FREE-TEXT is perfect). Flag those models so generation uses the CLIENT-SIDE
40
+ # free-text + tag-parse path (answer_generator.answer_model_for) instead of the guided-decoding seam.
41
+ client_side_structured: bool = False
42
+ # Applied by the seam only to the forced structured call, never to the base client.
43
+ structured_extra_body: Optional[dict[str, Any]] = Field(default=None)
44
+ # The free-text counterpart to `structured_extra_body`: applied by the seam to the FREE-TEXT generation path
45
+ # (`astream_text` / the client-side tag-parse path), never to a forced structured call. A reasoning model must
46
+ # have its reasoning EXPLICITLY set on this path too -- leaving it unset falls to the provider default, which
47
+ # for qwen3.8 on the streaming path is inconsistent (intermittently returns reasoning-only / empty content).
48
+ # `structured_extra_body` does not reach here (it binds only to `with_structured_output`), so this is a separate
49
+ # slot; it also lets the free-text extraction use a DIFFERENT reasoning setting than the forced-structured judge.
50
+ text_extra_body: Optional[dict[str, Any]] = Field(default=None)
51
+ # Applied by the seam to the BASE client (every call to this model). Carries request-level provider
52
+ # routing (e.g. OpenRouter `{"provider": {"sort": "throughput"}}`) -- a provider flag, so it lives in
53
+ # config + a dated ADR, never in node/agent code (ADR-0027). Grounded: `extra_body` is a real
54
+ # `BaseChatOpenAI` field for exactly this purpose.
55
+ extra_body: Optional[dict[str, Any]] = Field(default=None)
56
+
57
+ # ADR-0100: the profile also carries HOW to REACH the model, so a model STRING fully describes both the
58
+ # model and its access -- the engine builds the client from this, and callers only need the string. A string
59
+ # can pin a backend (mix OpenRouter + self-hosted vLLM/Modal by using different strings), or leave `backend`
60
+ # None to fall back to the global `RAG_SERVING` default (back-compat).
61
+ backend: Optional[Literal["openrouter", "vllm", "ollama"]] = None # None -> RAG_SERVING default
62
+ # the id the backend actually expects (OpenRouter slug / vLLM `--served-model-name`), often != our string.
63
+ served_model_id: Optional[str] = None # default = model_id
64
+ base_url_env: Optional[str] = None # env var holding the base_url; default per backend (VLLM_BASE_URL, ...)
65
+ api_key_env: Optional[str] = None # env var holding the api key; default per backend (VLLM_API_KEY, ...)
66
+
67
+
68
+ class ModelRole(str, Enum):
69
+ """Which model does which job. The mapping to ids lives in config, not in capability code."""
70
+
71
+ STRUCTURED_REASONING = "structured_reasoning" # forced schema under reasoning: extraction, grading, synthesis
72
+ STRUCTURED_REASONING_SECONDARY = "structured_reasoning_secondary" # same call class, selectable fallback
73
+ GENERAL = "general" # reasoning, generation, vision-to-text, RLM; the local-deployment default
74
+ SUMMARIZATION = "summarization" # a smaller model for chunking and summarization (FR-I.6 tiering)
75
+ OKF_ENRICHMENT = "okf_enrichment" # cheap classify + one-line description for OKF signposts (FR-K.2, ADR-0023)
76
+ # issue 0005: the ingest clause-function classifier as its OWN role, so it can run on a different model than
77
+ # GENERAL (e.g. Gemma-4) without moving the other stages. Route (b) is CLIENT-SIDE tag-parse -> works on any
78
+ # model. Defaults to the product LLM (unchanged behavior); set RAG_MODEL_FUNCTION_CLASSIFY to override.
79
+ FUNCTION_CLASSIFY = "function_classify"
80
+ # 0009-VLM: VLM-based OCR escalation for degraded scans (via docling ApiVlmOptions -> OpenRouter). Defaults to
81
+ # Gemma-4 (a vision model, unlike the Granite product LLM); set RAG_MODEL_VISION_OCR to swap the model.
82
+ VISION_OCR = "vision_ocr"
83
+
84
+
85
+ # Default model ids per role, confirmed against the live OpenRouter catalog at T12 (ADR-0006).
86
+ # Overridable by env (`_ROLE_ENV`), so a later slug change stays config, not code.
87
+ DEFAULT_STRUCTURED_REASONING = "deepseek/deepseek-v4-pro"
88
+ DEFAULT_STRUCTURED_REASONING_SECONDARY = "qwen/qwen3.7-plus"
89
+ DEFAULT_GENERAL = "google/gemma-4-31b-it"
90
+ DEFAULT_SUMMARIZATION = "deepseek/deepseek-v4-flash" # the smaller/faster DeepSeek (FR-I.6)
91
+ # OKF signpost enrichment is a simple classify-and-describe task; a cheap Gemma matched DeepSeek V4 Pro
92
+ # on it (100% category agreement, good one-liners, ~4x cheaper/faster) at the 2026-07-22 bench (ADR-0023).
93
+ # THIS TASK ONLY; every other call class stays on its DeepSeek/Gemma role above.
94
+ DEFAULT_OKF_ENRICHMENT = "google/gemma-4-26b-a4b-it"
95
+
96
+ # MS1-2 (ADR-0039): the product substrate is a SINGLE self-hosted model on the A100. EVERY role defaults to
97
+ # Granite -- Gemma/DeepSeek are DROPPED from the product default but stay REGISTERED in `PROFILES` below, so a
98
+ # dev run can still select any of them via the `RAG_MODEL_*` / `RAG_MODEL_ALL` env overrides. The DEFAULT_*
99
+ # constants above are kept as those foundation-model profile keys + documented dev-override values. (OKF
100
+ # signpost enrichment -- the one-time ADR-0023 Gemma exception -- is NOT wired into the ingestion/query
101
+ # pipeline (only `okf/enrich.py`), so it too defaults to Granite; ADR-0023's Gemma choice is now vestigial.)
102
+ _PRODUCT_LLM = "qwen3.8-27b-modal-or" # engine-wide default (ADR-0100): Qwen via OpenRouter for now
103
+
104
+ _ROLE_ENV: dict[ModelRole, tuple[str, str]] = {
105
+ ModelRole.STRUCTURED_REASONING: ("RAG_MODEL_STRUCTURED_REASONING", _PRODUCT_LLM),
106
+ ModelRole.STRUCTURED_REASONING_SECONDARY: ("RAG_MODEL_STRUCTURED_REASONING_SECONDARY", _PRODUCT_LLM),
107
+ ModelRole.GENERAL: ("RAG_MODEL_GENERAL", _PRODUCT_LLM),
108
+ ModelRole.SUMMARIZATION: ("RAG_MODEL_SUMMARIZATION", _PRODUCT_LLM),
109
+ ModelRole.OKF_ENRICHMENT: ("RAG_MODEL_OKF_ENRICHMENT", _PRODUCT_LLM),
110
+ ModelRole.FUNCTION_CLASSIFY: ("RAG_MODEL_FUNCTION_CLASSIFY", _PRODUCT_LLM),
111
+ # 0009-VLM: defaults to Gemma-4 (a vision model), NOT the Granite product LLM -- OCR needs vision.
112
+ ModelRole.VISION_OCR: ("RAG_MODEL_VISION_OCR", DEFAULT_GENERAL),
113
+ }
114
+
115
+ # Registered profiles keyed by model id. A model without an entry falls back to the safe default
116
+ # (function_calling, no extra_body) via `profile_for`, so no call site special-cases a model.
117
+ # Structured-output methods and extra bodies below are empirical, confirmed by a live forced-schema
118
+ # call at T12 and recorded in ADR-0006.
119
+ PROFILES: dict[str, ModelProfile] = {
120
+ # DeepSeek V4 Pro honors the forced tool call while reasoning; no thinking-disable needed. Route by
121
+ # THROUGHPUT so OpenRouter prefers the fastest provider over the cheapest (which throttled the bulk
122
+ # property extraction, T58); keeps V4 Pro, model rule intact (ADR-0027).
123
+ DEFAULT_STRUCTURED_REASONING: ModelProfile(
124
+ model_id=DEFAULT_STRUCTURED_REASONING,
125
+ extra_body={"provider": {"sort": "throughput"}},
126
+ ),
127
+ # Qwen 3.7 Plus rejects `tool_choice` object/required in thinking mode ("<400> ... does not
128
+ # support being set to required or object in thinking mode"); disabling reasoning on the forced
129
+ # structured call alone fixes it, leaving its free-text/reasoning calls untouched (ADR-0006).
130
+ DEFAULT_STRUCTURED_REASONING_SECONDARY: ModelProfile(
131
+ model_id=DEFAULT_STRUCTURED_REASONING_SECONDARY,
132
+ structured_extra_body={"reasoning": {"enabled": False}},
133
+ ),
134
+ # Gemma 4, like Qwen, rejects/returns-None on forced structured output when its reasoning mode is on for
135
+ # richer schemas (nullable fields): the CU-C1 NL->type emit returned None on every call until reasoning was
136
+ # disabled on the forced structured call. Same fix as the secondary (structured-only, so free-text/reasoning
137
+ # calls -- and the two-step reason node -- are untouched); simpler schemas (chunking _BoundaryList/_Summary)
138
+ # verified still valid with it. Empirical, dated: CU-D2 / ADR-0032.
139
+ DEFAULT_GENERAL: ModelProfile(
140
+ model_id=DEFAULT_GENERAL,
141
+ structured_extra_body={"reasoning": {"enabled": False}},
142
+ ),
143
+ # DeepSeek V4 Flash does reasoning + structured output together, like V4 Pro (ADR-0006); no
144
+ # thinking-disable needed. As the T58 BULK property extractor (Flash->Pro cascade, ADR-0028) it routes
145
+ # by THROUGHPUT too, to dodge the cheapest-provider throttle (ADR-0027).
146
+ DEFAULT_SUMMARIZATION: ModelProfile(
147
+ model_id=DEFAULT_SUMMARIZATION,
148
+ extra_body={"provider": {"sort": "throughput"}},
149
+ ),
150
+ # Gemma 4 26b-a4b takes the forced tool call cleanly (default function_calling, no extra_body);
151
+ # 0 structured-output errors across the 20-clause bench (ADR-0023).
152
+ DEFAULT_OKF_ENRICHMENT: ModelProfile(model_id=DEFAULT_OKF_ENRICHMENT),
153
+ # The ADOPTED default (2026-09-04): DeepSeek V4 Flash via OpenRouter's auto-updating `~...-latest` alias.
154
+ # It does reasoning + structured output together (function_calling, no thinking-disable needed) and returns
155
+ # content directly. The `~...-latest` alias otherwise routes to a SLOW/flaky provider (measured 6.9s vs 1.0s),
156
+ # so pin `provider.sort=latency` -- OpenRouter routes to the lowest-latency provider (applied to every seam
157
+ # call; the docling-graph extraction path injects the same routing separately in `dg_extraction`). Replaces the
158
+ # de-listed `ibm-granite/granite-4.1-8b` (OpenRouter 404: no endpoints); granite-4.2-8b needed reasoning
159
+ # disabled to avoid empty content, so DeepSeek Flash is the lower-friction default.
160
+ # Product default (ADR-0079). Granite-4.1-8b was de-listed on OpenRouter (404); granite-4.2-8b is its direct
161
+ # successor and the replacement default -- same family, same reasoning-off handling, and RELIABLE on the
162
+ # docling-graph extraction path (measured 5/5 clean clause extractions vs deepseek-v4-flash's flaky ~2/5,
163
+ # which produced "no models" a large fraction of the time regardless of provider). `reasoning:{enabled:false}`
164
+ # is LOAD-BEARING: granite-4.2 is a reasoning model and returns empty `content` on a forced structured call
165
+ # unless reasoning is disabled. `provider:{sort:latency}` picks the fastest of its (few) providers.
166
+ "ibm-granite/granite-4.2-8b": ModelProfile(
167
+ model_id="ibm-granite/granite-4.2-8b",
168
+ extra_body={"provider": {"sort": "latency"}, "reasoning": {"enabled": False}},
169
+ ),
170
+ # Qwen 3.8 27b: UNLIKE qwen3.7-plus, it ACCEPTS a forced `tool_choice` object/required WHILE reasoning is ON
171
+ # (measured 2026-09-08), so we keep its deep reasoning on the single-shot forced-structured path (the compliance
172
+ # judge) -- that reasoning is the point of picking Qwen. `structured_extra_body={"reasoning":{"enabled":True}}`
173
+ # EXPLICITLY forces reasoning on the forced structured call: leaving it UNSET falls to the provider default,
174
+ # which for a forced tool call does little/no reasoning (measured ~5s + shallow vs ~32s deep) -- the opposite of
175
+ # the intended "reasoning-on, accept the latency" choice. Measured trade-offs on the judge: reasoning-ON ~32s
176
+ # (deep, chosen), reasoning-OFF ~5s (shallow).
177
+ # `text_extra_body={"reasoning":{"enabled":False}}` (issue 0020): the FREE-TEXT/tag-parse path (query constraint
178
+ # extraction) needs reasoning set EXPLICITLY too -- unset, qwen3.8 on the streaming path intermittently returns
179
+ # empty content (~1/3 of runs, measured), dropping the query's constraints. OFF (not ON) because constraint
180
+ # extraction is mechanical: OFF is deterministic + cheaper and drops the redundant raw-phrase `cap_quantum` that
181
+ # reasoning-ON adds. So the judge reasons deeply while query extraction does not -- two settings, two slots.
182
+ # provider PIN (ADR-0027, dated 2026-09-20): `{"provider":{"only":["deepinfra/bf16"],"allow_fallbacks":false}}`
183
+ # HARD-pins Qwen3.8-27B to DeepInfra's bf16 endpoint on OpenRouter -- no fallbacks, so OpenRouter can never
184
+ # route to a quantized (fp8/int) provider variant. Guarantees full-precision weights for consistent behavior
185
+ # (supersedes the earlier `sort:throughput`). Applies to all three OpenRouter Qwen3.8-27b strings below; the
186
+ # extraction surface honors the same pin via the profile (dg_extraction), so it holds engine-wide.
187
+ "qwen/qwen3.8-27b": ModelProfile(
188
+ model_id="qwen/qwen3.8-27b",
189
+ structured_extra_body={"reasoning": {"enabled": True}},
190
+ text_extra_body={"reasoning": {"enabled": False}},
191
+ extra_body={"provider": {"only": ["deepinfra/bf16"], "allow_fallbacks": False}},
192
+ ),
193
+ # ADR-0100: backend-PINNED strings for the same Qwen3.8-27B -- pick the string, get the backend. `-or` routes
194
+ # to OpenRouter (slug qwen/qwen3.8-27b, its provider-routing + reasoning-field flags); `-modal` routes to the
195
+ # self-hosted vLLM server (served-name Qwen/Qwen3.8-27B, reasoning via vLLM's `chat_template_kwargs`, and NO
196
+ # OpenRouter `provider` routing). The product uses one string; mixing backends per stage is just two strings.
197
+ "qwen3.8-27b-or": ModelProfile(
198
+ model_id="qwen3.8-27b-or", backend="openrouter", served_model_id="qwen/qwen3.8-27b",
199
+ structured_extra_body={"reasoning": {"enabled": True}},
200
+ text_extra_body={"reasoning": {"enabled": False}},
201
+ extra_body={"provider": {"only": ["deepinfra/bf16"], "allow_fallbacks": False}},
202
+ ),
203
+ "qwen3.8-27b-modal": ModelProfile(
204
+ model_id="qwen3.8-27b-modal", backend="vllm", served_model_id="Qwen/Qwen3.8-27B",
205
+ # vLLM controls Qwen3 reasoning via chat_template_kwargs (enable_thinking), not OpenRouter's reasoning
206
+ # field; base_url/key come from VLLM_BASE_URL/VLLM_API_KEY (override with base_url_env for a 2nd server).
207
+ structured_extra_body={"chat_template_kwargs": {"enable_thinking": True}},
208
+ text_extra_body={"chat_template_kwargs": {"enable_thinking": False}},
209
+ ),
210
+ # Reasoning-OFF variant of `qwen3.8-27b-modal` for BULK closed-vocab classification (e.g. silver mining): the
211
+ # forced structured pick needs no chain-of-thought, and thinking-on is ~5-10x slower/costlier per call on the
212
+ # self-hosted A100. Same served name + function_calling (the server's hermes tool-parser handles it); only the
213
+ # structured call's `enable_thinking` flips to False. Not a default anywhere — opt in via the model string.
214
+ "qwen3.8-27b-modal-nothink": ModelProfile(
215
+ model_id="qwen3.8-27b-modal-nothink", backend="vllm", served_model_id="Qwen/Qwen3.8-27B",
216
+ structured_extra_body={"chat_template_kwargs": {"enable_thinking": False}},
217
+ text_extra_body={"chat_template_kwargs": {"enable_thinking": False}},
218
+ ),
219
+ # Quantization ACCURACY EVAL profile (ADR-0110 follow-up): points the JUDGE role at a raw Modal vLLM endpoint
220
+ # (`scripts/modal_qwen3_vllm_server.py`, served as "qwen3-eval") so bf16-reference vs FP8 configs are the SAME
221
+ # profile with only VLLM_BASE_URL swapped between deployments. Reasoning ON for the structured judge (matches
222
+ # the production `-modal-or` judge), off for free-text. base_url/key from VLLM_BASE_URL/VLLM_API_KEY.
223
+ # Structured via vLLM NATIVE guided decoding (`json_schema`/xgrammar) -- raw vLLM 400s on the default
224
+ # `function_calling` ("tool_choice=function requires --tool-call-parser"), and open models reject a forced
225
+ # schema while THINKING, so thinking is OFF on the forced structured call (the self-hosted pattern, matching
226
+ # the Gemma profiles). Free-text stays thinking-off too. ref vs FP8 use the SAME config -> a clean delta.
227
+ "qwen3-eval": ModelProfile(
228
+ model_id="qwen3-eval", backend="vllm", served_model_id="qwen3-eval",
229
+ structured_method="json_schema",
230
+ structured_extra_body={"chat_template_kwargs": {"enable_thinking": False}},
231
+ text_extra_body={"chat_template_kwargs": {"enable_thinking": False}},
232
+ ),
233
+ # THE ENGINE-WIDE DEFAULT (_PRODUCT_LLM / _PRODUCT_EXTRACT_DEFAULT / graph-extract default all point here).
234
+ # "-modal-or": the model destined for our Modal/vLLM deployment, but routed via OPENROUTER FOR NOW because the
235
+ # Modal cold-start/warmup is unsolved (snapshot path fails today). WHEN warmup is solved, flip THIS ONE entry's
236
+ # `backend` to "vllm" + `served_model_id="Qwen/Qwen3.8-27B"` + the vLLM reasoning flags, and every default
237
+ # follows -- no other change. Carries the OpenRouter Qwen flags (reasoning on judge / off free-text).
238
+ "qwen3.8-27b-modal-or": ModelProfile(
239
+ model_id="qwen3.8-27b-modal-or", backend="openrouter", served_model_id="qwen/qwen3.8-27b",
240
+ structured_extra_body={"reasoning": {"enabled": True}},
241
+ text_extra_body={"reasoning": {"enabled": False}},
242
+ extra_body={"provider": {"only": ["deepinfra/bf16"], "allow_fallbacks": False}},
243
+ ),
244
+ # Kimi-k3 (Moonshot) shows the same `function_calling` degeneracy as granite (empty structured result on
245
+ # some queries); `json_schema` fixes it. Registered only for the KG-6 query-side model comparison (not
246
+ # adopted). Empirical, KG-6 / ADR-0034.
247
+ "moonshotai/kimi-k3": ModelProfile(
248
+ model_id="moonshotai/kimi-k3",
249
+ structured_method="json_schema",
250
+ ),
251
+ # Self-hosted Gemma 4 QAT on vLLM: forced `function_calling` returns HTTP 400 ("tool_choice=function ...
252
+ # requires --tool-call-parser to be set") -- vLLM won't honor a forced named tool without a tool-call
253
+ # parser. `json_schema` routes through vLLM's NATIVE guided decoding (xgrammar), which needs no parser.
254
+ # Note this is a DIFFERENT id from the OpenRouter `google/gemma-4-31b-it` profile above (which uses the
255
+ # OpenRouter-only `{"reasoning": {"enabled": False}}` extra_body -- inapplicable to vLLM). Empirical, dated
256
+ # 2026-08-08: GATE-2 400 on function_calling, clean on json_schema. Candidate self-hosted GENERAL model.
257
+ # NOTE (2026-08-08, dated finding): self-hosted Gemma 4 on vLLM 0.26 could NOT do reliable GRAMMAR-
258
+ # CONSTRAINED structured output -- BOTH `json_schema` and `json_mode` (xgrammar guided decoding) RUN AWAY
259
+ # to max_model_len instead of emitting EOS, while FREE-TEXT generation terminates perfectly (1s, coherent,
260
+ # correct, finish=stop) and even emits inline [chunk_id] citations. function_calling needs
261
+ # --tool-call-parser gemma4 (concurrency <pad> bug, vllm#39392). So structured output on this stack is the
262
+ # open problem, NOT model quality/VRAM. Path forward: generate the answer as FREE-TEXT and extract the
263
+ # GeneratedAnswer (citations, abstained) DETERMINISTICALLY from it, bypassing guided decoding. `json_schema`
264
+ # kept here to match the Granite precedent; it does NOT yet work for this model on vLLM.
265
+ "google/gemma-4-31B-it-qat-w4a16-ct": ModelProfile(
266
+ model_id="google/gemma-4-31B-it-qat-w4a16-ct",
267
+ structured_method="json_schema", # unused: server-side guided decoding runs away on this stack
268
+ client_side_structured=True, # -> free-text + client-side tag parse instead
269
+ ),
270
+ "google/gemma-4-26B-A4B-it": ModelProfile(
271
+ model_id="google/gemma-4-26B-A4B-it",
272
+ structured_method="json_schema",
273
+ client_side_structured=True,
274
+ ),
275
+ }
276
+
277
+
278
+ def model_for(role: ModelRole) -> str:
279
+ """Resolve a role to a model id. Precedence (all config, MS1-2): the ROLE-SPECIFIC override
280
+ (`RAG_MODEL_<ROLE>`) > the ALL-ROLES override (`RAG_MODEL_ALL`, to point every role at one model for a
281
+ quick cross-model test) > the documented default. So a run can swap one role, or every role, purely by env.
282
+ """
283
+ env_var, default = _ROLE_ENV[role]
284
+ return os.getenv(env_var) or os.getenv("RAG_MODEL_ALL") or default
285
+
286
+
287
+ def profile_for(model_id: str) -> ModelProfile:
288
+ """The registered profile for a model id, or a safe default profile for an unregistered one."""
289
+ return PROFILES.get(model_id, ModelProfile(model_id=model_id))
290
+
291
+
292
+ # --- ADR-0119: typed-DECISION model profiles (a separate API surface from the chat/structured LLM profiles) ---
293
+
294
+ class DecisionModelProfile(BaseModel):
295
+ """How to reach a typed-DECISION model (TypeSafe Jev, or an on-prem Laya decisions server). This is a DIFFERENT
296
+ API surface from `ModelProfile`: a decision model POSTs a `state` + typed `questions` to a Decisions endpoint
297
+ and returns calibrated typed answers (`noul`/`choice`/`score`), not chat completions -- so it needs its own
298
+ profile shape (endpoint + served id + key env + default thresholds), not the structured-output fields.
299
+
300
+ Keyed by model id so swapping `jev-1.13` -> `jev-latest`, or Jev -> an on-prem Laya decisions endpoint, is a
301
+ config change, not a code edit (the Laya fallback, ADR-0119). The capability (`capabilities/jev_decision.py`)
302
+ reads this; the model id is never hardcoded in capability/node code (the model-neutrality rule)."""
303
+
304
+ model_config = ConfigDict(frozen=True, protected_namespaces=())
305
+
306
+ model_id: str
307
+ served_model_id: Optional[str] = None # the id the endpoint expects (default = model_id)
308
+ endpoint: str = "https://openrouter.ai/api/alpha/decisions"
309
+ api_key_env: str = "OPENROUTER_API_KEY"
310
+ timeout_s: float = 60.0
311
+ # default decision thresholds (callers may override): yes/no (noul) acceptance + multi-label noul cutoff.
312
+ op_threshold: float = 0.5
313
+ multilabel_threshold: float = 0.6
314
+
315
+ @property
316
+ def served(self) -> str:
317
+ return self.served_model_id or self.model_id
318
+
319
+
320
+ DECISION_PROFILES: dict[str, DecisionModelProfile] = {
321
+ "jev-1.13": DecisionModelProfile(model_id="jev-1.13", served_model_id="typesafe/jev-1.13"),
322
+ "jev-latest": DecisionModelProfile(model_id="jev-latest", served_model_id="typesafe/jev-latest"),
323
+ }
324
+ _DEFAULT_DECISION_MODEL = "jev-1.13"
325
+
326
+
327
+ def decision_profile(model_id: Optional[str] = None) -> DecisionModelProfile:
328
+ """The decision-model profile for `model_id` (or `RAG_DECISION_MODEL`, else the built-in default `jev-1.13`).
329
+ An unregistered id gets a default profile using that id as its served id (so a raw slug still works)."""
330
+ mid = model_id or os.getenv("RAG_DECISION_MODEL") or _DEFAULT_DECISION_MODEL
331
+ return DECISION_PROFILES.get(mid, DecisionModelProfile(model_id=mid, served_model_id=mid))