rag-wright 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- rag_wright/__init__.py +13 -0
- rag_wright/api/__init__.py +33 -0
- rag_wright/api/config.py +59 -0
- rag_wright/api/discover.py +70 -0
- rag_wright/api/documents.py +39 -0
- rag_wright/api/ids.py +31 -0
- rag_wright/api/invoke.py +99 -0
- rag_wright/api/kg.py +61 -0
- rag_wright/api/mcp.py +94 -0
- rag_wright/api/usage.py +30 -0
- rag_wright/api/workspace.py +85 -0
- rag_wright/capabilities/__init__.py +8 -0
- rag_wright/capabilities/answer_generator.py +427 -0
- rag_wright/capabilities/ard.py +286 -0
- rag_wright/capabilities/assertion_extraction.py +79 -0
- rag_wright/capabilities/chunk_read.py +58 -0
- rag_wright/capabilities/chunk_write.py +163 -0
- rag_wright/capabilities/claim_extraction.py +153 -0
- rag_wright/capabilities/clause_exception_linking.py +117 -0
- rag_wright/capabilities/compliance_judgment.py +322 -0
- rag_wright/capabilities/compliance_store.py +87 -0
- rag_wright/capabilities/contract_kg_serve.py +156 -0
- rag_wright/capabilities/contract_kg_store.py +251 -0
- rag_wright/capabilities/dg_extraction.py +585 -0
- rag_wright/capabilities/disambiguation.py +163 -0
- rag_wright/capabilities/document_parse.py +87 -0
- rag_wright/capabilities/document_scope.py +49 -0
- rag_wright/capabilities/embedding.py +164 -0
- rag_wright/capabilities/embedding_profiles.py +43 -0
- rag_wright/capabilities/entity_resolution.py +154 -0
- rag_wright/capabilities/fusion.py +64 -0
- rag_wright/capabilities/graph_extraction.py +243 -0
- rag_wright/capabilities/graph_query.py +73 -0
- rag_wright/capabilities/graph_storage.py +111 -0
- rag_wright/capabilities/highlight_serve.py +142 -0
- rag_wright/capabilities/hybrid_search.py +65 -0
- rag_wright/capabilities/invoke.py +31 -0
- rag_wright/capabilities/jev_decision.py +38 -0
- rag_wright/capabilities/manifests.py +872 -0
- rag_wright/capabilities/okf_navigate.py +456 -0
- rag_wright/capabilities/parsing.py +286 -0
- rag_wright/capabilities/property_boosted_retrieval.py +125 -0
- rag_wright/capabilities/query_function_classifier.py +94 -0
- rag_wright/capabilities/query_understanding.py +109 -0
- rag_wright/capabilities/registry.py +262 -0
- rag_wright/capabilities/remote_encoders.py +94 -0
- rag_wright/capabilities/requirement_extraction.py +247 -0
- rag_wright/capabilities/reranking.py +123 -0
- rag_wright/capabilities/retrieval_core.py +126 -0
- rag_wright/capabilities/rlm_chunking.py +808 -0
- rag_wright/capabilities/rlm_synthesis.py +316 -0
- rag_wright/capabilities/scan_quality.py +136 -0
- rag_wright/capabilities/span_relevance_judgment.py +191 -0
- rag_wright/capabilities/vision_to_text.py +85 -0
- rag_wright/capabilities/vlm_ocr.py +85 -0
- rag_wright/contracts/__init__.py +6 -0
- rag_wright/contracts/chunk.py +79 -0
- rag_wright/contracts/compliance.py +303 -0
- rag_wright/contracts/contract_meta.py +27 -0
- rag_wright/contracts/extraction.py +130 -0
- rag_wright/contracts/function.py +167 -0
- rag_wright/contracts/function_routing.py +91 -0
- rag_wright/contracts/highlight.py +74 -0
- rag_wright/contracts/identifiers.py +153 -0
- rag_wright/contracts/jurisdiction.py +96 -0
- rag_wright/contracts/ontology.py +142 -0
- rag_wright/contracts/property.py +201 -0
- rag_wright/contracts/provenance.py +78 -0
- rag_wright/contracts/query_intent.py +53 -0
- rag_wright/contracts/span.py +76 -0
- rag_wright/contracts/value_match.py +84 -0
- rag_wright/corpus/__init__.py +0 -0
- rag_wright/corpus/canonicalize.py +116 -0
- rag_wright/corpus/cuad.py +153 -0
- rag_wright/corpus/cuad_ingestion.py +72 -0
- rag_wright/corpus/document_parser.py +299 -0
- rag_wright/corpus/edgar.py +231 -0
- rag_wright/corpus/gcs_ingestion.py +120 -0
- rag_wright/corpus/http.py +110 -0
- rag_wright/corpus/selection.py +152 -0
- rag_wright/mcp/__init__.py +11 -0
- rag_wright/mcp/compliance_server.py +299 -0
- rag_wright/mcp/intra_document_qa_server.py +170 -0
- rag_wright/mcp/relational_qa_server.py +171 -0
- rag_wright/mcp/session_store.py +64 -0
- rag_wright/mcp/typed_property_retrieval_server.py +191 -0
- rag_wright/models/__init__.py +8 -0
- rag_wright/models/profiles.py +331 -0
- rag_wright/models/seam.py +497 -0
- rag_wright/models/tag_structured.py +285 -0
- rag_wright/models/tracing.py +179 -0
- rag_wright/models/usage.py +102 -0
- rag_wright/okf/__init__.py +11 -0
- rag_wright/okf/compile.py +292 -0
- rag_wright/okf/document.py +47 -0
- rag_wright/okf/enrich.py +176 -0
- rag_wright/okf/links.py +190 -0
- rag_wright/okf/lint.py +105 -0
- rag_wright/ontology/__init__.py +6 -0
- rag_wright/ontology/_generated_template_meta.py +60 -0
- rag_wright/ontology/_generated_vocab.py +52 -0
- rag_wright/ontology/clause_template.py +964 -0
- rag_wright/ontology/codegen.py +84 -0
- rag_wright/ontology/compliance_bridge.ttl +186 -0
- rag_wright/ontology/contract_bridge.ttl +2685 -0
- rag_wright/ontology/contract_taxonomy.py +24 -0
- rag_wright/ontology/derive.py +58 -0
- rag_wright/ontology/loader.py +435 -0
- rag_wright/ontology/packs/ftc_16cfr255.ttl +29 -0
- rag_wright/ontology/registry.py +87 -0
- rag_wright/ontology/template_introspect.py +100 -0
- rag_wright/py.typed +0 -0
- rag_wright/reference/__init__.py +2 -0
- rag_wright/reference/compliance.py +41 -0
- rag_wright/reference/contract_seam.py +123 -0
- rag_wright/skills/__init__.py +7 -0
- rag_wright/skills/claim_extraction/SKILL.md +47 -0
- rag_wright/skills/claim_extraction/__init__.py +1 -0
- rag_wright/skills/claim_extraction/template.py +50 -0
- rag_wright/skills/compliance_judgment/SKILL.md +59 -0
- rag_wright/skills/corpus_ingest/SKILL.md +106 -0
- rag_wright/skills/extraction_semantic_judge/SKILL.md +51 -0
- rag_wright/skills/extraction_semantic_judge/__init__.py +1 -0
- rag_wright/skills/generation/SKILL.md +64 -0
- rag_wright/skills/generation/__init__.py +1 -0
- rag_wright/skills/generic_compliance_judgment/SKILL.md +58 -0
- rag_wright/skills/okf_navigate/SKILL.md +137 -0
- rag_wright/skills/requirement_extraction/SKILL.md +47 -0
- rag_wright/skills/requirement_extraction/__init__.py +1 -0
- rag_wright/skills/requirement_extraction/template.py +50 -0
- rag_wright/skills/rlm/SKILL.md +186 -0
- rag_wright/skills/rlm/__init__.py +31 -0
- rag_wright/skills/rlm/agent.py +292 -0
- rag_wright/skills/span_relevance_judgment/SKILL.md +67 -0
- rag_wright/skills/vision_to_text/SKILL.md +36 -0
- rag_wright/skills/vision_to_text/__init__.py +1 -0
- rag_wright/spans/__init__.py +1 -0
- rag_wright/spans/boundary.py +78 -0
- rag_wright/spans/clause_function_classifier.py +490 -0
- rag_wright/spans/clause_kg_extractor.py +337 -0
- rag_wright/spans/cuad_labels.py +81 -0
- rag_wright/spans/dim_classifier.py +158 -0
- rag_wright/spans/dim_fleet.json +411 -0
- rag_wright/spans/function_classifier.py +77 -0
- rag_wright/spans/function_families.py +62 -0
- rag_wright/spans/hybrid_classifier.py +103 -0
- rag_wright/spans/legalbert_classifier.py +83 -0
- rag_wright/spans/model_capabilities.py +107 -0
- rag_wright/spans/new_function_labels.py +111 -0
- rag_wright/spans/page_map.py +68 -0
- rag_wright/spans/property_extractor.py +365 -0
- rag_wright/spans/property_grounding.py +182 -0
- rag_wright/spans/reclassify.py +77 -0
- rag_wright/spans/scarce_function_labels.py +105 -0
- rag_wright/spans/segment.py +341 -0
- rag_wright/spans/semantic_judge.py +197 -0
- rag_wright/spans/symbolic_validation.py +131 -0
- rag_wright/spans/tag_clause_extractor.py +182 -0
- rag_wright/store/__init__.py +6 -0
- rag_wright/store/arcadedb.py +1135 -0
- rag_wright/store/chunk_text.py +66 -0
- rag_wright/store/seam.py +213 -0
- rag_wright/subgraphs/__init__.py +0 -0
- rag_wright/subgraphs/async_ingestion.py +204 -0
- rag_wright/subgraphs/compliance_check.py +1042 -0
- rag_wright/subgraphs/compliance_ingestion.py +306 -0
- rag_wright/subgraphs/contract_ingestion_pipeline.py +999 -0
- rag_wright/subgraphs/graph_extraction.py +102 -0
- rag_wright/subgraphs/intra_document_qa.py +328 -0
- rag_wright/subgraphs/observability.py +140 -0
- rag_wright/subgraphs/query_constraint_extraction.py +73 -0
- rag_wright/subgraphs/relational_qa.py +165 -0
- rag_wright/subgraphs/requirement_extraction.py +137 -0
- rag_wright/subgraphs/scaffold.py +65 -0
- rag_wright/subgraphs/semantic_chunking.py +183 -0
- rag_wright/subgraphs/typed_clause_extraction.py +172 -0
- rag_wright/subgraphs/typed_property_retrieval.py +278 -0
- rag_wright/util/__init__.py +1 -0
- rag_wright/util/concurrent.py +153 -0
- rag_wright/util/spacy_model.py +45 -0
- rag_wright-0.1.0.dist-info/METADATA +168 -0
- rag_wright-0.1.0.dist-info/RECORD +184 -0
- rag_wright-0.1.0.dist-info/WHEEL +4 -0
- rag_wright-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,331 @@
|
|
|
1
|
+
"""Per-model profiles and role resolution for the model seam (T11).
|
|
2
|
+
|
|
3
|
+
A profile is keyed by model id and carries only what the seam needs to make a *structured-output*
|
|
4
|
+
call safely: the structured-output method (default `function_calling`, more broadly supported across
|
|
5
|
+
open models than `json_schema`) and an optional structured-only `extra_body`. The `extra_body` is
|
|
6
|
+
applied by the seam only to the forced structured call (for example to disable thinking on the
|
|
7
|
+
structured emit), leaving free-text and reasoning calls unaffected. Provider/model flags live here
|
|
8
|
+
in config and in a dated ADR, never in a capability or node call site (CLAUDE.md standing rule).
|
|
9
|
+
|
|
10
|
+
Model priority for the structured-output-under-reasoning call class is a config decision, not
|
|
11
|
+
capability code: DeepSeek V4 Pro is the primary, Qwen 3.7 Plus the selectable secondary, and the
|
|
12
|
+
Gemma 4 class is the general / local-deployment default (SPEC section 4, tech stack; Phase 2 ledger).
|
|
13
|
+
The exact OpenRouter slug and the empirical structured profile (method + `extra_body`) for the
|
|
14
|
+
structured-reasoning model are confirmed against a live call at T12 and recorded in its ADR; each id
|
|
15
|
+
below is a documented default, overridable by env so the empirical slug needs no code change.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
import os
|
|
21
|
+
from enum import Enum
|
|
22
|
+
from typing import Any, Literal, Optional
|
|
23
|
+
|
|
24
|
+
from pydantic import BaseModel, ConfigDict, Field
|
|
25
|
+
|
|
26
|
+
StructuredMethod = Literal["function_calling", "json_mode", "json_schema"]
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
class ModelProfile(BaseModel):
|
|
30
|
+
"""How a given model id must be driven for a forced structured-output call."""
|
|
31
|
+
|
|
32
|
+
# `model_id` sits in Pydantic's protected `model_` namespace; opt out (it is a plain field).
|
|
33
|
+
model_config = ConfigDict(frozen=True, protected_namespaces=())
|
|
34
|
+
|
|
35
|
+
model_id: str
|
|
36
|
+
structured_method: StructuredMethod = "function_calling"
|
|
37
|
+
# Some serving stacks cannot do SERVER-SIDE grammar-constrained structured output at all (self-hosted
|
|
38
|
+
# Gemma 4 on vLLM 0.26: json_schema/json_mode run away to max_model_len; function_calling needs a buggy
|
|
39
|
+
# tool-parser -- while FREE-TEXT is perfect). Flag those models so generation uses the CLIENT-SIDE
|
|
40
|
+
# free-text + tag-parse path (answer_generator.answer_model_for) instead of the guided-decoding seam.
|
|
41
|
+
client_side_structured: bool = False
|
|
42
|
+
# Applied by the seam only to the forced structured call, never to the base client.
|
|
43
|
+
structured_extra_body: Optional[dict[str, Any]] = Field(default=None)
|
|
44
|
+
# The free-text counterpart to `structured_extra_body`: applied by the seam to the FREE-TEXT generation path
|
|
45
|
+
# (`astream_text` / the client-side tag-parse path), never to a forced structured call. A reasoning model must
|
|
46
|
+
# have its reasoning EXPLICITLY set on this path too -- leaving it unset falls to the provider default, which
|
|
47
|
+
# for qwen3.8 on the streaming path is inconsistent (intermittently returns reasoning-only / empty content).
|
|
48
|
+
# `structured_extra_body` does not reach here (it binds only to `with_structured_output`), so this is a separate
|
|
49
|
+
# slot; it also lets the free-text extraction use a DIFFERENT reasoning setting than the forced-structured judge.
|
|
50
|
+
text_extra_body: Optional[dict[str, Any]] = Field(default=None)
|
|
51
|
+
# Applied by the seam to the BASE client (every call to this model). Carries request-level provider
|
|
52
|
+
# routing (e.g. OpenRouter `{"provider": {"sort": "throughput"}}`) -- a provider flag, so it lives in
|
|
53
|
+
# config + a dated ADR, never in node/agent code (ADR-0027). Grounded: `extra_body` is a real
|
|
54
|
+
# `BaseChatOpenAI` field for exactly this purpose.
|
|
55
|
+
extra_body: Optional[dict[str, Any]] = Field(default=None)
|
|
56
|
+
|
|
57
|
+
# ADR-0100: the profile also carries HOW to REACH the model, so a model STRING fully describes both the
|
|
58
|
+
# model and its access -- the engine builds the client from this, and callers only need the string. A string
|
|
59
|
+
# can pin a backend (mix OpenRouter + self-hosted vLLM/Modal by using different strings), or leave `backend`
|
|
60
|
+
# None to fall back to the global `RAG_SERVING` default (back-compat).
|
|
61
|
+
backend: Optional[Literal["openrouter", "vllm", "ollama"]] = None # None -> RAG_SERVING default
|
|
62
|
+
# the id the backend actually expects (OpenRouter slug / vLLM `--served-model-name`), often != our string.
|
|
63
|
+
served_model_id: Optional[str] = None # default = model_id
|
|
64
|
+
base_url_env: Optional[str] = None # env var holding the base_url; default per backend (VLLM_BASE_URL, ...)
|
|
65
|
+
api_key_env: Optional[str] = None # env var holding the api key; default per backend (VLLM_API_KEY, ...)
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
class ModelRole(str, Enum):
|
|
69
|
+
"""Which model does which job. The mapping to ids lives in config, not in capability code."""
|
|
70
|
+
|
|
71
|
+
STRUCTURED_REASONING = "structured_reasoning" # forced schema under reasoning: extraction, grading, synthesis
|
|
72
|
+
STRUCTURED_REASONING_SECONDARY = "structured_reasoning_secondary" # same call class, selectable fallback
|
|
73
|
+
GENERAL = "general" # reasoning, generation, vision-to-text, RLM; the local-deployment default
|
|
74
|
+
SUMMARIZATION = "summarization" # a smaller model for chunking and summarization (FR-I.6 tiering)
|
|
75
|
+
OKF_ENRICHMENT = "okf_enrichment" # cheap classify + one-line description for OKF signposts (FR-K.2, ADR-0023)
|
|
76
|
+
# issue 0005: the ingest clause-function classifier as its OWN role, so it can run on a different model than
|
|
77
|
+
# GENERAL (e.g. Gemma-4) without moving the other stages. Route (b) is CLIENT-SIDE tag-parse -> works on any
|
|
78
|
+
# model. Defaults to the product LLM (unchanged behavior); set RAG_MODEL_FUNCTION_CLASSIFY to override.
|
|
79
|
+
FUNCTION_CLASSIFY = "function_classify"
|
|
80
|
+
# 0009-VLM: VLM-based OCR escalation for degraded scans (via docling ApiVlmOptions -> OpenRouter). Defaults to
|
|
81
|
+
# Gemma-4 (a vision model, unlike the Granite product LLM); set RAG_MODEL_VISION_OCR to swap the model.
|
|
82
|
+
VISION_OCR = "vision_ocr"
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
# Default model ids per role, confirmed against the live OpenRouter catalog at T12 (ADR-0006).
|
|
86
|
+
# Overridable by env (`_ROLE_ENV`), so a later slug change stays config, not code.
|
|
87
|
+
DEFAULT_STRUCTURED_REASONING = "deepseek/deepseek-v4-pro"
|
|
88
|
+
DEFAULT_STRUCTURED_REASONING_SECONDARY = "qwen/qwen3.7-plus"
|
|
89
|
+
DEFAULT_GENERAL = "google/gemma-4-31b-it"
|
|
90
|
+
DEFAULT_SUMMARIZATION = "deepseek/deepseek-v4-flash" # the smaller/faster DeepSeek (FR-I.6)
|
|
91
|
+
# OKF signpost enrichment is a simple classify-and-describe task; a cheap Gemma matched DeepSeek V4 Pro
|
|
92
|
+
# on it (100% category agreement, good one-liners, ~4x cheaper/faster) at the 2026-07-22 bench (ADR-0023).
|
|
93
|
+
# THIS TASK ONLY; every other call class stays on its DeepSeek/Gemma role above.
|
|
94
|
+
DEFAULT_OKF_ENRICHMENT = "google/gemma-4-26b-a4b-it"
|
|
95
|
+
|
|
96
|
+
# MS1-2 (ADR-0039): the product substrate is a SINGLE self-hosted model on the A100. EVERY role defaults to
|
|
97
|
+
# Granite -- Gemma/DeepSeek are DROPPED from the product default but stay REGISTERED in `PROFILES` below, so a
|
|
98
|
+
# dev run can still select any of them via the `RAG_MODEL_*` / `RAG_MODEL_ALL` env overrides. The DEFAULT_*
|
|
99
|
+
# constants above are kept as those foundation-model profile keys + documented dev-override values. (OKF
|
|
100
|
+
# signpost enrichment -- the one-time ADR-0023 Gemma exception -- is NOT wired into the ingestion/query
|
|
101
|
+
# pipeline (only `okf/enrich.py`), so it too defaults to Granite; ADR-0023's Gemma choice is now vestigial.)
|
|
102
|
+
_PRODUCT_LLM = "qwen3.8-27b-modal-or" # engine-wide default (ADR-0100): Qwen via OpenRouter for now
|
|
103
|
+
|
|
104
|
+
_ROLE_ENV: dict[ModelRole, tuple[str, str]] = {
|
|
105
|
+
ModelRole.STRUCTURED_REASONING: ("RAG_MODEL_STRUCTURED_REASONING", _PRODUCT_LLM),
|
|
106
|
+
ModelRole.STRUCTURED_REASONING_SECONDARY: ("RAG_MODEL_STRUCTURED_REASONING_SECONDARY", _PRODUCT_LLM),
|
|
107
|
+
ModelRole.GENERAL: ("RAG_MODEL_GENERAL", _PRODUCT_LLM),
|
|
108
|
+
ModelRole.SUMMARIZATION: ("RAG_MODEL_SUMMARIZATION", _PRODUCT_LLM),
|
|
109
|
+
ModelRole.OKF_ENRICHMENT: ("RAG_MODEL_OKF_ENRICHMENT", _PRODUCT_LLM),
|
|
110
|
+
ModelRole.FUNCTION_CLASSIFY: ("RAG_MODEL_FUNCTION_CLASSIFY", _PRODUCT_LLM),
|
|
111
|
+
# 0009-VLM: defaults to Gemma-4 (a vision model), NOT the Granite product LLM -- OCR needs vision.
|
|
112
|
+
ModelRole.VISION_OCR: ("RAG_MODEL_VISION_OCR", DEFAULT_GENERAL),
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
# Registered profiles keyed by model id. A model without an entry falls back to the safe default
|
|
116
|
+
# (function_calling, no extra_body) via `profile_for`, so no call site special-cases a model.
|
|
117
|
+
# Structured-output methods and extra bodies below are empirical, confirmed by a live forced-schema
|
|
118
|
+
# call at T12 and recorded in ADR-0006.
|
|
119
|
+
PROFILES: dict[str, ModelProfile] = {
|
|
120
|
+
# DeepSeek V4 Pro honors the forced tool call while reasoning; no thinking-disable needed. Route by
|
|
121
|
+
# THROUGHPUT so OpenRouter prefers the fastest provider over the cheapest (which throttled the bulk
|
|
122
|
+
# property extraction, T58); keeps V4 Pro, model rule intact (ADR-0027).
|
|
123
|
+
DEFAULT_STRUCTURED_REASONING: ModelProfile(
|
|
124
|
+
model_id=DEFAULT_STRUCTURED_REASONING,
|
|
125
|
+
extra_body={"provider": {"sort": "throughput"}},
|
|
126
|
+
),
|
|
127
|
+
# Qwen 3.7 Plus rejects `tool_choice` object/required in thinking mode ("<400> ... does not
|
|
128
|
+
# support being set to required or object in thinking mode"); disabling reasoning on the forced
|
|
129
|
+
# structured call alone fixes it, leaving its free-text/reasoning calls untouched (ADR-0006).
|
|
130
|
+
DEFAULT_STRUCTURED_REASONING_SECONDARY: ModelProfile(
|
|
131
|
+
model_id=DEFAULT_STRUCTURED_REASONING_SECONDARY,
|
|
132
|
+
structured_extra_body={"reasoning": {"enabled": False}},
|
|
133
|
+
),
|
|
134
|
+
# Gemma 4, like Qwen, rejects/returns-None on forced structured output when its reasoning mode is on for
|
|
135
|
+
# richer schemas (nullable fields): the CU-C1 NL->type emit returned None on every call until reasoning was
|
|
136
|
+
# disabled on the forced structured call. Same fix as the secondary (structured-only, so free-text/reasoning
|
|
137
|
+
# calls -- and the two-step reason node -- are untouched); simpler schemas (chunking _BoundaryList/_Summary)
|
|
138
|
+
# verified still valid with it. Empirical, dated: CU-D2 / ADR-0032.
|
|
139
|
+
DEFAULT_GENERAL: ModelProfile(
|
|
140
|
+
model_id=DEFAULT_GENERAL,
|
|
141
|
+
structured_extra_body={"reasoning": {"enabled": False}},
|
|
142
|
+
),
|
|
143
|
+
# DeepSeek V4 Flash does reasoning + structured output together, like V4 Pro (ADR-0006); no
|
|
144
|
+
# thinking-disable needed. As the T58 BULK property extractor (Flash->Pro cascade, ADR-0028) it routes
|
|
145
|
+
# by THROUGHPUT too, to dodge the cheapest-provider throttle (ADR-0027).
|
|
146
|
+
DEFAULT_SUMMARIZATION: ModelProfile(
|
|
147
|
+
model_id=DEFAULT_SUMMARIZATION,
|
|
148
|
+
extra_body={"provider": {"sort": "throughput"}},
|
|
149
|
+
),
|
|
150
|
+
# Gemma 4 26b-a4b takes the forced tool call cleanly (default function_calling, no extra_body);
|
|
151
|
+
# 0 structured-output errors across the 20-clause bench (ADR-0023).
|
|
152
|
+
DEFAULT_OKF_ENRICHMENT: ModelProfile(model_id=DEFAULT_OKF_ENRICHMENT),
|
|
153
|
+
# The ADOPTED default (2026-09-04): DeepSeek V4 Flash via OpenRouter's auto-updating `~...-latest` alias.
|
|
154
|
+
# It does reasoning + structured output together (function_calling, no thinking-disable needed) and returns
|
|
155
|
+
# content directly. The `~...-latest` alias otherwise routes to a SLOW/flaky provider (measured 6.9s vs 1.0s),
|
|
156
|
+
# so pin `provider.sort=latency` -- OpenRouter routes to the lowest-latency provider (applied to every seam
|
|
157
|
+
# call; the docling-graph extraction path injects the same routing separately in `dg_extraction`). Replaces the
|
|
158
|
+
# de-listed `ibm-granite/granite-4.1-8b` (OpenRouter 404: no endpoints); granite-4.2-8b needed reasoning
|
|
159
|
+
# disabled to avoid empty content, so DeepSeek Flash is the lower-friction default.
|
|
160
|
+
# Product default (ADR-0079). Granite-4.1-8b was de-listed on OpenRouter (404); granite-4.2-8b is its direct
|
|
161
|
+
# successor and the replacement default -- same family, same reasoning-off handling, and RELIABLE on the
|
|
162
|
+
# docling-graph extraction path (measured 5/5 clean clause extractions vs deepseek-v4-flash's flaky ~2/5,
|
|
163
|
+
# which produced "no models" a large fraction of the time regardless of provider). `reasoning:{enabled:false}`
|
|
164
|
+
# is LOAD-BEARING: granite-4.2 is a reasoning model and returns empty `content` on a forced structured call
|
|
165
|
+
# unless reasoning is disabled. `provider:{sort:latency}` picks the fastest of its (few) providers.
|
|
166
|
+
"ibm-granite/granite-4.2-8b": ModelProfile(
|
|
167
|
+
model_id="ibm-granite/granite-4.2-8b",
|
|
168
|
+
extra_body={"provider": {"sort": "latency"}, "reasoning": {"enabled": False}},
|
|
169
|
+
),
|
|
170
|
+
# Qwen 3.8 27b: UNLIKE qwen3.7-plus, it ACCEPTS a forced `tool_choice` object/required WHILE reasoning is ON
|
|
171
|
+
# (measured 2026-09-08), so we keep its deep reasoning on the single-shot forced-structured path (the compliance
|
|
172
|
+
# judge) -- that reasoning is the point of picking Qwen. `structured_extra_body={"reasoning":{"enabled":True}}`
|
|
173
|
+
# EXPLICITLY forces reasoning on the forced structured call: leaving it UNSET falls to the provider default,
|
|
174
|
+
# which for a forced tool call does little/no reasoning (measured ~5s + shallow vs ~32s deep) -- the opposite of
|
|
175
|
+
# the intended "reasoning-on, accept the latency" choice. Measured trade-offs on the judge: reasoning-ON ~32s
|
|
176
|
+
# (deep, chosen), reasoning-OFF ~5s (shallow).
|
|
177
|
+
# `text_extra_body={"reasoning":{"enabled":False}}` (issue 0020): the FREE-TEXT/tag-parse path (query constraint
|
|
178
|
+
# extraction) needs reasoning set EXPLICITLY too -- unset, qwen3.8 on the streaming path intermittently returns
|
|
179
|
+
# empty content (~1/3 of runs, measured), dropping the query's constraints. OFF (not ON) because constraint
|
|
180
|
+
# extraction is mechanical: OFF is deterministic + cheaper and drops the redundant raw-phrase `cap_quantum` that
|
|
181
|
+
# reasoning-ON adds. So the judge reasons deeply while query extraction does not -- two settings, two slots.
|
|
182
|
+
# provider PIN (ADR-0027, dated 2026-09-20): `{"provider":{"only":["deepinfra/bf16"],"allow_fallbacks":false}}`
|
|
183
|
+
# HARD-pins Qwen3.8-27B to DeepInfra's bf16 endpoint on OpenRouter -- no fallbacks, so OpenRouter can never
|
|
184
|
+
# route to a quantized (fp8/int) provider variant. Guarantees full-precision weights for consistent behavior
|
|
185
|
+
# (supersedes the earlier `sort:throughput`). Applies to all three OpenRouter Qwen3.8-27b strings below; the
|
|
186
|
+
# extraction surface honors the same pin via the profile (dg_extraction), so it holds engine-wide.
|
|
187
|
+
"qwen/qwen3.8-27b": ModelProfile(
|
|
188
|
+
model_id="qwen/qwen3.8-27b",
|
|
189
|
+
structured_extra_body={"reasoning": {"enabled": True}},
|
|
190
|
+
text_extra_body={"reasoning": {"enabled": False}},
|
|
191
|
+
extra_body={"provider": {"only": ["deepinfra/bf16"], "allow_fallbacks": False}},
|
|
192
|
+
),
|
|
193
|
+
# ADR-0100: backend-PINNED strings for the same Qwen3.8-27B -- pick the string, get the backend. `-or` routes
|
|
194
|
+
# to OpenRouter (slug qwen/qwen3.8-27b, its provider-routing + reasoning-field flags); `-modal` routes to the
|
|
195
|
+
# self-hosted vLLM server (served-name Qwen/Qwen3.8-27B, reasoning via vLLM's `chat_template_kwargs`, and NO
|
|
196
|
+
# OpenRouter `provider` routing). The product uses one string; mixing backends per stage is just two strings.
|
|
197
|
+
"qwen3.8-27b-or": ModelProfile(
|
|
198
|
+
model_id="qwen3.8-27b-or", backend="openrouter", served_model_id="qwen/qwen3.8-27b",
|
|
199
|
+
structured_extra_body={"reasoning": {"enabled": True}},
|
|
200
|
+
text_extra_body={"reasoning": {"enabled": False}},
|
|
201
|
+
extra_body={"provider": {"only": ["deepinfra/bf16"], "allow_fallbacks": False}},
|
|
202
|
+
),
|
|
203
|
+
"qwen3.8-27b-modal": ModelProfile(
|
|
204
|
+
model_id="qwen3.8-27b-modal", backend="vllm", served_model_id="Qwen/Qwen3.8-27B",
|
|
205
|
+
# vLLM controls Qwen3 reasoning via chat_template_kwargs (enable_thinking), not OpenRouter's reasoning
|
|
206
|
+
# field; base_url/key come from VLLM_BASE_URL/VLLM_API_KEY (override with base_url_env for a 2nd server).
|
|
207
|
+
structured_extra_body={"chat_template_kwargs": {"enable_thinking": True}},
|
|
208
|
+
text_extra_body={"chat_template_kwargs": {"enable_thinking": False}},
|
|
209
|
+
),
|
|
210
|
+
# Reasoning-OFF variant of `qwen3.8-27b-modal` for BULK closed-vocab classification (e.g. silver mining): the
|
|
211
|
+
# forced structured pick needs no chain-of-thought, and thinking-on is ~5-10x slower/costlier per call on the
|
|
212
|
+
# self-hosted A100. Same served name + function_calling (the server's hermes tool-parser handles it); only the
|
|
213
|
+
# structured call's `enable_thinking` flips to False. Not a default anywhere — opt in via the model string.
|
|
214
|
+
"qwen3.8-27b-modal-nothink": ModelProfile(
|
|
215
|
+
model_id="qwen3.8-27b-modal-nothink", backend="vllm", served_model_id="Qwen/Qwen3.8-27B",
|
|
216
|
+
structured_extra_body={"chat_template_kwargs": {"enable_thinking": False}},
|
|
217
|
+
text_extra_body={"chat_template_kwargs": {"enable_thinking": False}},
|
|
218
|
+
),
|
|
219
|
+
# Quantization ACCURACY EVAL profile (ADR-0110 follow-up): points the JUDGE role at a raw Modal vLLM endpoint
|
|
220
|
+
# (`scripts/modal_qwen3_vllm_server.py`, served as "qwen3-eval") so bf16-reference vs FP8 configs are the SAME
|
|
221
|
+
# profile with only VLLM_BASE_URL swapped between deployments. Reasoning ON for the structured judge (matches
|
|
222
|
+
# the production `-modal-or` judge), off for free-text. base_url/key from VLLM_BASE_URL/VLLM_API_KEY.
|
|
223
|
+
# Structured via vLLM NATIVE guided decoding (`json_schema`/xgrammar) -- raw vLLM 400s on the default
|
|
224
|
+
# `function_calling` ("tool_choice=function requires --tool-call-parser"), and open models reject a forced
|
|
225
|
+
# schema while THINKING, so thinking is OFF on the forced structured call (the self-hosted pattern, matching
|
|
226
|
+
# the Gemma profiles). Free-text stays thinking-off too. ref vs FP8 use the SAME config -> a clean delta.
|
|
227
|
+
"qwen3-eval": ModelProfile(
|
|
228
|
+
model_id="qwen3-eval", backend="vllm", served_model_id="qwen3-eval",
|
|
229
|
+
structured_method="json_schema",
|
|
230
|
+
structured_extra_body={"chat_template_kwargs": {"enable_thinking": False}},
|
|
231
|
+
text_extra_body={"chat_template_kwargs": {"enable_thinking": False}},
|
|
232
|
+
),
|
|
233
|
+
# THE ENGINE-WIDE DEFAULT (_PRODUCT_LLM / _PRODUCT_EXTRACT_DEFAULT / graph-extract default all point here).
|
|
234
|
+
# "-modal-or": the model destined for our Modal/vLLM deployment, but routed via OPENROUTER FOR NOW because the
|
|
235
|
+
# Modal cold-start/warmup is unsolved (snapshot path fails today). WHEN warmup is solved, flip THIS ONE entry's
|
|
236
|
+
# `backend` to "vllm" + `served_model_id="Qwen/Qwen3.8-27B"` + the vLLM reasoning flags, and every default
|
|
237
|
+
# follows -- no other change. Carries the OpenRouter Qwen flags (reasoning on judge / off free-text).
|
|
238
|
+
"qwen3.8-27b-modal-or": ModelProfile(
|
|
239
|
+
model_id="qwen3.8-27b-modal-or", backend="openrouter", served_model_id="qwen/qwen3.8-27b",
|
|
240
|
+
structured_extra_body={"reasoning": {"enabled": True}},
|
|
241
|
+
text_extra_body={"reasoning": {"enabled": False}},
|
|
242
|
+
extra_body={"provider": {"only": ["deepinfra/bf16"], "allow_fallbacks": False}},
|
|
243
|
+
),
|
|
244
|
+
# Kimi-k3 (Moonshot) shows the same `function_calling` degeneracy as granite (empty structured result on
|
|
245
|
+
# some queries); `json_schema` fixes it. Registered only for the KG-6 query-side model comparison (not
|
|
246
|
+
# adopted). Empirical, KG-6 / ADR-0034.
|
|
247
|
+
"moonshotai/kimi-k3": ModelProfile(
|
|
248
|
+
model_id="moonshotai/kimi-k3",
|
|
249
|
+
structured_method="json_schema",
|
|
250
|
+
),
|
|
251
|
+
# Self-hosted Gemma 4 QAT on vLLM: forced `function_calling` returns HTTP 400 ("tool_choice=function ...
|
|
252
|
+
# requires --tool-call-parser to be set") -- vLLM won't honor a forced named tool without a tool-call
|
|
253
|
+
# parser. `json_schema` routes through vLLM's NATIVE guided decoding (xgrammar), which needs no parser.
|
|
254
|
+
# Note this is a DIFFERENT id from the OpenRouter `google/gemma-4-31b-it` profile above (which uses the
|
|
255
|
+
# OpenRouter-only `{"reasoning": {"enabled": False}}` extra_body -- inapplicable to vLLM). Empirical, dated
|
|
256
|
+
# 2026-08-08: GATE-2 400 on function_calling, clean on json_schema. Candidate self-hosted GENERAL model.
|
|
257
|
+
# NOTE (2026-08-08, dated finding): self-hosted Gemma 4 on vLLM 0.26 could NOT do reliable GRAMMAR-
|
|
258
|
+
# CONSTRAINED structured output -- BOTH `json_schema` and `json_mode` (xgrammar guided decoding) RUN AWAY
|
|
259
|
+
# to max_model_len instead of emitting EOS, while FREE-TEXT generation terminates perfectly (1s, coherent,
|
|
260
|
+
# correct, finish=stop) and even emits inline [chunk_id] citations. function_calling needs
|
|
261
|
+
# --tool-call-parser gemma4 (concurrency <pad> bug, vllm#39392). So structured output on this stack is the
|
|
262
|
+
# open problem, NOT model quality/VRAM. Path forward: generate the answer as FREE-TEXT and extract the
|
|
263
|
+
# GeneratedAnswer (citations, abstained) DETERMINISTICALLY from it, bypassing guided decoding. `json_schema`
|
|
264
|
+
# kept here to match the Granite precedent; it does NOT yet work for this model on vLLM.
|
|
265
|
+
"google/gemma-4-31B-it-qat-w4a16-ct": ModelProfile(
|
|
266
|
+
model_id="google/gemma-4-31B-it-qat-w4a16-ct",
|
|
267
|
+
structured_method="json_schema", # unused: server-side guided decoding runs away on this stack
|
|
268
|
+
client_side_structured=True, # -> free-text + client-side tag parse instead
|
|
269
|
+
),
|
|
270
|
+
"google/gemma-4-26B-A4B-it": ModelProfile(
|
|
271
|
+
model_id="google/gemma-4-26B-A4B-it",
|
|
272
|
+
structured_method="json_schema",
|
|
273
|
+
client_side_structured=True,
|
|
274
|
+
),
|
|
275
|
+
}
|
|
276
|
+
|
|
277
|
+
|
|
278
|
+
def model_for(role: ModelRole) -> str:
|
|
279
|
+
"""Resolve a role to a model id. Precedence (all config, MS1-2): the ROLE-SPECIFIC override
|
|
280
|
+
(`RAG_MODEL_<ROLE>`) > the ALL-ROLES override (`RAG_MODEL_ALL`, to point every role at one model for a
|
|
281
|
+
quick cross-model test) > the documented default. So a run can swap one role, or every role, purely by env.
|
|
282
|
+
"""
|
|
283
|
+
env_var, default = _ROLE_ENV[role]
|
|
284
|
+
return os.getenv(env_var) or os.getenv("RAG_MODEL_ALL") or default
|
|
285
|
+
|
|
286
|
+
|
|
287
|
+
def profile_for(model_id: str) -> ModelProfile:
|
|
288
|
+
"""The registered profile for a model id, or a safe default profile for an unregistered one."""
|
|
289
|
+
return PROFILES.get(model_id, ModelProfile(model_id=model_id))
|
|
290
|
+
|
|
291
|
+
|
|
292
|
+
# --- ADR-0119: typed-DECISION model profiles (a separate API surface from the chat/structured LLM profiles) ---
|
|
293
|
+
|
|
294
|
+
class DecisionModelProfile(BaseModel):
|
|
295
|
+
"""How to reach a typed-DECISION model (TypeSafe Jev, or an on-prem Laya decisions server). This is a DIFFERENT
|
|
296
|
+
API surface from `ModelProfile`: a decision model POSTs a `state` + typed `questions` to a Decisions endpoint
|
|
297
|
+
and returns calibrated typed answers (`noul`/`choice`/`score`), not chat completions -- so it needs its own
|
|
298
|
+
profile shape (endpoint + served id + key env + default thresholds), not the structured-output fields.
|
|
299
|
+
|
|
300
|
+
Keyed by model id so swapping `jev-1.13` -> `jev-latest`, or Jev -> an on-prem Laya decisions endpoint, is a
|
|
301
|
+
config change, not a code edit (the Laya fallback, ADR-0119). The capability (`capabilities/jev_decision.py`)
|
|
302
|
+
reads this; the model id is never hardcoded in capability/node code (the model-neutrality rule)."""
|
|
303
|
+
|
|
304
|
+
model_config = ConfigDict(frozen=True, protected_namespaces=())
|
|
305
|
+
|
|
306
|
+
model_id: str
|
|
307
|
+
served_model_id: Optional[str] = None # the id the endpoint expects (default = model_id)
|
|
308
|
+
endpoint: str = "https://openrouter.ai/api/alpha/decisions"
|
|
309
|
+
api_key_env: str = "OPENROUTER_API_KEY"
|
|
310
|
+
timeout_s: float = 60.0
|
|
311
|
+
# default decision thresholds (callers may override): yes/no (noul) acceptance + multi-label noul cutoff.
|
|
312
|
+
op_threshold: float = 0.5
|
|
313
|
+
multilabel_threshold: float = 0.6
|
|
314
|
+
|
|
315
|
+
@property
|
|
316
|
+
def served(self) -> str:
|
|
317
|
+
return self.served_model_id or self.model_id
|
|
318
|
+
|
|
319
|
+
|
|
320
|
+
DECISION_PROFILES: dict[str, DecisionModelProfile] = {
|
|
321
|
+
"jev-1.13": DecisionModelProfile(model_id="jev-1.13", served_model_id="typesafe/jev-1.13"),
|
|
322
|
+
"jev-latest": DecisionModelProfile(model_id="jev-latest", served_model_id="typesafe/jev-latest"),
|
|
323
|
+
}
|
|
324
|
+
_DEFAULT_DECISION_MODEL = "jev-1.13"
|
|
325
|
+
|
|
326
|
+
|
|
327
|
+
def decision_profile(model_id: Optional[str] = None) -> DecisionModelProfile:
|
|
328
|
+
"""The decision-model profile for `model_id` (or `RAG_DECISION_MODEL`, else the built-in default `jev-1.13`).
|
|
329
|
+
An unregistered id gets a default profile using that id as its served id (so a raw slug still works)."""
|
|
330
|
+
mid = model_id or os.getenv("RAG_DECISION_MODEL") or _DEFAULT_DECISION_MODEL
|
|
331
|
+
return DECISION_PROFILES.get(mid, DecisionModelProfile(model_id=mid, served_model_id=mid))
|