rag-wright 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- rag_wright/__init__.py +13 -0
- rag_wright/api/__init__.py +33 -0
- rag_wright/api/config.py +59 -0
- rag_wright/api/discover.py +70 -0
- rag_wright/api/documents.py +39 -0
- rag_wright/api/ids.py +31 -0
- rag_wright/api/invoke.py +99 -0
- rag_wright/api/kg.py +61 -0
- rag_wright/api/mcp.py +94 -0
- rag_wright/api/usage.py +30 -0
- rag_wright/api/workspace.py +85 -0
- rag_wright/capabilities/__init__.py +8 -0
- rag_wright/capabilities/answer_generator.py +427 -0
- rag_wright/capabilities/ard.py +286 -0
- rag_wright/capabilities/assertion_extraction.py +79 -0
- rag_wright/capabilities/chunk_read.py +58 -0
- rag_wright/capabilities/chunk_write.py +163 -0
- rag_wright/capabilities/claim_extraction.py +153 -0
- rag_wright/capabilities/clause_exception_linking.py +117 -0
- rag_wright/capabilities/compliance_judgment.py +322 -0
- rag_wright/capabilities/compliance_store.py +87 -0
- rag_wright/capabilities/contract_kg_serve.py +156 -0
- rag_wright/capabilities/contract_kg_store.py +251 -0
- rag_wright/capabilities/dg_extraction.py +585 -0
- rag_wright/capabilities/disambiguation.py +163 -0
- rag_wright/capabilities/document_parse.py +87 -0
- rag_wright/capabilities/document_scope.py +49 -0
- rag_wright/capabilities/embedding.py +164 -0
- rag_wright/capabilities/embedding_profiles.py +43 -0
- rag_wright/capabilities/entity_resolution.py +154 -0
- rag_wright/capabilities/fusion.py +64 -0
- rag_wright/capabilities/graph_extraction.py +243 -0
- rag_wright/capabilities/graph_query.py +73 -0
- rag_wright/capabilities/graph_storage.py +111 -0
- rag_wright/capabilities/highlight_serve.py +142 -0
- rag_wright/capabilities/hybrid_search.py +65 -0
- rag_wright/capabilities/invoke.py +31 -0
- rag_wright/capabilities/jev_decision.py +38 -0
- rag_wright/capabilities/manifests.py +872 -0
- rag_wright/capabilities/okf_navigate.py +456 -0
- rag_wright/capabilities/parsing.py +286 -0
- rag_wright/capabilities/property_boosted_retrieval.py +125 -0
- rag_wright/capabilities/query_function_classifier.py +94 -0
- rag_wright/capabilities/query_understanding.py +109 -0
- rag_wright/capabilities/registry.py +262 -0
- rag_wright/capabilities/remote_encoders.py +94 -0
- rag_wright/capabilities/requirement_extraction.py +247 -0
- rag_wright/capabilities/reranking.py +123 -0
- rag_wright/capabilities/retrieval_core.py +126 -0
- rag_wright/capabilities/rlm_chunking.py +808 -0
- rag_wright/capabilities/rlm_synthesis.py +316 -0
- rag_wright/capabilities/scan_quality.py +136 -0
- rag_wright/capabilities/span_relevance_judgment.py +191 -0
- rag_wright/capabilities/vision_to_text.py +85 -0
- rag_wright/capabilities/vlm_ocr.py +85 -0
- rag_wright/contracts/__init__.py +6 -0
- rag_wright/contracts/chunk.py +79 -0
- rag_wright/contracts/compliance.py +303 -0
- rag_wright/contracts/contract_meta.py +27 -0
- rag_wright/contracts/extraction.py +130 -0
- rag_wright/contracts/function.py +167 -0
- rag_wright/contracts/function_routing.py +91 -0
- rag_wright/contracts/highlight.py +74 -0
- rag_wright/contracts/identifiers.py +153 -0
- rag_wright/contracts/jurisdiction.py +96 -0
- rag_wright/contracts/ontology.py +142 -0
- rag_wright/contracts/property.py +201 -0
- rag_wright/contracts/provenance.py +78 -0
- rag_wright/contracts/query_intent.py +53 -0
- rag_wright/contracts/span.py +76 -0
- rag_wright/contracts/value_match.py +84 -0
- rag_wright/corpus/__init__.py +0 -0
- rag_wright/corpus/canonicalize.py +116 -0
- rag_wright/corpus/cuad.py +153 -0
- rag_wright/corpus/cuad_ingestion.py +72 -0
- rag_wright/corpus/document_parser.py +299 -0
- rag_wright/corpus/edgar.py +231 -0
- rag_wright/corpus/gcs_ingestion.py +120 -0
- rag_wright/corpus/http.py +110 -0
- rag_wright/corpus/selection.py +152 -0
- rag_wright/mcp/__init__.py +11 -0
- rag_wright/mcp/compliance_server.py +299 -0
- rag_wright/mcp/intra_document_qa_server.py +170 -0
- rag_wright/mcp/relational_qa_server.py +171 -0
- rag_wright/mcp/session_store.py +64 -0
- rag_wright/mcp/typed_property_retrieval_server.py +191 -0
- rag_wright/models/__init__.py +8 -0
- rag_wright/models/profiles.py +331 -0
- rag_wright/models/seam.py +497 -0
- rag_wright/models/tag_structured.py +285 -0
- rag_wright/models/tracing.py +179 -0
- rag_wright/models/usage.py +102 -0
- rag_wright/okf/__init__.py +11 -0
- rag_wright/okf/compile.py +292 -0
- rag_wright/okf/document.py +47 -0
- rag_wright/okf/enrich.py +176 -0
- rag_wright/okf/links.py +190 -0
- rag_wright/okf/lint.py +105 -0
- rag_wright/ontology/__init__.py +6 -0
- rag_wright/ontology/_generated_template_meta.py +60 -0
- rag_wright/ontology/_generated_vocab.py +52 -0
- rag_wright/ontology/clause_template.py +964 -0
- rag_wright/ontology/codegen.py +84 -0
- rag_wright/ontology/compliance_bridge.ttl +186 -0
- rag_wright/ontology/contract_bridge.ttl +2685 -0
- rag_wright/ontology/contract_taxonomy.py +24 -0
- rag_wright/ontology/derive.py +58 -0
- rag_wright/ontology/loader.py +435 -0
- rag_wright/ontology/packs/ftc_16cfr255.ttl +29 -0
- rag_wright/ontology/registry.py +87 -0
- rag_wright/ontology/template_introspect.py +100 -0
- rag_wright/py.typed +0 -0
- rag_wright/reference/__init__.py +2 -0
- rag_wright/reference/compliance.py +41 -0
- rag_wright/reference/contract_seam.py +123 -0
- rag_wright/skills/__init__.py +7 -0
- rag_wright/skills/claim_extraction/SKILL.md +47 -0
- rag_wright/skills/claim_extraction/__init__.py +1 -0
- rag_wright/skills/claim_extraction/template.py +50 -0
- rag_wright/skills/compliance_judgment/SKILL.md +59 -0
- rag_wright/skills/corpus_ingest/SKILL.md +106 -0
- rag_wright/skills/extraction_semantic_judge/SKILL.md +51 -0
- rag_wright/skills/extraction_semantic_judge/__init__.py +1 -0
- rag_wright/skills/generation/SKILL.md +64 -0
- rag_wright/skills/generation/__init__.py +1 -0
- rag_wright/skills/generic_compliance_judgment/SKILL.md +58 -0
- rag_wright/skills/okf_navigate/SKILL.md +137 -0
- rag_wright/skills/requirement_extraction/SKILL.md +47 -0
- rag_wright/skills/requirement_extraction/__init__.py +1 -0
- rag_wright/skills/requirement_extraction/template.py +50 -0
- rag_wright/skills/rlm/SKILL.md +186 -0
- rag_wright/skills/rlm/__init__.py +31 -0
- rag_wright/skills/rlm/agent.py +292 -0
- rag_wright/skills/span_relevance_judgment/SKILL.md +67 -0
- rag_wright/skills/vision_to_text/SKILL.md +36 -0
- rag_wright/skills/vision_to_text/__init__.py +1 -0
- rag_wright/spans/__init__.py +1 -0
- rag_wright/spans/boundary.py +78 -0
- rag_wright/spans/clause_function_classifier.py +490 -0
- rag_wright/spans/clause_kg_extractor.py +337 -0
- rag_wright/spans/cuad_labels.py +81 -0
- rag_wright/spans/dim_classifier.py +158 -0
- rag_wright/spans/dim_fleet.json +411 -0
- rag_wright/spans/function_classifier.py +77 -0
- rag_wright/spans/function_families.py +62 -0
- rag_wright/spans/hybrid_classifier.py +103 -0
- rag_wright/spans/legalbert_classifier.py +83 -0
- rag_wright/spans/model_capabilities.py +107 -0
- rag_wright/spans/new_function_labels.py +111 -0
- rag_wright/spans/page_map.py +68 -0
- rag_wright/spans/property_extractor.py +365 -0
- rag_wright/spans/property_grounding.py +182 -0
- rag_wright/spans/reclassify.py +77 -0
- rag_wright/spans/scarce_function_labels.py +105 -0
- rag_wright/spans/segment.py +341 -0
- rag_wright/spans/semantic_judge.py +197 -0
- rag_wright/spans/symbolic_validation.py +131 -0
- rag_wright/spans/tag_clause_extractor.py +182 -0
- rag_wright/store/__init__.py +6 -0
- rag_wright/store/arcadedb.py +1135 -0
- rag_wright/store/chunk_text.py +66 -0
- rag_wright/store/seam.py +213 -0
- rag_wright/subgraphs/__init__.py +0 -0
- rag_wright/subgraphs/async_ingestion.py +204 -0
- rag_wright/subgraphs/compliance_check.py +1042 -0
- rag_wright/subgraphs/compliance_ingestion.py +306 -0
- rag_wright/subgraphs/contract_ingestion_pipeline.py +999 -0
- rag_wright/subgraphs/graph_extraction.py +102 -0
- rag_wright/subgraphs/intra_document_qa.py +328 -0
- rag_wright/subgraphs/observability.py +140 -0
- rag_wright/subgraphs/query_constraint_extraction.py +73 -0
- rag_wright/subgraphs/relational_qa.py +165 -0
- rag_wright/subgraphs/requirement_extraction.py +137 -0
- rag_wright/subgraphs/scaffold.py +65 -0
- rag_wright/subgraphs/semantic_chunking.py +183 -0
- rag_wright/subgraphs/typed_clause_extraction.py +172 -0
- rag_wright/subgraphs/typed_property_retrieval.py +278 -0
- rag_wright/util/__init__.py +1 -0
- rag_wright/util/concurrent.py +153 -0
- rag_wright/util/spacy_model.py +45 -0
- rag_wright-0.1.0.dist-info/METADATA +168 -0
- rag_wright-0.1.0.dist-info/RECORD +184 -0
- rag_wright-0.1.0.dist-info/WHEEL +4 -0
- rag_wright-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,427 @@
|
|
|
1
|
+
"""Answer generation (FR-C.9, FR-Q.6, T29): grounded, cited, confidence-aware, abstention-willing.
|
|
2
|
+
|
|
3
|
+
Produces the final answer from the query-side evidence (fusion/synthesis). It enforces the spec's hard
|
|
4
|
+
rule — **no claim without a citation** (FR-Q.6): every non-abstaining answer must cite `chunk_id`s that
|
|
5
|
+
are actually in the evidence, and a question the evidence does not support yields an **abstention**, not
|
|
6
|
+
a fabrication. It is **confidence-aware**: graph-derived facts carry their confidence tag into the
|
|
7
|
+
evidence the model sees (T26 surfaces it; here it is put in front of the generator). Vision-to-text is a
|
|
8
|
+
separate capability/slug (`vision_to_text.py`, ADR-0014), though both run on the Gemma 4 class model.
|
|
9
|
+
|
|
10
|
+
Grounding and citation are enforced in CODE around the model, not left to the prompt: fabricated
|
|
11
|
+
citations (ids not in the evidence) are dropped, and an answer that ends up with no valid citation is
|
|
12
|
+
coerced to an abstention. The model choice is the `GENERAL` role (no flag here).
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import re
|
|
18
|
+
from enum import Enum
|
|
19
|
+
from pathlib import Path
|
|
20
|
+
from typing import Optional, Protocol, runtime_checkable
|
|
21
|
+
|
|
22
|
+
from pydantic import BaseModel, model_validator
|
|
23
|
+
|
|
24
|
+
from rag_wright.capabilities.registry import CapabilityRegistry
|
|
25
|
+
from rag_wright.models.profiles import ModelRole, model_for
|
|
26
|
+
from rag_wright.models.seam import astream_text, build_model, build_structured
|
|
27
|
+
from rag_wright.util.concurrent import map_concurrent
|
|
28
|
+
|
|
29
|
+
_ABSTENTION = "The retrieved context does not support an answer."
|
|
30
|
+
_SKILL_PATH = Path(__file__).parents[1] / "skills" / "generation" / "SKILL.md"
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def generation_method() -> str:
|
|
34
|
+
"""The grounded-answer method (the `generation` SKILL body, YAML frontmatter stripped) used as the
|
|
35
|
+
generator's instruction. Authored knowledge (skills/generation/SKILL.md), not a hardcoded string. The
|
|
36
|
+
citation/abstention GUARANTEES are still enforced in code around the model (see `generate_answer`)."""
|
|
37
|
+
text = _SKILL_PATH.read_text(encoding="utf-8")
|
|
38
|
+
if text.startswith("---"):
|
|
39
|
+
marker = text.find("\n---", 3)
|
|
40
|
+
if marker != -1:
|
|
41
|
+
text = text[marker + 4 :]
|
|
42
|
+
return text.strip()
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
class EvidenceItem(BaseModel):
|
|
46
|
+
"""One piece of grounding evidence: a chunk's text, its `chunk_id` (the citation), and — for a
|
|
47
|
+
graph-derived fact — its confidence tag (surfaced to the generator, FR-S.4).
|
|
48
|
+
|
|
49
|
+
Engine issue 0011 / ADR-0064: a clause's typed properties ride OUT-OF-BAND here, NOT concatenated into
|
|
50
|
+
`text`. `_evidence_block` renders only `text`, so the generator never sees the `dimension=value` schema
|
|
51
|
+
tokens and cannot paraphrase them into prose ("the typed property cap_quantum=..."). The structured facts
|
|
52
|
+
stay available on this field for a caller that wants them (the product's UI chips); they are never fed to
|
|
53
|
+
the model. This is the ADR-0054 treatment (function label) applied to properties, but out-of-band rather
|
|
54
|
+
than dropped, because the properties do real work elsewhere."""
|
|
55
|
+
|
|
56
|
+
chunk_id: str
|
|
57
|
+
text: str
|
|
58
|
+
confidence: Optional[str] = None # graph-fact confidence tag; None for plain retrieved text
|
|
59
|
+
properties: Optional[list[dict]] = None # code-generated {dimension, value}; out-of-band, never in `text`
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
class AnswerKind(str, Enum):
|
|
63
|
+
"""The sufficiency of a generated answer (PREC-1a): a first-class signal so an honest hedge is distinct
|
|
64
|
+
from a confident over-answer, both in the contract the caller receives and in evaluation."""
|
|
65
|
+
|
|
66
|
+
ANSWERED = "answered" # the evidence supports the answer
|
|
67
|
+
PARTIAL = "partial" # answered, but the evidence does NOT fully support it -> caveated, low-confidence
|
|
68
|
+
ABSTAINED = "abstained" # the evidence supports no answer -> abstention (no fabrication)
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
class GeneratedAnswer(BaseModel):
|
|
72
|
+
"""The generated answer (FR-C.9): grounded text, the cited chunk_ids, whether it abstained, and its
|
|
73
|
+
sufficiency `answer_kind` (PREC-1a). `abstained` is kept (backward-compat) and `answer_kind` is kept in
|
|
74
|
+
sync: constructing with `abstained` alone derives the kind (ABSTAINED/ANSWERED); passing `answer_kind`
|
|
75
|
+
(e.g. PARTIAL) wins and sets `abstained` accordingly. So no existing `abstained=`-only caller changes."""
|
|
76
|
+
|
|
77
|
+
answer: str
|
|
78
|
+
citations: list[str] # chunk_ids actually in the evidence (no claim without a citation, FR-Q.6)
|
|
79
|
+
abstained: bool = False # kept for backward-compat; reconciled with answer_kind by the validator below
|
|
80
|
+
answer_kind: Optional[AnswerKind] = None # None at input -> derived from `abstained`; else it wins
|
|
81
|
+
|
|
82
|
+
@model_validator(mode="after")
|
|
83
|
+
def _sync_kind(self) -> "GeneratedAnswer":
|
|
84
|
+
if self.answer_kind is None:
|
|
85
|
+
self.answer_kind = AnswerKind.ABSTAINED if self.abstained else AnswerKind.ANSWERED
|
|
86
|
+
else:
|
|
87
|
+
self.abstained = self.answer_kind is AnswerKind.ABSTAINED
|
|
88
|
+
return self
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
@runtime_checkable
|
|
92
|
+
class AnswerModel(Protocol):
|
|
93
|
+
"""The generation seam: produce a `GeneratedAnswer` for a grounded prompt (structured output)."""
|
|
94
|
+
|
|
95
|
+
def generate(self, prompt: str) -> GeneratedAnswer: ...
|
|
96
|
+
async def agenerate(self, prompt: str) -> GeneratedAnswer: ... # ASYNC-C1 (ADR-0057): true-deadline twin
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
class SeamAnswerModel:
|
|
100
|
+
"""The real generator: structured output through the model-profile seam (GENERAL role). `temperature`
|
|
101
|
+
defaults to 0; the best-of-N strategy constructs one at temperature>0 to sample diverse completions."""
|
|
102
|
+
|
|
103
|
+
def __init__(
|
|
104
|
+
self, model_id: str | None = None, *, temperature: float = 0.0, max_tokens: int | None = None
|
|
105
|
+
) -> None:
|
|
106
|
+
self._model_id = model_id or model_for(ModelRole.GENERAL)
|
|
107
|
+
self._temperature = temperature
|
|
108
|
+
self._max_tokens = max_tokens
|
|
109
|
+
|
|
110
|
+
def generate(self, prompt: str) -> GeneratedAnswer:
|
|
111
|
+
return build_structured(
|
|
112
|
+
self._model_id, GeneratedAnswer, temperature=self._temperature, max_tokens=self._max_tokens
|
|
113
|
+
).invoke(prompt)
|
|
114
|
+
|
|
115
|
+
async def agenerate(self, prompt: str) -> GeneratedAnswer:
|
|
116
|
+
# ASYNC-C1 (ADR-0057): the structured seam's async path (build_structured's .ainvoke = true wall-clock
|
|
117
|
+
# deadline). Kept in step with .generate so a SeamAnswerModel injected into agenerate_answer works.
|
|
118
|
+
return await build_structured(
|
|
119
|
+
self._model_id, GeneratedAnswer, temperature=self._temperature, max_tokens=self._max_tokens
|
|
120
|
+
).ainvoke(prompt)
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
@runtime_checkable
|
|
124
|
+
class ReasonModel(Protocol):
|
|
125
|
+
"""The free-text reasoning seam (B): analyze the evidence in prose, no forced schema."""
|
|
126
|
+
|
|
127
|
+
def reason(self, prompt: str) -> str: ...
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
class SeamReasonModel:
|
|
131
|
+
"""The real reasoner: a plain free-text call through the seam (GENERAL role). Free-text avoids the
|
|
132
|
+
forced-structured/thinking-mode conflict that makes the one-shot structured generate flaky."""
|
|
133
|
+
|
|
134
|
+
def __init__(self, model_id: str | None = None) -> None:
|
|
135
|
+
self._model_id = model_id or model_for(ModelRole.GENERAL)
|
|
136
|
+
|
|
137
|
+
def reason(self, prompt: str) -> str:
|
|
138
|
+
return str(build_model(self._model_id).invoke(prompt).content)
|
|
139
|
+
|
|
140
|
+
async def areason(self, prompt: str) -> str:
|
|
141
|
+
# ASYNC-A3 (ADR-0057): free-text via astream (idle-drip detection + true wall-clock deadline).
|
|
142
|
+
return await astream_text(self._model_id, prompt)
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
# --- client-side structured output: free-text + light XML tags, parsed here (no server guided decoding) ------
|
|
146
|
+
#
|
|
147
|
+
# Why: on some serving stacks (self-hosted Gemma 4 on vLLM) SERVER-SIDE grammar-constrained structured output
|
|
148
|
+
# runs away to max_model_len, while plain FREE-TEXT terminates cleanly. So we ask the model to answer in light
|
|
149
|
+
# XML tags and parse them CLIENT-SIDE into GeneratedAnswer. Tags (not JSON) because the `answer` body is long
|
|
150
|
+
# legal prose full of quotes/brackets/newlines -- which is exactly what breaks JSON string escaping; a tagged
|
|
151
|
+
# body needs no escaping. Robust by design: a missing <citations> block falls back to the inline [chunk_id]s the
|
|
152
|
+
# model already emits, and missing tags fall back to treating the whole text as the answer -- so retries are rare
|
|
153
|
+
# and _finalize (drop non-evidence citations, coerce uncited -> abstain) still enforces the contract downstream.
|
|
154
|
+
|
|
155
|
+
_TAG_INSTRUCTIONS = (
|
|
156
|
+
"\n\nReturn your response using EXACTLY these tags:\n"
|
|
157
|
+
"<answer>\nYour grounded answer, citing each supporting evidence item inline as [chunk_id] (the bracketed "
|
|
158
|
+
"id shown for that item).\n</answer>\n"
|
|
159
|
+
"<citations>\nThe chunk_id of every evidence item you used, one per line; use only ids present in the "
|
|
160
|
+
"evidence above.\n</citations>\n"
|
|
161
|
+
"If the evidence does not support an answer at all, output exactly <abstain/> and nothing else. If the "
|
|
162
|
+
"evidence only PARTIALLY or TANGENTIALLY addresses the question -- it mentions related material but does "
|
|
163
|
+
"not actually state the answer -- give what the evidence does support with citations, add the marker "
|
|
164
|
+
"<partial/>, and say plainly what the evidence does not establish (do NOT present a tangential mention as a "
|
|
165
|
+
"confident answer)."
|
|
166
|
+
)
|
|
167
|
+
_ANSWER_RE = re.compile(r"<answer>(.*?)</answer>", re.DOTALL | re.IGNORECASE)
|
|
168
|
+
_CITE_BLOCK_RE = re.compile(r"<citations>(.*?)</citations>", re.DOTALL | re.IGNORECASE)
|
|
169
|
+
_ABSTAIN_RE = re.compile(r"<abstain\s*/?>", re.IGNORECASE)
|
|
170
|
+
_PARTIAL_RE = re.compile(r"<partial\s*/?>", re.IGNORECASE)
|
|
171
|
+
# an inline citation: [<contract_id>:<index>:<hex hash>]; contract_id is delimiter-safe (no brackets).
|
|
172
|
+
_INLINE_CITE_RE = re.compile(r"\[([^\[\]]+:\d+:[0-9a-fA-F]{8,})\]")
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
def parse_tagged_answer(text: str) -> GeneratedAnswer:
|
|
176
|
+
"""Parse a free-text tagged response into a GeneratedAnswer (Pydantic then validates the contract; the
|
|
177
|
+
capability's _finalize drops any citation not in the evidence). Tolerant: <answer> tag -> its body; else an
|
|
178
|
+
<abstain/> marker -> abstain; else the whole text is the answer. Citations come from the <citations> block
|
|
179
|
+
AND the inline [chunk_id]s in the answer body (deduped), so a missing block still yields citations."""
|
|
180
|
+
t = text.strip()
|
|
181
|
+
if not t:
|
|
182
|
+
return _abstain()
|
|
183
|
+
match = _ANSWER_RE.search(t)
|
|
184
|
+
if match:
|
|
185
|
+
answer = match.group(1).strip()
|
|
186
|
+
elif _ABSTAIN_RE.search(t):
|
|
187
|
+
return _abstain()
|
|
188
|
+
else:
|
|
189
|
+
answer = _PARTIAL_RE.sub("", t).strip() # no tags -> prose (with inline [chunk_id]s); drop any marker
|
|
190
|
+
if not answer:
|
|
191
|
+
return _abstain()
|
|
192
|
+
citations: list[str] = []
|
|
193
|
+
block = _CITE_BLOCK_RE.search(t)
|
|
194
|
+
if block:
|
|
195
|
+
citations = [c for c in re.split(r"[\s,]+", block.group(1).strip()) if c]
|
|
196
|
+
for cid in _INLINE_CITE_RE.findall(answer): # supplement with inline ids (dedup, order-preserving)
|
|
197
|
+
if cid not in citations:
|
|
198
|
+
citations.append(cid)
|
|
199
|
+
kind = AnswerKind.PARTIAL if _PARTIAL_RE.search(t) else AnswerKind.ANSWERED # a flagged partial/hedge
|
|
200
|
+
return GeneratedAnswer(answer=answer, citations=citations, answer_kind=kind)
|
|
201
|
+
|
|
202
|
+
|
|
203
|
+
class TaggedFreeTextAnswerModel:
|
|
204
|
+
"""Generation with NO server-side guided decoding: a plain free-text call (`build_model`, no
|
|
205
|
+
`response_format`) that the model answers in light XML tags, parsed client-side (`parse_tagged_answer`).
|
|
206
|
+
`max_tokens` is a generous safety cap only -- free-text terminates on its own."""
|
|
207
|
+
|
|
208
|
+
def __init__(
|
|
209
|
+
self, model_id: str | None = None, *, temperature: float = 0.0, max_tokens: int = 2048
|
|
210
|
+
) -> None:
|
|
211
|
+
self._model_id = model_id or model_for(ModelRole.GENERAL)
|
|
212
|
+
self._temperature = temperature
|
|
213
|
+
self._max_tokens = max_tokens
|
|
214
|
+
|
|
215
|
+
def generate(self, prompt: str) -> GeneratedAnswer:
|
|
216
|
+
text = build_model(
|
|
217
|
+
self._model_id, temperature=self._temperature, max_tokens=self._max_tokens
|
|
218
|
+
).invoke(prompt + _TAG_INSTRUCTIONS).content
|
|
219
|
+
return parse_tagged_answer(str(text))
|
|
220
|
+
|
|
221
|
+
async def agenerate(self, prompt: str) -> GeneratedAnswer:
|
|
222
|
+
# ASYNC-A3 (ADR-0057): stream the free-text answer (idle-drip detection + true deadline), then parse the
|
|
223
|
+
# light tags client-side -- same GeneratedAnswer the sync path yields.
|
|
224
|
+
text = await astream_text(self._model_id, prompt + _TAG_INSTRUCTIONS,
|
|
225
|
+
temperature=self._temperature, max_tokens=self._max_tokens)
|
|
226
|
+
return parse_tagged_answer(text)
|
|
227
|
+
|
|
228
|
+
|
|
229
|
+
def answer_model_for(
|
|
230
|
+
model_id: str | None = None, *, temperature: float = 0.0, max_tokens: int | None = None
|
|
231
|
+
) -> AnswerModel:
|
|
232
|
+
"""The generation strategy for a model: ALWAYS the free-text + client-side tag-parse path (ADR-0045).
|
|
233
|
+
Server-side guided decoding is not portable (runs away on self-hosted Gemma 4, ~60s/call on Cerebras), so
|
|
234
|
+
generation no longer depends on it for any model -- one LLM-agnostic path, so production and evals stay in
|
|
235
|
+
step. `SeamAnswerModel` remains for an explicit opt-in (constructed directly), but is never the default."""
|
|
236
|
+
mid = model_id or model_for(ModelRole.GENERAL)
|
|
237
|
+
return TaggedFreeTextAnswerModel(mid, temperature=temperature, max_tokens=max_tokens or 2048)
|
|
238
|
+
|
|
239
|
+
|
|
240
|
+
def _evidence_block(evidence: list[EvidenceItem]) -> str:
|
|
241
|
+
# engine issue 0002 (ADR-0055): NO inline [confidence: ...] marker. It used to sit in the evidence text,
|
|
242
|
+
# where the model narrated it to the reader (~100% conditional on citing an uncertain clause). Confidence is
|
|
243
|
+
# now delivered out-of-band as a hedging directive (see `_confidence_directive`); the evidence block is just
|
|
244
|
+
# the cited text.
|
|
245
|
+
return "\n".join(f"[{item.chunk_id}] {item.text}" for item in evidence)
|
|
246
|
+
|
|
247
|
+
|
|
248
|
+
# Confidence, OUT-OF-BAND (engine issue 0002 / ADR-0055). Instead of an inline [confidence: ...] marker the model
|
|
249
|
+
# can quote, the worst-case certainty across the evidence becomes a HEDGING DIRECTIVE the prompt consumes -- a
|
|
250
|
+
# tone instruction, appended after the evidence, never quotable. It does NOT name the internal enum tokens
|
|
251
|
+
# (INFERRED / AMBIGUOUS), so they cannot be echoed. This preserves the FR-S.4 / ADR-0028 hedging while removing
|
|
252
|
+
# the narratable surface -- the same move that closed the auto-tag leak (ADR-0054).
|
|
253
|
+
def _confidence_directive(evidence: list[EvidenceItem]) -> str:
|
|
254
|
+
confs = {(item.confidence or "").upper() for item in evidence}
|
|
255
|
+
if "AMBIGUOUS" in confs:
|
|
256
|
+
return ("\n\nCertainty note (do NOT mention this to the reader): some of the evidence is uncertain. "
|
|
257
|
+
"Where your answer depends on it, be tentative and do not state those points as settled. Let this "
|
|
258
|
+
"shape only how tentatively you write; never mention certainty, confidence, or any internal label.")
|
|
259
|
+
if "INFERRED" in confs:
|
|
260
|
+
return ("\n\nCertainty note (do NOT mention this to the reader): some of the evidence is inferred rather "
|
|
261
|
+
"than directly stated. Present any point that depends on it as an inference, not a settled fact. "
|
|
262
|
+
"Let this shape only how you phrase it; never mention certainty, confidence, or any internal label.")
|
|
263
|
+
return ""
|
|
264
|
+
|
|
265
|
+
|
|
266
|
+
def _abstain(text: str = _ABSTENTION) -> GeneratedAnswer:
|
|
267
|
+
return GeneratedAnswer(answer=text, citations=[], abstained=True)
|
|
268
|
+
|
|
269
|
+
|
|
270
|
+
# --- output hygiene: keep the engine's internal annotations out of user-facing prose (engine issue 0001) -----
|
|
271
|
+
#
|
|
272
|
+
# The evidence block feeds the model machine-internal markers -- inline citation ids [id:idx:hash], the
|
|
273
|
+
# [auto-tag: TYPE] classification (the engine's own sometimes-wrong guess), the [confidence: ...] tag, the
|
|
274
|
+
# [dimension=value; ...] typed-property string, and the [Exception ... (inferred)] carve-out framing. These are
|
|
275
|
+
# INPUTS to the model's judgement; a reader must never see them (a narrated auto-tag asserts a possibly-wrong
|
|
276
|
+
# clause type in the engine's voice, and a raw id looks broken). Citation ids belong in `citations` only. The
|
|
277
|
+
# SKILL now tells the model not to narrate them; this code is the hard guarantee for the bracketed forms it may
|
|
278
|
+
# still echo. TARGETED, not a blanket bracket strip: only the known annotation formats and the exact evidence
|
|
279
|
+
# chunk_ids are removed, so a legitimately quoted bracket (a defined term like "[Party A]") survives.
|
|
280
|
+
_CID_SHAPE = re.compile(r"^[^\[\]]+:\d+:[0-9a-fA-F]{8,}$")
|
|
281
|
+
_PROSE_ANNOTATION_RES = [
|
|
282
|
+
re.compile(r"\[[^\[\]]+:\d+:[0-9a-fA-F]{8,}\]"), # a bracketed citation id (incl. a fabricated one)
|
|
283
|
+
re.compile(r"\[auto-tag:[^\[\]]*\]", re.IGNORECASE),
|
|
284
|
+
re.compile(r"\[confidence:[^\[\]]*\]", re.IGNORECASE),
|
|
285
|
+
re.compile(r"\[Exception[^\[\]]*\]", re.IGNORECASE), # the inferred carve-out framing
|
|
286
|
+
re.compile(r"\[[^\[\]]*=[^\[\]]*\]"), # a typed-property fact group [dim=value; ...]
|
|
287
|
+
# engine issue 0002: a literal schema FIELD NAME written where a citation would go (not an id) -- engine
|
|
288
|
+
# vocabulary, never legitimate in a contract answer.
|
|
289
|
+
re.compile(r"\[(?:chunk_id|clause_id|source_doc_id|span_id|answer_kind)\]", re.IGNORECASE),
|
|
290
|
+
# engine issue 0026: the ONE non-bracketed marker. `<partial/>` is read (parse_tagged_answer) to set
|
|
291
|
+
# answer_kind, then must be stripped from the prose. The bracket-shape assumption above is exactly what let it
|
|
292
|
+
# reach the reader, so it lives here explicitly. Placed before the whitespace/punctuation passes so an inline
|
|
293
|
+
# marker's orphaned space/comma (issue 0022) is tidied after removal.
|
|
294
|
+
_PARTIAL_RE,
|
|
295
|
+
]
|
|
296
|
+
|
|
297
|
+
|
|
298
|
+
def _scrub_prose(text: str, evidence: list[EvidenceItem]) -> str:
|
|
299
|
+
"""Remove the engine's internal annotation tokens from user-facing answer prose (issue 0001): the exact
|
|
300
|
+
evidence chunk_ids (bracketed and, for citation-shaped ids, bare), then the known bracketed annotation
|
|
301
|
+
formats, then tidy the whitespace/punctuation the removals leave behind. Quoted clause text and any other
|
|
302
|
+
bracketed text are left intact -- only the known formats and the exact ids are stripped."""
|
|
303
|
+
out = text
|
|
304
|
+
for item in evidence: # the exact ids we know are in play (precise; avoids guessing)
|
|
305
|
+
out = re.sub(rf"\[\s*{re.escape(item.chunk_id)}\s*\]", "", out)
|
|
306
|
+
if _CID_SHAPE.match(item.chunk_id): # bare removal only for real citation-shaped ids (not short test ids)
|
|
307
|
+
out = out.replace(item.chunk_id, "")
|
|
308
|
+
for rx in _PROSE_ANNOTATION_RES:
|
|
309
|
+
out = rx.sub("", out)
|
|
310
|
+
out = re.sub(r"\(\s*\)", "", out) # empty parens left by a removed token
|
|
311
|
+
out = re.sub(r"[ \t]{2,}", " ", out) # collapse runs of spaces
|
|
312
|
+
out = re.sub(r"[ \t]+([,.;:)])", r"\1", out) # no space before punctuation
|
|
313
|
+
# issue 0022: removing the markers of a citation LIST orphans the separator that joined them (",." / ",;" /
|
|
314
|
+
# a run like ",," / a trailing ","). Drop separator(s) that now sit immediately before terminal punctuation or
|
|
315
|
+
# at newline/end. A separator followed by real content (a clause-separating "; the term ...") is untouched.
|
|
316
|
+
out = re.sub(r"(?:[ \t]*[,;])+([ \t]*[.;:)])", r"\1", out) # separator(s) before terminal punctuation -> drop
|
|
317
|
+
out = re.sub(r"(?:[ \t]*[,;])+(?=[ \t]*(?:\n|$))", "", out) # trailing separator(s) at newline / end -> drop
|
|
318
|
+
out = re.sub(r"\n[ \t]+", "\n", out)
|
|
319
|
+
return out.strip()
|
|
320
|
+
|
|
321
|
+
|
|
322
|
+
def _finalize(raw: GeneratedAnswer, evidence: list[EvidenceItem]) -> GeneratedAnswer:
|
|
323
|
+
"""The code-level guarantees applied to a raw model answer (shared by every generation strategy):
|
|
324
|
+
an abstention stays an abstention; a citation not present in the evidence is dropped (no fabrication);
|
|
325
|
+
an answer left with no valid citation is coerced to an abstention (no claim without a citation, FR-Q.6);
|
|
326
|
+
and the answer prose is scrubbed of internal annotations/ids (issue 0001) -- if that leaves no readable
|
|
327
|
+
prose, abstain rather than return an empty answer."""
|
|
328
|
+
if raw.abstained:
|
|
329
|
+
return _abstain(raw.answer or _ABSTENTION)
|
|
330
|
+
valid_ids = {item.chunk_id for item in evidence}
|
|
331
|
+
citations = [chunk_id for chunk_id in raw.citations if chunk_id in valid_ids] # drop fabricated
|
|
332
|
+
if not citations:
|
|
333
|
+
return _abstain() # no valid citation -> abstain (even a PARTIAL needs a citation, FR-Q.6)
|
|
334
|
+
answer = _scrub_prose(raw.answer, evidence) # keep internal annotations/ids out of the reader's prose
|
|
335
|
+
if not answer:
|
|
336
|
+
return _abstain() # the prose was nothing but annotations -> abstain
|
|
337
|
+
return GeneratedAnswer(answer=answer, citations=citations, answer_kind=raw.answer_kind)
|
|
338
|
+
|
|
339
|
+
|
|
340
|
+
def _answer_prompt(query: str, evidence: list[EvidenceItem]) -> str:
|
|
341
|
+
return (f"{generation_method()}\n\nQuestion: {query}\n\nEvidence:\n{_evidence_block(evidence)}"
|
|
342
|
+
f"{_confidence_directive(evidence)}")
|
|
343
|
+
|
|
344
|
+
|
|
345
|
+
def generate_answer(
|
|
346
|
+
query: str, evidence: list[EvidenceItem], *, model: AnswerModel
|
|
347
|
+
) -> GeneratedAnswer:
|
|
348
|
+
"""Generate a grounded, cited answer — or abstain — enforcing no-claim-without-a-citation in code.
|
|
349
|
+
|
|
350
|
+
Empty evidence abstains without a model call. Otherwise the model answers over the evidence block
|
|
351
|
+
(with confidence tags surfaced); any citation not present in the evidence is dropped, and an answer
|
|
352
|
+
left with no valid citation is coerced to an abstention. This is the single-call baseline strategy.
|
|
353
|
+
"""
|
|
354
|
+
if not evidence:
|
|
355
|
+
return _abstain()
|
|
356
|
+
return _finalize(model.generate(_answer_prompt(query, evidence)), evidence)
|
|
357
|
+
|
|
358
|
+
|
|
359
|
+
async def agenerate_answer(
|
|
360
|
+
query: str, evidence: list[EvidenceItem], *, model: AnswerModel
|
|
361
|
+
) -> GeneratedAnswer:
|
|
362
|
+
"""ASYNC-C1 (ADR-0057): the async twin of `generate_answer` -- the single-call baseline strategy on the
|
|
363
|
+
async generation seam (`model.agenerate`, a true wall-clock deadline on the model call). Identical
|
|
364
|
+
guarantees: empty evidence abstains WITHOUT a model call; any citation not in the evidence is dropped; an
|
|
365
|
+
answer left with no valid citation is coerced to an abstention (no claim without a citation, FR-Q.6)."""
|
|
366
|
+
if not evidence:
|
|
367
|
+
return _abstain()
|
|
368
|
+
return _finalize(await model.agenerate(_answer_prompt(query, evidence)), evidence)
|
|
369
|
+
|
|
370
|
+
|
|
371
|
+
_REASON_HEADER = (
|
|
372
|
+
"STEP 1 — ANALYSIS (not the final answer). Work through ONLY the evidence below: does it support an "
|
|
373
|
+
"answer to the question? Name the specific [chunk_id] items that support each part of a would-be answer; "
|
|
374
|
+
"if an item is framed as an inferred exception/carve-out, note the rule together with its exception. If "
|
|
375
|
+
"the evidence genuinely does not support an answer, say so and why. Do NOT write the final answer yet."
|
|
376
|
+
)
|
|
377
|
+
|
|
378
|
+
|
|
379
|
+
def generate_answer_reasoned(
|
|
380
|
+
query: str, evidence: list[EvidenceItem], *, reason_model: ReasonModel, emit_model: AnswerModel
|
|
381
|
+
) -> GeneratedAnswer:
|
|
382
|
+
"""Strategy B: split generation into a FREE-TEXT reasoning node then a STRUCTURED emit node. The reason
|
|
383
|
+
node analyzes the evidence in prose (where the model is strongest and the forced-structured/thinking-mode
|
|
384
|
+
conflict does not apply); the emit node only FORMATS that conclusion into the `GeneratedAnswer` contract,
|
|
385
|
+
a more constrained call than reason-and-emit in one shot. Same code-level guarantees via `_finalize`;
|
|
386
|
+
empty evidence still abstains without any model call."""
|
|
387
|
+
if not evidence:
|
|
388
|
+
return _abstain()
|
|
389
|
+
block = _evidence_block(evidence)
|
|
390
|
+
directive = _confidence_directive(evidence) # out-of-band hedging (ADR-0055), applied to both nodes
|
|
391
|
+
analysis = reason_model.reason(
|
|
392
|
+
f"{generation_method()}\n\n{_REASON_HEADER}\n\nQuestion: {query}\n\nEvidence:\n{block}{directive}")
|
|
393
|
+
emit_prompt = (
|
|
394
|
+
f"{generation_method()}\n\nQuestion: {query}\n\nEvidence:\n{block}{directive}\n\n"
|
|
395
|
+
f"STEP 2 — using your STEP 1 analysis below, emit the final grounded, cited answer now, or abstain "
|
|
396
|
+
f"if the analysis concluded the evidence does not support one.\n\nSTEP 1 analysis:\n{analysis}"
|
|
397
|
+
)
|
|
398
|
+
return _finalize(emit_model.generate(emit_prompt), evidence)
|
|
399
|
+
|
|
400
|
+
|
|
401
|
+
def generate_answer_best_of_n(
|
|
402
|
+
query: str, evidence: list[EvidenceItem], *, model: AnswerModel, n: int = 5, min_answers: int = 1,
|
|
403
|
+
max_concurrency: int = 5,
|
|
404
|
+
) -> GeneratedAnswer:
|
|
405
|
+
"""Strategy C: sample the single-call generation `n` times (supply a temperature>0 `model` for genuine
|
|
406
|
+
diversity), run CONCURRENTLY, and take the best-cited NON-abstaining sample — abstaining only if fewer
|
|
407
|
+
than `min_answers` samples produced a valid cited answer. Self-consistency against the near-boundary
|
|
408
|
+
abstain flip: one good grounded sample is enough to answer; `min_answers`>1 demands agreement. Empty
|
|
409
|
+
evidence abstains without any model call."""
|
|
410
|
+
if not evidence:
|
|
411
|
+
return _abstain()
|
|
412
|
+
prompt = _answer_prompt(query, evidence)
|
|
413
|
+
raws = map_concurrent([prompt] * n, model.generate, max_concurrency=max_concurrency)
|
|
414
|
+
answered = [f for f in (_finalize(r, evidence) for r in raws if r is not None) if not f.abstained]
|
|
415
|
+
if len(answered) < min_answers:
|
|
416
|
+
return _abstain()
|
|
417
|
+
return max(answered, key=lambda f: len(f.citations))
|
|
418
|
+
|
|
419
|
+
|
|
420
|
+
def register_generation(registry: CapabilityRegistry) -> None:
|
|
421
|
+
"""Register answer generation under FR-C.9 (`generation`; vision-to-text is its own slug, ADR-0014)."""
|
|
422
|
+
registry.register(
|
|
423
|
+
"generation",
|
|
424
|
+
contract=GeneratedAnswer,
|
|
425
|
+
kind="agent_skill", # a single grounded/cited LLM act (CAP-REG-1)
|
|
426
|
+
display_name="Answer generation (grounded, cited, abstains)",
|
|
427
|
+
)
|