rag-wright 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (184) hide show
  1. rag_wright/__init__.py +13 -0
  2. rag_wright/api/__init__.py +33 -0
  3. rag_wright/api/config.py +59 -0
  4. rag_wright/api/discover.py +70 -0
  5. rag_wright/api/documents.py +39 -0
  6. rag_wright/api/ids.py +31 -0
  7. rag_wright/api/invoke.py +99 -0
  8. rag_wright/api/kg.py +61 -0
  9. rag_wright/api/mcp.py +94 -0
  10. rag_wright/api/usage.py +30 -0
  11. rag_wright/api/workspace.py +85 -0
  12. rag_wright/capabilities/__init__.py +8 -0
  13. rag_wright/capabilities/answer_generator.py +427 -0
  14. rag_wright/capabilities/ard.py +286 -0
  15. rag_wright/capabilities/assertion_extraction.py +79 -0
  16. rag_wright/capabilities/chunk_read.py +58 -0
  17. rag_wright/capabilities/chunk_write.py +163 -0
  18. rag_wright/capabilities/claim_extraction.py +153 -0
  19. rag_wright/capabilities/clause_exception_linking.py +117 -0
  20. rag_wright/capabilities/compliance_judgment.py +322 -0
  21. rag_wright/capabilities/compliance_store.py +87 -0
  22. rag_wright/capabilities/contract_kg_serve.py +156 -0
  23. rag_wright/capabilities/contract_kg_store.py +251 -0
  24. rag_wright/capabilities/dg_extraction.py +585 -0
  25. rag_wright/capabilities/disambiguation.py +163 -0
  26. rag_wright/capabilities/document_parse.py +87 -0
  27. rag_wright/capabilities/document_scope.py +49 -0
  28. rag_wright/capabilities/embedding.py +164 -0
  29. rag_wright/capabilities/embedding_profiles.py +43 -0
  30. rag_wright/capabilities/entity_resolution.py +154 -0
  31. rag_wright/capabilities/fusion.py +64 -0
  32. rag_wright/capabilities/graph_extraction.py +243 -0
  33. rag_wright/capabilities/graph_query.py +73 -0
  34. rag_wright/capabilities/graph_storage.py +111 -0
  35. rag_wright/capabilities/highlight_serve.py +142 -0
  36. rag_wright/capabilities/hybrid_search.py +65 -0
  37. rag_wright/capabilities/invoke.py +31 -0
  38. rag_wright/capabilities/jev_decision.py +38 -0
  39. rag_wright/capabilities/manifests.py +872 -0
  40. rag_wright/capabilities/okf_navigate.py +456 -0
  41. rag_wright/capabilities/parsing.py +286 -0
  42. rag_wright/capabilities/property_boosted_retrieval.py +125 -0
  43. rag_wright/capabilities/query_function_classifier.py +94 -0
  44. rag_wright/capabilities/query_understanding.py +109 -0
  45. rag_wright/capabilities/registry.py +262 -0
  46. rag_wright/capabilities/remote_encoders.py +94 -0
  47. rag_wright/capabilities/requirement_extraction.py +247 -0
  48. rag_wright/capabilities/reranking.py +123 -0
  49. rag_wright/capabilities/retrieval_core.py +126 -0
  50. rag_wright/capabilities/rlm_chunking.py +808 -0
  51. rag_wright/capabilities/rlm_synthesis.py +316 -0
  52. rag_wright/capabilities/scan_quality.py +136 -0
  53. rag_wright/capabilities/span_relevance_judgment.py +191 -0
  54. rag_wright/capabilities/vision_to_text.py +85 -0
  55. rag_wright/capabilities/vlm_ocr.py +85 -0
  56. rag_wright/contracts/__init__.py +6 -0
  57. rag_wright/contracts/chunk.py +79 -0
  58. rag_wright/contracts/compliance.py +303 -0
  59. rag_wright/contracts/contract_meta.py +27 -0
  60. rag_wright/contracts/extraction.py +130 -0
  61. rag_wright/contracts/function.py +167 -0
  62. rag_wright/contracts/function_routing.py +91 -0
  63. rag_wright/contracts/highlight.py +74 -0
  64. rag_wright/contracts/identifiers.py +153 -0
  65. rag_wright/contracts/jurisdiction.py +96 -0
  66. rag_wright/contracts/ontology.py +142 -0
  67. rag_wright/contracts/property.py +201 -0
  68. rag_wright/contracts/provenance.py +78 -0
  69. rag_wright/contracts/query_intent.py +53 -0
  70. rag_wright/contracts/span.py +76 -0
  71. rag_wright/contracts/value_match.py +84 -0
  72. rag_wright/corpus/__init__.py +0 -0
  73. rag_wright/corpus/canonicalize.py +116 -0
  74. rag_wright/corpus/cuad.py +153 -0
  75. rag_wright/corpus/cuad_ingestion.py +72 -0
  76. rag_wright/corpus/document_parser.py +299 -0
  77. rag_wright/corpus/edgar.py +231 -0
  78. rag_wright/corpus/gcs_ingestion.py +120 -0
  79. rag_wright/corpus/http.py +110 -0
  80. rag_wright/corpus/selection.py +152 -0
  81. rag_wright/mcp/__init__.py +11 -0
  82. rag_wright/mcp/compliance_server.py +299 -0
  83. rag_wright/mcp/intra_document_qa_server.py +170 -0
  84. rag_wright/mcp/relational_qa_server.py +171 -0
  85. rag_wright/mcp/session_store.py +64 -0
  86. rag_wright/mcp/typed_property_retrieval_server.py +191 -0
  87. rag_wright/models/__init__.py +8 -0
  88. rag_wright/models/profiles.py +331 -0
  89. rag_wright/models/seam.py +497 -0
  90. rag_wright/models/tag_structured.py +285 -0
  91. rag_wright/models/tracing.py +179 -0
  92. rag_wright/models/usage.py +102 -0
  93. rag_wright/okf/__init__.py +11 -0
  94. rag_wright/okf/compile.py +292 -0
  95. rag_wright/okf/document.py +47 -0
  96. rag_wright/okf/enrich.py +176 -0
  97. rag_wright/okf/links.py +190 -0
  98. rag_wright/okf/lint.py +105 -0
  99. rag_wright/ontology/__init__.py +6 -0
  100. rag_wright/ontology/_generated_template_meta.py +60 -0
  101. rag_wright/ontology/_generated_vocab.py +52 -0
  102. rag_wright/ontology/clause_template.py +964 -0
  103. rag_wright/ontology/codegen.py +84 -0
  104. rag_wright/ontology/compliance_bridge.ttl +186 -0
  105. rag_wright/ontology/contract_bridge.ttl +2685 -0
  106. rag_wright/ontology/contract_taxonomy.py +24 -0
  107. rag_wright/ontology/derive.py +58 -0
  108. rag_wright/ontology/loader.py +435 -0
  109. rag_wright/ontology/packs/ftc_16cfr255.ttl +29 -0
  110. rag_wright/ontology/registry.py +87 -0
  111. rag_wright/ontology/template_introspect.py +100 -0
  112. rag_wright/py.typed +0 -0
  113. rag_wright/reference/__init__.py +2 -0
  114. rag_wright/reference/compliance.py +41 -0
  115. rag_wright/reference/contract_seam.py +123 -0
  116. rag_wright/skills/__init__.py +7 -0
  117. rag_wright/skills/claim_extraction/SKILL.md +47 -0
  118. rag_wright/skills/claim_extraction/__init__.py +1 -0
  119. rag_wright/skills/claim_extraction/template.py +50 -0
  120. rag_wright/skills/compliance_judgment/SKILL.md +59 -0
  121. rag_wright/skills/corpus_ingest/SKILL.md +106 -0
  122. rag_wright/skills/extraction_semantic_judge/SKILL.md +51 -0
  123. rag_wright/skills/extraction_semantic_judge/__init__.py +1 -0
  124. rag_wright/skills/generation/SKILL.md +64 -0
  125. rag_wright/skills/generation/__init__.py +1 -0
  126. rag_wright/skills/generic_compliance_judgment/SKILL.md +58 -0
  127. rag_wright/skills/okf_navigate/SKILL.md +137 -0
  128. rag_wright/skills/requirement_extraction/SKILL.md +47 -0
  129. rag_wright/skills/requirement_extraction/__init__.py +1 -0
  130. rag_wright/skills/requirement_extraction/template.py +50 -0
  131. rag_wright/skills/rlm/SKILL.md +186 -0
  132. rag_wright/skills/rlm/__init__.py +31 -0
  133. rag_wright/skills/rlm/agent.py +292 -0
  134. rag_wright/skills/span_relevance_judgment/SKILL.md +67 -0
  135. rag_wright/skills/vision_to_text/SKILL.md +36 -0
  136. rag_wright/skills/vision_to_text/__init__.py +1 -0
  137. rag_wright/spans/__init__.py +1 -0
  138. rag_wright/spans/boundary.py +78 -0
  139. rag_wright/spans/clause_function_classifier.py +490 -0
  140. rag_wright/spans/clause_kg_extractor.py +337 -0
  141. rag_wright/spans/cuad_labels.py +81 -0
  142. rag_wright/spans/dim_classifier.py +158 -0
  143. rag_wright/spans/dim_fleet.json +411 -0
  144. rag_wright/spans/function_classifier.py +77 -0
  145. rag_wright/spans/function_families.py +62 -0
  146. rag_wright/spans/hybrid_classifier.py +103 -0
  147. rag_wright/spans/legalbert_classifier.py +83 -0
  148. rag_wright/spans/model_capabilities.py +107 -0
  149. rag_wright/spans/new_function_labels.py +111 -0
  150. rag_wright/spans/page_map.py +68 -0
  151. rag_wright/spans/property_extractor.py +365 -0
  152. rag_wright/spans/property_grounding.py +182 -0
  153. rag_wright/spans/reclassify.py +77 -0
  154. rag_wright/spans/scarce_function_labels.py +105 -0
  155. rag_wright/spans/segment.py +341 -0
  156. rag_wright/spans/semantic_judge.py +197 -0
  157. rag_wright/spans/symbolic_validation.py +131 -0
  158. rag_wright/spans/tag_clause_extractor.py +182 -0
  159. rag_wright/store/__init__.py +6 -0
  160. rag_wright/store/arcadedb.py +1135 -0
  161. rag_wright/store/chunk_text.py +66 -0
  162. rag_wright/store/seam.py +213 -0
  163. rag_wright/subgraphs/__init__.py +0 -0
  164. rag_wright/subgraphs/async_ingestion.py +204 -0
  165. rag_wright/subgraphs/compliance_check.py +1042 -0
  166. rag_wright/subgraphs/compliance_ingestion.py +306 -0
  167. rag_wright/subgraphs/contract_ingestion_pipeline.py +999 -0
  168. rag_wright/subgraphs/graph_extraction.py +102 -0
  169. rag_wright/subgraphs/intra_document_qa.py +328 -0
  170. rag_wright/subgraphs/observability.py +140 -0
  171. rag_wright/subgraphs/query_constraint_extraction.py +73 -0
  172. rag_wright/subgraphs/relational_qa.py +165 -0
  173. rag_wright/subgraphs/requirement_extraction.py +137 -0
  174. rag_wright/subgraphs/scaffold.py +65 -0
  175. rag_wright/subgraphs/semantic_chunking.py +183 -0
  176. rag_wright/subgraphs/typed_clause_extraction.py +172 -0
  177. rag_wright/subgraphs/typed_property_retrieval.py +278 -0
  178. rag_wright/util/__init__.py +1 -0
  179. rag_wright/util/concurrent.py +153 -0
  180. rag_wright/util/spacy_model.py +45 -0
  181. rag_wright-0.1.0.dist-info/METADATA +168 -0
  182. rag_wright-0.1.0.dist-info/RECORD +184 -0
  183. rag_wright-0.1.0.dist-info/WHEEL +4 -0
  184. rag_wright-0.1.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,427 @@
1
+ """Answer generation (FR-C.9, FR-Q.6, T29): grounded, cited, confidence-aware, abstention-willing.
2
+
3
+ Produces the final answer from the query-side evidence (fusion/synthesis). It enforces the spec's hard
4
+ rule — **no claim without a citation** (FR-Q.6): every non-abstaining answer must cite `chunk_id`s that
5
+ are actually in the evidence, and a question the evidence does not support yields an **abstention**, not
6
+ a fabrication. It is **confidence-aware**: graph-derived facts carry their confidence tag into the
7
+ evidence the model sees (T26 surfaces it; here it is put in front of the generator). Vision-to-text is a
8
+ separate capability/slug (`vision_to_text.py`, ADR-0014), though both run on the Gemma 4 class model.
9
+
10
+ Grounding and citation are enforced in CODE around the model, not left to the prompt: fabricated
11
+ citations (ids not in the evidence) are dropped, and an answer that ends up with no valid citation is
12
+ coerced to an abstention. The model choice is the `GENERAL` role (no flag here).
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ import re
18
+ from enum import Enum
19
+ from pathlib import Path
20
+ from typing import Optional, Protocol, runtime_checkable
21
+
22
+ from pydantic import BaseModel, model_validator
23
+
24
+ from rag_wright.capabilities.registry import CapabilityRegistry
25
+ from rag_wright.models.profiles import ModelRole, model_for
26
+ from rag_wright.models.seam import astream_text, build_model, build_structured
27
+ from rag_wright.util.concurrent import map_concurrent
28
+
29
+ _ABSTENTION = "The retrieved context does not support an answer."
30
+ _SKILL_PATH = Path(__file__).parents[1] / "skills" / "generation" / "SKILL.md"
31
+
32
+
33
+ def generation_method() -> str:
34
+ """The grounded-answer method (the `generation` SKILL body, YAML frontmatter stripped) used as the
35
+ generator's instruction. Authored knowledge (skills/generation/SKILL.md), not a hardcoded string. The
36
+ citation/abstention GUARANTEES are still enforced in code around the model (see `generate_answer`)."""
37
+ text = _SKILL_PATH.read_text(encoding="utf-8")
38
+ if text.startswith("---"):
39
+ marker = text.find("\n---", 3)
40
+ if marker != -1:
41
+ text = text[marker + 4 :]
42
+ return text.strip()
43
+
44
+
45
+ class EvidenceItem(BaseModel):
46
+ """One piece of grounding evidence: a chunk's text, its `chunk_id` (the citation), and — for a
47
+ graph-derived fact — its confidence tag (surfaced to the generator, FR-S.4).
48
+
49
+ Engine issue 0011 / ADR-0064: a clause's typed properties ride OUT-OF-BAND here, NOT concatenated into
50
+ `text`. `_evidence_block` renders only `text`, so the generator never sees the `dimension=value` schema
51
+ tokens and cannot paraphrase them into prose ("the typed property cap_quantum=..."). The structured facts
52
+ stay available on this field for a caller that wants them (the product's UI chips); they are never fed to
53
+ the model. This is the ADR-0054 treatment (function label) applied to properties, but out-of-band rather
54
+ than dropped, because the properties do real work elsewhere."""
55
+
56
+ chunk_id: str
57
+ text: str
58
+ confidence: Optional[str] = None # graph-fact confidence tag; None for plain retrieved text
59
+ properties: Optional[list[dict]] = None # code-generated {dimension, value}; out-of-band, never in `text`
60
+
61
+
62
+ class AnswerKind(str, Enum):
63
+ """The sufficiency of a generated answer (PREC-1a): a first-class signal so an honest hedge is distinct
64
+ from a confident over-answer, both in the contract the caller receives and in evaluation."""
65
+
66
+ ANSWERED = "answered" # the evidence supports the answer
67
+ PARTIAL = "partial" # answered, but the evidence does NOT fully support it -> caveated, low-confidence
68
+ ABSTAINED = "abstained" # the evidence supports no answer -> abstention (no fabrication)
69
+
70
+
71
+ class GeneratedAnswer(BaseModel):
72
+ """The generated answer (FR-C.9): grounded text, the cited chunk_ids, whether it abstained, and its
73
+ sufficiency `answer_kind` (PREC-1a). `abstained` is kept (backward-compat) and `answer_kind` is kept in
74
+ sync: constructing with `abstained` alone derives the kind (ABSTAINED/ANSWERED); passing `answer_kind`
75
+ (e.g. PARTIAL) wins and sets `abstained` accordingly. So no existing `abstained=`-only caller changes."""
76
+
77
+ answer: str
78
+ citations: list[str] # chunk_ids actually in the evidence (no claim without a citation, FR-Q.6)
79
+ abstained: bool = False # kept for backward-compat; reconciled with answer_kind by the validator below
80
+ answer_kind: Optional[AnswerKind] = None # None at input -> derived from `abstained`; else it wins
81
+
82
+ @model_validator(mode="after")
83
+ def _sync_kind(self) -> "GeneratedAnswer":
84
+ if self.answer_kind is None:
85
+ self.answer_kind = AnswerKind.ABSTAINED if self.abstained else AnswerKind.ANSWERED
86
+ else:
87
+ self.abstained = self.answer_kind is AnswerKind.ABSTAINED
88
+ return self
89
+
90
+
91
+ @runtime_checkable
92
+ class AnswerModel(Protocol):
93
+ """The generation seam: produce a `GeneratedAnswer` for a grounded prompt (structured output)."""
94
+
95
+ def generate(self, prompt: str) -> GeneratedAnswer: ...
96
+ async def agenerate(self, prompt: str) -> GeneratedAnswer: ... # ASYNC-C1 (ADR-0057): true-deadline twin
97
+
98
+
99
+ class SeamAnswerModel:
100
+ """The real generator: structured output through the model-profile seam (GENERAL role). `temperature`
101
+ defaults to 0; the best-of-N strategy constructs one at temperature>0 to sample diverse completions."""
102
+
103
+ def __init__(
104
+ self, model_id: str | None = None, *, temperature: float = 0.0, max_tokens: int | None = None
105
+ ) -> None:
106
+ self._model_id = model_id or model_for(ModelRole.GENERAL)
107
+ self._temperature = temperature
108
+ self._max_tokens = max_tokens
109
+
110
+ def generate(self, prompt: str) -> GeneratedAnswer:
111
+ return build_structured(
112
+ self._model_id, GeneratedAnswer, temperature=self._temperature, max_tokens=self._max_tokens
113
+ ).invoke(prompt)
114
+
115
+ async def agenerate(self, prompt: str) -> GeneratedAnswer:
116
+ # ASYNC-C1 (ADR-0057): the structured seam's async path (build_structured's .ainvoke = true wall-clock
117
+ # deadline). Kept in step with .generate so a SeamAnswerModel injected into agenerate_answer works.
118
+ return await build_structured(
119
+ self._model_id, GeneratedAnswer, temperature=self._temperature, max_tokens=self._max_tokens
120
+ ).ainvoke(prompt)
121
+
122
+
123
+ @runtime_checkable
124
+ class ReasonModel(Protocol):
125
+ """The free-text reasoning seam (B): analyze the evidence in prose, no forced schema."""
126
+
127
+ def reason(self, prompt: str) -> str: ...
128
+
129
+
130
+ class SeamReasonModel:
131
+ """The real reasoner: a plain free-text call through the seam (GENERAL role). Free-text avoids the
132
+ forced-structured/thinking-mode conflict that makes the one-shot structured generate flaky."""
133
+
134
+ def __init__(self, model_id: str | None = None) -> None:
135
+ self._model_id = model_id or model_for(ModelRole.GENERAL)
136
+
137
+ def reason(self, prompt: str) -> str:
138
+ return str(build_model(self._model_id).invoke(prompt).content)
139
+
140
+ async def areason(self, prompt: str) -> str:
141
+ # ASYNC-A3 (ADR-0057): free-text via astream (idle-drip detection + true wall-clock deadline).
142
+ return await astream_text(self._model_id, prompt)
143
+
144
+
145
+ # --- client-side structured output: free-text + light XML tags, parsed here (no server guided decoding) ------
146
+ #
147
+ # Why: on some serving stacks (self-hosted Gemma 4 on vLLM) SERVER-SIDE grammar-constrained structured output
148
+ # runs away to max_model_len, while plain FREE-TEXT terminates cleanly. So we ask the model to answer in light
149
+ # XML tags and parse them CLIENT-SIDE into GeneratedAnswer. Tags (not JSON) because the `answer` body is long
150
+ # legal prose full of quotes/brackets/newlines -- which is exactly what breaks JSON string escaping; a tagged
151
+ # body needs no escaping. Robust by design: a missing <citations> block falls back to the inline [chunk_id]s the
152
+ # model already emits, and missing tags fall back to treating the whole text as the answer -- so retries are rare
153
+ # and _finalize (drop non-evidence citations, coerce uncited -> abstain) still enforces the contract downstream.
154
+
155
+ _TAG_INSTRUCTIONS = (
156
+ "\n\nReturn your response using EXACTLY these tags:\n"
157
+ "<answer>\nYour grounded answer, citing each supporting evidence item inline as [chunk_id] (the bracketed "
158
+ "id shown for that item).\n</answer>\n"
159
+ "<citations>\nThe chunk_id of every evidence item you used, one per line; use only ids present in the "
160
+ "evidence above.\n</citations>\n"
161
+ "If the evidence does not support an answer at all, output exactly <abstain/> and nothing else. If the "
162
+ "evidence only PARTIALLY or TANGENTIALLY addresses the question -- it mentions related material but does "
163
+ "not actually state the answer -- give what the evidence does support with citations, add the marker "
164
+ "<partial/>, and say plainly what the evidence does not establish (do NOT present a tangential mention as a "
165
+ "confident answer)."
166
+ )
167
+ _ANSWER_RE = re.compile(r"<answer>(.*?)</answer>", re.DOTALL | re.IGNORECASE)
168
+ _CITE_BLOCK_RE = re.compile(r"<citations>(.*?)</citations>", re.DOTALL | re.IGNORECASE)
169
+ _ABSTAIN_RE = re.compile(r"<abstain\s*/?>", re.IGNORECASE)
170
+ _PARTIAL_RE = re.compile(r"<partial\s*/?>", re.IGNORECASE)
171
+ # an inline citation: [<contract_id>:<index>:<hex hash>]; contract_id is delimiter-safe (no brackets).
172
+ _INLINE_CITE_RE = re.compile(r"\[([^\[\]]+:\d+:[0-9a-fA-F]{8,})\]")
173
+
174
+
175
+ def parse_tagged_answer(text: str) -> GeneratedAnswer:
176
+ """Parse a free-text tagged response into a GeneratedAnswer (Pydantic then validates the contract; the
177
+ capability's _finalize drops any citation not in the evidence). Tolerant: <answer> tag -> its body; else an
178
+ <abstain/> marker -> abstain; else the whole text is the answer. Citations come from the <citations> block
179
+ AND the inline [chunk_id]s in the answer body (deduped), so a missing block still yields citations."""
180
+ t = text.strip()
181
+ if not t:
182
+ return _abstain()
183
+ match = _ANSWER_RE.search(t)
184
+ if match:
185
+ answer = match.group(1).strip()
186
+ elif _ABSTAIN_RE.search(t):
187
+ return _abstain()
188
+ else:
189
+ answer = _PARTIAL_RE.sub("", t).strip() # no tags -> prose (with inline [chunk_id]s); drop any marker
190
+ if not answer:
191
+ return _abstain()
192
+ citations: list[str] = []
193
+ block = _CITE_BLOCK_RE.search(t)
194
+ if block:
195
+ citations = [c for c in re.split(r"[\s,]+", block.group(1).strip()) if c]
196
+ for cid in _INLINE_CITE_RE.findall(answer): # supplement with inline ids (dedup, order-preserving)
197
+ if cid not in citations:
198
+ citations.append(cid)
199
+ kind = AnswerKind.PARTIAL if _PARTIAL_RE.search(t) else AnswerKind.ANSWERED # a flagged partial/hedge
200
+ return GeneratedAnswer(answer=answer, citations=citations, answer_kind=kind)
201
+
202
+
203
+ class TaggedFreeTextAnswerModel:
204
+ """Generation with NO server-side guided decoding: a plain free-text call (`build_model`, no
205
+ `response_format`) that the model answers in light XML tags, parsed client-side (`parse_tagged_answer`).
206
+ `max_tokens` is a generous safety cap only -- free-text terminates on its own."""
207
+
208
+ def __init__(
209
+ self, model_id: str | None = None, *, temperature: float = 0.0, max_tokens: int = 2048
210
+ ) -> None:
211
+ self._model_id = model_id or model_for(ModelRole.GENERAL)
212
+ self._temperature = temperature
213
+ self._max_tokens = max_tokens
214
+
215
+ def generate(self, prompt: str) -> GeneratedAnswer:
216
+ text = build_model(
217
+ self._model_id, temperature=self._temperature, max_tokens=self._max_tokens
218
+ ).invoke(prompt + _TAG_INSTRUCTIONS).content
219
+ return parse_tagged_answer(str(text))
220
+
221
+ async def agenerate(self, prompt: str) -> GeneratedAnswer:
222
+ # ASYNC-A3 (ADR-0057): stream the free-text answer (idle-drip detection + true deadline), then parse the
223
+ # light tags client-side -- same GeneratedAnswer the sync path yields.
224
+ text = await astream_text(self._model_id, prompt + _TAG_INSTRUCTIONS,
225
+ temperature=self._temperature, max_tokens=self._max_tokens)
226
+ return parse_tagged_answer(text)
227
+
228
+
229
+ def answer_model_for(
230
+ model_id: str | None = None, *, temperature: float = 0.0, max_tokens: int | None = None
231
+ ) -> AnswerModel:
232
+ """The generation strategy for a model: ALWAYS the free-text + client-side tag-parse path (ADR-0045).
233
+ Server-side guided decoding is not portable (runs away on self-hosted Gemma 4, ~60s/call on Cerebras), so
234
+ generation no longer depends on it for any model -- one LLM-agnostic path, so production and evals stay in
235
+ step. `SeamAnswerModel` remains for an explicit opt-in (constructed directly), but is never the default."""
236
+ mid = model_id or model_for(ModelRole.GENERAL)
237
+ return TaggedFreeTextAnswerModel(mid, temperature=temperature, max_tokens=max_tokens or 2048)
238
+
239
+
240
+ def _evidence_block(evidence: list[EvidenceItem]) -> str:
241
+ # engine issue 0002 (ADR-0055): NO inline [confidence: ...] marker. It used to sit in the evidence text,
242
+ # where the model narrated it to the reader (~100% conditional on citing an uncertain clause). Confidence is
243
+ # now delivered out-of-band as a hedging directive (see `_confidence_directive`); the evidence block is just
244
+ # the cited text.
245
+ return "\n".join(f"[{item.chunk_id}] {item.text}" for item in evidence)
246
+
247
+
248
+ # Confidence, OUT-OF-BAND (engine issue 0002 / ADR-0055). Instead of an inline [confidence: ...] marker the model
249
+ # can quote, the worst-case certainty across the evidence becomes a HEDGING DIRECTIVE the prompt consumes -- a
250
+ # tone instruction, appended after the evidence, never quotable. It does NOT name the internal enum tokens
251
+ # (INFERRED / AMBIGUOUS), so they cannot be echoed. This preserves the FR-S.4 / ADR-0028 hedging while removing
252
+ # the narratable surface -- the same move that closed the auto-tag leak (ADR-0054).
253
+ def _confidence_directive(evidence: list[EvidenceItem]) -> str:
254
+ confs = {(item.confidence or "").upper() for item in evidence}
255
+ if "AMBIGUOUS" in confs:
256
+ return ("\n\nCertainty note (do NOT mention this to the reader): some of the evidence is uncertain. "
257
+ "Where your answer depends on it, be tentative and do not state those points as settled. Let this "
258
+ "shape only how tentatively you write; never mention certainty, confidence, or any internal label.")
259
+ if "INFERRED" in confs:
260
+ return ("\n\nCertainty note (do NOT mention this to the reader): some of the evidence is inferred rather "
261
+ "than directly stated. Present any point that depends on it as an inference, not a settled fact. "
262
+ "Let this shape only how you phrase it; never mention certainty, confidence, or any internal label.")
263
+ return ""
264
+
265
+
266
+ def _abstain(text: str = _ABSTENTION) -> GeneratedAnswer:
267
+ return GeneratedAnswer(answer=text, citations=[], abstained=True)
268
+
269
+
270
+ # --- output hygiene: keep the engine's internal annotations out of user-facing prose (engine issue 0001) -----
271
+ #
272
+ # The evidence block feeds the model machine-internal markers -- inline citation ids [id:idx:hash], the
273
+ # [auto-tag: TYPE] classification (the engine's own sometimes-wrong guess), the [confidence: ...] tag, the
274
+ # [dimension=value; ...] typed-property string, and the [Exception ... (inferred)] carve-out framing. These are
275
+ # INPUTS to the model's judgement; a reader must never see them (a narrated auto-tag asserts a possibly-wrong
276
+ # clause type in the engine's voice, and a raw id looks broken). Citation ids belong in `citations` only. The
277
+ # SKILL now tells the model not to narrate them; this code is the hard guarantee for the bracketed forms it may
278
+ # still echo. TARGETED, not a blanket bracket strip: only the known annotation formats and the exact evidence
279
+ # chunk_ids are removed, so a legitimately quoted bracket (a defined term like "[Party A]") survives.
280
+ _CID_SHAPE = re.compile(r"^[^\[\]]+:\d+:[0-9a-fA-F]{8,}$")
281
+ _PROSE_ANNOTATION_RES = [
282
+ re.compile(r"\[[^\[\]]+:\d+:[0-9a-fA-F]{8,}\]"), # a bracketed citation id (incl. a fabricated one)
283
+ re.compile(r"\[auto-tag:[^\[\]]*\]", re.IGNORECASE),
284
+ re.compile(r"\[confidence:[^\[\]]*\]", re.IGNORECASE),
285
+ re.compile(r"\[Exception[^\[\]]*\]", re.IGNORECASE), # the inferred carve-out framing
286
+ re.compile(r"\[[^\[\]]*=[^\[\]]*\]"), # a typed-property fact group [dim=value; ...]
287
+ # engine issue 0002: a literal schema FIELD NAME written where a citation would go (not an id) -- engine
288
+ # vocabulary, never legitimate in a contract answer.
289
+ re.compile(r"\[(?:chunk_id|clause_id|source_doc_id|span_id|answer_kind)\]", re.IGNORECASE),
290
+ # engine issue 0026: the ONE non-bracketed marker. `<partial/>` is read (parse_tagged_answer) to set
291
+ # answer_kind, then must be stripped from the prose. The bracket-shape assumption above is exactly what let it
292
+ # reach the reader, so it lives here explicitly. Placed before the whitespace/punctuation passes so an inline
293
+ # marker's orphaned space/comma (issue 0022) is tidied after removal.
294
+ _PARTIAL_RE,
295
+ ]
296
+
297
+
298
+ def _scrub_prose(text: str, evidence: list[EvidenceItem]) -> str:
299
+ """Remove the engine's internal annotation tokens from user-facing answer prose (issue 0001): the exact
300
+ evidence chunk_ids (bracketed and, for citation-shaped ids, bare), then the known bracketed annotation
301
+ formats, then tidy the whitespace/punctuation the removals leave behind. Quoted clause text and any other
302
+ bracketed text are left intact -- only the known formats and the exact ids are stripped."""
303
+ out = text
304
+ for item in evidence: # the exact ids we know are in play (precise; avoids guessing)
305
+ out = re.sub(rf"\[\s*{re.escape(item.chunk_id)}\s*\]", "", out)
306
+ if _CID_SHAPE.match(item.chunk_id): # bare removal only for real citation-shaped ids (not short test ids)
307
+ out = out.replace(item.chunk_id, "")
308
+ for rx in _PROSE_ANNOTATION_RES:
309
+ out = rx.sub("", out)
310
+ out = re.sub(r"\(\s*\)", "", out) # empty parens left by a removed token
311
+ out = re.sub(r"[ \t]{2,}", " ", out) # collapse runs of spaces
312
+ out = re.sub(r"[ \t]+([,.;:)])", r"\1", out) # no space before punctuation
313
+ # issue 0022: removing the markers of a citation LIST orphans the separator that joined them (",." / ",;" /
314
+ # a run like ",," / a trailing ","). Drop separator(s) that now sit immediately before terminal punctuation or
315
+ # at newline/end. A separator followed by real content (a clause-separating "; the term ...") is untouched.
316
+ out = re.sub(r"(?:[ \t]*[,;])+([ \t]*[.;:)])", r"\1", out) # separator(s) before terminal punctuation -> drop
317
+ out = re.sub(r"(?:[ \t]*[,;])+(?=[ \t]*(?:\n|$))", "", out) # trailing separator(s) at newline / end -> drop
318
+ out = re.sub(r"\n[ \t]+", "\n", out)
319
+ return out.strip()
320
+
321
+
322
+ def _finalize(raw: GeneratedAnswer, evidence: list[EvidenceItem]) -> GeneratedAnswer:
323
+ """The code-level guarantees applied to a raw model answer (shared by every generation strategy):
324
+ an abstention stays an abstention; a citation not present in the evidence is dropped (no fabrication);
325
+ an answer left with no valid citation is coerced to an abstention (no claim without a citation, FR-Q.6);
326
+ and the answer prose is scrubbed of internal annotations/ids (issue 0001) -- if that leaves no readable
327
+ prose, abstain rather than return an empty answer."""
328
+ if raw.abstained:
329
+ return _abstain(raw.answer or _ABSTENTION)
330
+ valid_ids = {item.chunk_id for item in evidence}
331
+ citations = [chunk_id for chunk_id in raw.citations if chunk_id in valid_ids] # drop fabricated
332
+ if not citations:
333
+ return _abstain() # no valid citation -> abstain (even a PARTIAL needs a citation, FR-Q.6)
334
+ answer = _scrub_prose(raw.answer, evidence) # keep internal annotations/ids out of the reader's prose
335
+ if not answer:
336
+ return _abstain() # the prose was nothing but annotations -> abstain
337
+ return GeneratedAnswer(answer=answer, citations=citations, answer_kind=raw.answer_kind)
338
+
339
+
340
+ def _answer_prompt(query: str, evidence: list[EvidenceItem]) -> str:
341
+ return (f"{generation_method()}\n\nQuestion: {query}\n\nEvidence:\n{_evidence_block(evidence)}"
342
+ f"{_confidence_directive(evidence)}")
343
+
344
+
345
+ def generate_answer(
346
+ query: str, evidence: list[EvidenceItem], *, model: AnswerModel
347
+ ) -> GeneratedAnswer:
348
+ """Generate a grounded, cited answer — or abstain — enforcing no-claim-without-a-citation in code.
349
+
350
+ Empty evidence abstains without a model call. Otherwise the model answers over the evidence block
351
+ (with confidence tags surfaced); any citation not present in the evidence is dropped, and an answer
352
+ left with no valid citation is coerced to an abstention. This is the single-call baseline strategy.
353
+ """
354
+ if not evidence:
355
+ return _abstain()
356
+ return _finalize(model.generate(_answer_prompt(query, evidence)), evidence)
357
+
358
+
359
+ async def agenerate_answer(
360
+ query: str, evidence: list[EvidenceItem], *, model: AnswerModel
361
+ ) -> GeneratedAnswer:
362
+ """ASYNC-C1 (ADR-0057): the async twin of `generate_answer` -- the single-call baseline strategy on the
363
+ async generation seam (`model.agenerate`, a true wall-clock deadline on the model call). Identical
364
+ guarantees: empty evidence abstains WITHOUT a model call; any citation not in the evidence is dropped; an
365
+ answer left with no valid citation is coerced to an abstention (no claim without a citation, FR-Q.6)."""
366
+ if not evidence:
367
+ return _abstain()
368
+ return _finalize(await model.agenerate(_answer_prompt(query, evidence)), evidence)
369
+
370
+
371
+ _REASON_HEADER = (
372
+ "STEP 1 — ANALYSIS (not the final answer). Work through ONLY the evidence below: does it support an "
373
+ "answer to the question? Name the specific [chunk_id] items that support each part of a would-be answer; "
374
+ "if an item is framed as an inferred exception/carve-out, note the rule together with its exception. If "
375
+ "the evidence genuinely does not support an answer, say so and why. Do NOT write the final answer yet."
376
+ )
377
+
378
+
379
+ def generate_answer_reasoned(
380
+ query: str, evidence: list[EvidenceItem], *, reason_model: ReasonModel, emit_model: AnswerModel
381
+ ) -> GeneratedAnswer:
382
+ """Strategy B: split generation into a FREE-TEXT reasoning node then a STRUCTURED emit node. The reason
383
+ node analyzes the evidence in prose (where the model is strongest and the forced-structured/thinking-mode
384
+ conflict does not apply); the emit node only FORMATS that conclusion into the `GeneratedAnswer` contract,
385
+ a more constrained call than reason-and-emit in one shot. Same code-level guarantees via `_finalize`;
386
+ empty evidence still abstains without any model call."""
387
+ if not evidence:
388
+ return _abstain()
389
+ block = _evidence_block(evidence)
390
+ directive = _confidence_directive(evidence) # out-of-band hedging (ADR-0055), applied to both nodes
391
+ analysis = reason_model.reason(
392
+ f"{generation_method()}\n\n{_REASON_HEADER}\n\nQuestion: {query}\n\nEvidence:\n{block}{directive}")
393
+ emit_prompt = (
394
+ f"{generation_method()}\n\nQuestion: {query}\n\nEvidence:\n{block}{directive}\n\n"
395
+ f"STEP 2 — using your STEP 1 analysis below, emit the final grounded, cited answer now, or abstain "
396
+ f"if the analysis concluded the evidence does not support one.\n\nSTEP 1 analysis:\n{analysis}"
397
+ )
398
+ return _finalize(emit_model.generate(emit_prompt), evidence)
399
+
400
+
401
+ def generate_answer_best_of_n(
402
+ query: str, evidence: list[EvidenceItem], *, model: AnswerModel, n: int = 5, min_answers: int = 1,
403
+ max_concurrency: int = 5,
404
+ ) -> GeneratedAnswer:
405
+ """Strategy C: sample the single-call generation `n` times (supply a temperature>0 `model` for genuine
406
+ diversity), run CONCURRENTLY, and take the best-cited NON-abstaining sample — abstaining only if fewer
407
+ than `min_answers` samples produced a valid cited answer. Self-consistency against the near-boundary
408
+ abstain flip: one good grounded sample is enough to answer; `min_answers`>1 demands agreement. Empty
409
+ evidence abstains without any model call."""
410
+ if not evidence:
411
+ return _abstain()
412
+ prompt = _answer_prompt(query, evidence)
413
+ raws = map_concurrent([prompt] * n, model.generate, max_concurrency=max_concurrency)
414
+ answered = [f for f in (_finalize(r, evidence) for r in raws if r is not None) if not f.abstained]
415
+ if len(answered) < min_answers:
416
+ return _abstain()
417
+ return max(answered, key=lambda f: len(f.citations))
418
+
419
+
420
+ def register_generation(registry: CapabilityRegistry) -> None:
421
+ """Register answer generation under FR-C.9 (`generation`; vision-to-text is its own slug, ADR-0014)."""
422
+ registry.register(
423
+ "generation",
424
+ contract=GeneratedAnswer,
425
+ kind="agent_skill", # a single grounded/cited LLM act (CAP-REG-1)
426
+ display_name="Answer generation (grounded, cited, abstains)",
427
+ )