rag-wright 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (184) hide show
  1. rag_wright/__init__.py +13 -0
  2. rag_wright/api/__init__.py +33 -0
  3. rag_wright/api/config.py +59 -0
  4. rag_wright/api/discover.py +70 -0
  5. rag_wright/api/documents.py +39 -0
  6. rag_wright/api/ids.py +31 -0
  7. rag_wright/api/invoke.py +99 -0
  8. rag_wright/api/kg.py +61 -0
  9. rag_wright/api/mcp.py +94 -0
  10. rag_wright/api/usage.py +30 -0
  11. rag_wright/api/workspace.py +85 -0
  12. rag_wright/capabilities/__init__.py +8 -0
  13. rag_wright/capabilities/answer_generator.py +427 -0
  14. rag_wright/capabilities/ard.py +286 -0
  15. rag_wright/capabilities/assertion_extraction.py +79 -0
  16. rag_wright/capabilities/chunk_read.py +58 -0
  17. rag_wright/capabilities/chunk_write.py +163 -0
  18. rag_wright/capabilities/claim_extraction.py +153 -0
  19. rag_wright/capabilities/clause_exception_linking.py +117 -0
  20. rag_wright/capabilities/compliance_judgment.py +322 -0
  21. rag_wright/capabilities/compliance_store.py +87 -0
  22. rag_wright/capabilities/contract_kg_serve.py +156 -0
  23. rag_wright/capabilities/contract_kg_store.py +251 -0
  24. rag_wright/capabilities/dg_extraction.py +585 -0
  25. rag_wright/capabilities/disambiguation.py +163 -0
  26. rag_wright/capabilities/document_parse.py +87 -0
  27. rag_wright/capabilities/document_scope.py +49 -0
  28. rag_wright/capabilities/embedding.py +164 -0
  29. rag_wright/capabilities/embedding_profiles.py +43 -0
  30. rag_wright/capabilities/entity_resolution.py +154 -0
  31. rag_wright/capabilities/fusion.py +64 -0
  32. rag_wright/capabilities/graph_extraction.py +243 -0
  33. rag_wright/capabilities/graph_query.py +73 -0
  34. rag_wright/capabilities/graph_storage.py +111 -0
  35. rag_wright/capabilities/highlight_serve.py +142 -0
  36. rag_wright/capabilities/hybrid_search.py +65 -0
  37. rag_wright/capabilities/invoke.py +31 -0
  38. rag_wright/capabilities/jev_decision.py +38 -0
  39. rag_wright/capabilities/manifests.py +872 -0
  40. rag_wright/capabilities/okf_navigate.py +456 -0
  41. rag_wright/capabilities/parsing.py +286 -0
  42. rag_wright/capabilities/property_boosted_retrieval.py +125 -0
  43. rag_wright/capabilities/query_function_classifier.py +94 -0
  44. rag_wright/capabilities/query_understanding.py +109 -0
  45. rag_wright/capabilities/registry.py +262 -0
  46. rag_wright/capabilities/remote_encoders.py +94 -0
  47. rag_wright/capabilities/requirement_extraction.py +247 -0
  48. rag_wright/capabilities/reranking.py +123 -0
  49. rag_wright/capabilities/retrieval_core.py +126 -0
  50. rag_wright/capabilities/rlm_chunking.py +808 -0
  51. rag_wright/capabilities/rlm_synthesis.py +316 -0
  52. rag_wright/capabilities/scan_quality.py +136 -0
  53. rag_wright/capabilities/span_relevance_judgment.py +191 -0
  54. rag_wright/capabilities/vision_to_text.py +85 -0
  55. rag_wright/capabilities/vlm_ocr.py +85 -0
  56. rag_wright/contracts/__init__.py +6 -0
  57. rag_wright/contracts/chunk.py +79 -0
  58. rag_wright/contracts/compliance.py +303 -0
  59. rag_wright/contracts/contract_meta.py +27 -0
  60. rag_wright/contracts/extraction.py +130 -0
  61. rag_wright/contracts/function.py +167 -0
  62. rag_wright/contracts/function_routing.py +91 -0
  63. rag_wright/contracts/highlight.py +74 -0
  64. rag_wright/contracts/identifiers.py +153 -0
  65. rag_wright/contracts/jurisdiction.py +96 -0
  66. rag_wright/contracts/ontology.py +142 -0
  67. rag_wright/contracts/property.py +201 -0
  68. rag_wright/contracts/provenance.py +78 -0
  69. rag_wright/contracts/query_intent.py +53 -0
  70. rag_wright/contracts/span.py +76 -0
  71. rag_wright/contracts/value_match.py +84 -0
  72. rag_wright/corpus/__init__.py +0 -0
  73. rag_wright/corpus/canonicalize.py +116 -0
  74. rag_wright/corpus/cuad.py +153 -0
  75. rag_wright/corpus/cuad_ingestion.py +72 -0
  76. rag_wright/corpus/document_parser.py +299 -0
  77. rag_wright/corpus/edgar.py +231 -0
  78. rag_wright/corpus/gcs_ingestion.py +120 -0
  79. rag_wright/corpus/http.py +110 -0
  80. rag_wright/corpus/selection.py +152 -0
  81. rag_wright/mcp/__init__.py +11 -0
  82. rag_wright/mcp/compliance_server.py +299 -0
  83. rag_wright/mcp/intra_document_qa_server.py +170 -0
  84. rag_wright/mcp/relational_qa_server.py +171 -0
  85. rag_wright/mcp/session_store.py +64 -0
  86. rag_wright/mcp/typed_property_retrieval_server.py +191 -0
  87. rag_wright/models/__init__.py +8 -0
  88. rag_wright/models/profiles.py +331 -0
  89. rag_wright/models/seam.py +497 -0
  90. rag_wright/models/tag_structured.py +285 -0
  91. rag_wright/models/tracing.py +179 -0
  92. rag_wright/models/usage.py +102 -0
  93. rag_wright/okf/__init__.py +11 -0
  94. rag_wright/okf/compile.py +292 -0
  95. rag_wright/okf/document.py +47 -0
  96. rag_wright/okf/enrich.py +176 -0
  97. rag_wright/okf/links.py +190 -0
  98. rag_wright/okf/lint.py +105 -0
  99. rag_wright/ontology/__init__.py +6 -0
  100. rag_wright/ontology/_generated_template_meta.py +60 -0
  101. rag_wright/ontology/_generated_vocab.py +52 -0
  102. rag_wright/ontology/clause_template.py +964 -0
  103. rag_wright/ontology/codegen.py +84 -0
  104. rag_wright/ontology/compliance_bridge.ttl +186 -0
  105. rag_wright/ontology/contract_bridge.ttl +2685 -0
  106. rag_wright/ontology/contract_taxonomy.py +24 -0
  107. rag_wright/ontology/derive.py +58 -0
  108. rag_wright/ontology/loader.py +435 -0
  109. rag_wright/ontology/packs/ftc_16cfr255.ttl +29 -0
  110. rag_wright/ontology/registry.py +87 -0
  111. rag_wright/ontology/template_introspect.py +100 -0
  112. rag_wright/py.typed +0 -0
  113. rag_wright/reference/__init__.py +2 -0
  114. rag_wright/reference/compliance.py +41 -0
  115. rag_wright/reference/contract_seam.py +123 -0
  116. rag_wright/skills/__init__.py +7 -0
  117. rag_wright/skills/claim_extraction/SKILL.md +47 -0
  118. rag_wright/skills/claim_extraction/__init__.py +1 -0
  119. rag_wright/skills/claim_extraction/template.py +50 -0
  120. rag_wright/skills/compliance_judgment/SKILL.md +59 -0
  121. rag_wright/skills/corpus_ingest/SKILL.md +106 -0
  122. rag_wright/skills/extraction_semantic_judge/SKILL.md +51 -0
  123. rag_wright/skills/extraction_semantic_judge/__init__.py +1 -0
  124. rag_wright/skills/generation/SKILL.md +64 -0
  125. rag_wright/skills/generation/__init__.py +1 -0
  126. rag_wright/skills/generic_compliance_judgment/SKILL.md +58 -0
  127. rag_wright/skills/okf_navigate/SKILL.md +137 -0
  128. rag_wright/skills/requirement_extraction/SKILL.md +47 -0
  129. rag_wright/skills/requirement_extraction/__init__.py +1 -0
  130. rag_wright/skills/requirement_extraction/template.py +50 -0
  131. rag_wright/skills/rlm/SKILL.md +186 -0
  132. rag_wright/skills/rlm/__init__.py +31 -0
  133. rag_wright/skills/rlm/agent.py +292 -0
  134. rag_wright/skills/span_relevance_judgment/SKILL.md +67 -0
  135. rag_wright/skills/vision_to_text/SKILL.md +36 -0
  136. rag_wright/skills/vision_to_text/__init__.py +1 -0
  137. rag_wright/spans/__init__.py +1 -0
  138. rag_wright/spans/boundary.py +78 -0
  139. rag_wright/spans/clause_function_classifier.py +490 -0
  140. rag_wright/spans/clause_kg_extractor.py +337 -0
  141. rag_wright/spans/cuad_labels.py +81 -0
  142. rag_wright/spans/dim_classifier.py +158 -0
  143. rag_wright/spans/dim_fleet.json +411 -0
  144. rag_wright/spans/function_classifier.py +77 -0
  145. rag_wright/spans/function_families.py +62 -0
  146. rag_wright/spans/hybrid_classifier.py +103 -0
  147. rag_wright/spans/legalbert_classifier.py +83 -0
  148. rag_wright/spans/model_capabilities.py +107 -0
  149. rag_wright/spans/new_function_labels.py +111 -0
  150. rag_wright/spans/page_map.py +68 -0
  151. rag_wright/spans/property_extractor.py +365 -0
  152. rag_wright/spans/property_grounding.py +182 -0
  153. rag_wright/spans/reclassify.py +77 -0
  154. rag_wright/spans/scarce_function_labels.py +105 -0
  155. rag_wright/spans/segment.py +341 -0
  156. rag_wright/spans/semantic_judge.py +197 -0
  157. rag_wright/spans/symbolic_validation.py +131 -0
  158. rag_wright/spans/tag_clause_extractor.py +182 -0
  159. rag_wright/store/__init__.py +6 -0
  160. rag_wright/store/arcadedb.py +1135 -0
  161. rag_wright/store/chunk_text.py +66 -0
  162. rag_wright/store/seam.py +213 -0
  163. rag_wright/subgraphs/__init__.py +0 -0
  164. rag_wright/subgraphs/async_ingestion.py +204 -0
  165. rag_wright/subgraphs/compliance_check.py +1042 -0
  166. rag_wright/subgraphs/compliance_ingestion.py +306 -0
  167. rag_wright/subgraphs/contract_ingestion_pipeline.py +999 -0
  168. rag_wright/subgraphs/graph_extraction.py +102 -0
  169. rag_wright/subgraphs/intra_document_qa.py +328 -0
  170. rag_wright/subgraphs/observability.py +140 -0
  171. rag_wright/subgraphs/query_constraint_extraction.py +73 -0
  172. rag_wright/subgraphs/relational_qa.py +165 -0
  173. rag_wright/subgraphs/requirement_extraction.py +137 -0
  174. rag_wright/subgraphs/scaffold.py +65 -0
  175. rag_wright/subgraphs/semantic_chunking.py +183 -0
  176. rag_wright/subgraphs/typed_clause_extraction.py +172 -0
  177. rag_wright/subgraphs/typed_property_retrieval.py +278 -0
  178. rag_wright/util/__init__.py +1 -0
  179. rag_wright/util/concurrent.py +153 -0
  180. rag_wright/util/spacy_model.py +45 -0
  181. rag_wright-0.1.0.dist-info/METADATA +168 -0
  182. rag_wright-0.1.0.dist-info/RECORD +184 -0
  183. rag_wright-0.1.0.dist-info/WHEEL +4 -0
  184. rag_wright-0.1.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,285 @@
1
+ """Client-side XML-tag structured output -- the app's LLM-AGNOSTIC structured-output mechanism (ADR-0045).
2
+
3
+ Server-side guided decoding (`response_format`/`with_structured_output` json_schema) is NOT portable: it runs
4
+ away to max length on self-hosted Gemma-4/vLLM and costs ~60s/call on OpenRouter->Cerebras, while PLAIN free-text
5
+ generation is fast and correct everywhere. So instead of forcing a JSON schema at decode time, we ask the model
6
+ to answer in light XML tags and parse them CLIENT-SIDE into the Pydantic contract. Tags (not JSON) because a
7
+ field body needs no escaping -- unlike a JSON string full of legal quotes/brackets/newlines.
8
+
9
+ `build_tag_structured` is a DROP-IN for `models.seam.build_structured` (same `(model_id, schema)` -> runnable
10
+ with `.invoke(prompt) -> schema instance`), so a caller swaps the mechanism by swapping the factory.
11
+
12
+ Scope (ADR-0045 + TAGPARSE-INGEST-1a): FLAT schemas -- scalars (str/int/float/bool), enum/Literal, `str | None`,
13
+ `list[<scalar>]` -- AND NESTED schemas: a single nested `BaseModel` field and `list[<BaseModel>]`, emitted and
14
+ parsed by recursion (nested `<field><sub>..</sub></field>`; list items as repeated `<item>..</item>` blocks).
15
+ That covers the query-side schemas (generation, query understanding, highlight field-extract, reader judgments),
16
+ the flat ingest judges (extraction_semantic_judge -> SemanticVerdict), AND the ingestion extraction contracts
17
+ (Clause with its nested bounded_by/caps/governed_by + the `excepts` list; ContractParties with `parties`), which
18
+ TAGPARSE-INGEST-1 routes off docling-graph's server-side json_object onto this path. The ingest clause-function
19
+ classifier (nested `list[SpanFunctions]` / `list[RawScore]`) still uses its OWN bespoke free-text tags + client-
20
+ side parse in `spans/clause_function_classifier.py`, not this generic parser (issue 0005), and is unchanged.
21
+ """
22
+
23
+ from __future__ import annotations
24
+
25
+ import re
26
+ import types
27
+ import typing
28
+ from enum import Enum
29
+ from typing import Any, get_args, get_origin
30
+
31
+ from pydantic import BaseModel, ValidationError
32
+
33
+ from rag_wright.models.seam import astream_text, build_model
34
+
35
+
36
+ def _unwrap_optional(ann: Any) -> Any:
37
+ """`X | None` / `Optional[X]` -> `X`; otherwise unchanged."""
38
+ if get_origin(ann) in (typing.Union, getattr(types, "UnionType", None)):
39
+ non_none = [a for a in get_args(ann) if a is not type(None)]
40
+ if len(non_none) == 1:
41
+ return non_none[0]
42
+ return ann
43
+
44
+
45
+ def _classify(ann: Any) -> tuple[str, type[BaseModel] | None, str]:
46
+ """Classify a field annotation into `(kind, submodel, hint)`:
47
+ - 'scalar' : a single scalar/enum/Literal/bool/number value (submodel None)
48
+ - 'list_scalar' : `list[<scalar/enum>]` (submodel None)
49
+ - 'nested' : a single nested `BaseModel` (submodel = that model)
50
+ - 'nested_list' : `list[<BaseModel>]` (submodel = the item model)
51
+ Nested kinds (TAGPARSE-INGEST-1a) let the ingestion contracts (Clause's bounded_by/caps/governed_by + the
52
+ `excepts` list, ContractParties' `parties` list) round-trip through tags; the emitter/parser recurse."""
53
+ ann = _unwrap_optional(ann)
54
+ if get_origin(ann) is list:
55
+ item = (get_args(ann) or (str,))[0]
56
+ if isinstance(item, type) and issubclass(item, BaseModel):
57
+ return "nested_list", item, ""
58
+ return "list_scalar", None, "one value per line"
59
+ if isinstance(ann, type) and issubclass(ann, BaseModel):
60
+ return "nested", ann, ""
61
+ if isinstance(ann, type) and issubclass(ann, Enum):
62
+ return "scalar", None, "one of: " + " | ".join(str(e.value) for e in ann)
63
+ if get_origin(ann) is typing.Literal:
64
+ return "scalar", None, "one of: " + " | ".join(str(v) for v in get_args(ann))
65
+ if ann is bool:
66
+ return "scalar", None, "true or false"
67
+ if ann in (int, float):
68
+ return "scalar", None, "a number"
69
+ return "scalar", None, "text"
70
+
71
+
72
+ def _field_lines(schema: type[BaseModel], fields: set[str] | None = None) -> list[str]:
73
+ """One tag-template block per field of `schema`; recurses into nested models and list[model] items. `fields`
74
+ (if given) restricts to that subset -- for the per-group ingestion passes (TAGPARSE-INGEST-1b)."""
75
+ out: list[str] = []
76
+ for name, field in schema.model_fields.items():
77
+ if fields is not None and name not in fields:
78
+ continue
79
+ kind, sub, hint = _classify(field.annotation)
80
+ desc = (field.description or "").strip()
81
+ # Guidance goes AFTER the tags (a trailing `-- ...`), NOT inside them: a hint placed inside the tag body
82
+ # (e.g. `(one of: a | b)`) gets ECHOED by the model (`<f>(a)</f>`), which then fails value/enum parsing.
83
+ # The tag body is left EMPTY for the model to fill with ONLY the value.
84
+ guide = "; ".join(g for g in (hint, desc) if g)
85
+ tail = f" -- {guide}" if guide else ""
86
+ if kind == "scalar":
87
+ out.append(f"<{name}></{name}>{tail}")
88
+ elif kind == "list_scalar":
89
+ out.append(f"<{name}>\n</{name}>{tail}")
90
+ elif kind == "nested":
91
+ inner = "\n".join(_field_lines(sub)) # type: ignore[arg-type]
92
+ out.append(f"<{name}>{(' -- ' + desc) if desc else ''}\n{inner}\n</{name}>")
93
+ else: # nested_list
94
+ inner = "\n".join(_field_lines(sub)) # type: ignore[arg-type]
95
+ out.append(f"<{name}>{(' -- ' + desc) if desc else ''} (repeat the <item> block once per entry)\n"
96
+ f"<item>\n{inner}\n</item>\n</{name}>")
97
+ return out
98
+
99
+
100
+ def tag_instructions(schema: type[BaseModel], fields: set[str] | None = None) -> str:
101
+ """The prompt appendix telling the model to emit one XML-tag block per field of `schema` (recursing into
102
+ nested models and list[model] items). `fields` (if given) restricts to that subset."""
103
+ return "\n".join([
104
+ "Respond using EXACTLY these XML-style tags, one block per field. Fill EACH tag body with ONLY the raw "
105
+ "value -- no quotes, no JSON, no markdown fences, and do NOT copy the guidance. Any text after `--` is "
106
+ "guidance for you, not part of the value:",
107
+ *_field_lines(schema, fields),
108
+ "Omit the tag entirely for any field whose value is unknown or not applicable.",
109
+ ])
110
+
111
+
112
+ def _extract(text: str, name: str) -> str | None:
113
+ m = re.search(rf"<{re.escape(name)}>(.*?)</{re.escape(name)}>", text, re.DOTALL | re.IGNORECASE)
114
+ return m.group(1) if m else None
115
+
116
+
117
+ def _field_value(body: str, field: Any, lenient: bool, full_text: str | None = None) -> Any:
118
+ """The Python value for one `<field>` body: scalar/enum/bool/number verbatim; `list[<scalar>]` split
119
+ one-per-line (commas too); a nested `BaseModel` recursed; a `list[<BaseModel>]` split on its `<item>` blocks
120
+ and each recursed. In `lenient` mode an unbuildable list ITEM is dropped (kept in strict).
121
+
122
+ For a nested `BaseModel`, the sub-model is parsed from `full_text` when given, not just the `<field>` body:
123
+ models often FLATTEN a nested field -- emitting `<governed_by>Delaware</governed_by>` then the sub-fields
124
+ `<jurisdiction_name>...`/`<law_multiplicity>...` as SIBLINGS rather than nested inside. Sub-field tag names are
125
+ unique in these contracts, so scanning the full text finds them whether nested or flattened."""
126
+ kind, sub, _ = _classify(field.annotation)
127
+ if kind == "scalar":
128
+ return body.strip()
129
+ if kind == "list_scalar":
130
+ return [x.strip() for x in re.split(r"[\r\n,]+", body.strip()) if x.strip()]
131
+ if kind == "nested":
132
+ src = full_text if full_text is not None else body
133
+ if not any((_extract(src, sn) or "").strip() for sn in sub.model_fields): # type: ignore[union-attr]
134
+ return None # no non-empty sub-field anywhere -> the nested block is absent (never a hollow model)
135
+ return parse_tagged(src, sub, lenient=lenient) # type: ignore[arg-type]
136
+ # nested_list
137
+ out = []
138
+ for it in re.findall(r"<item>(.*?)</item>", body, re.DOTALL | re.IGNORECASE):
139
+ try:
140
+ out.append(parse_tagged(it, sub, lenient=lenient)) # type: ignore[arg-type]
141
+ except ValidationError:
142
+ if not lenient:
143
+ raise
144
+ return out
145
+
146
+
147
+ def _prune_invalid_optionals(schema: type[BaseModel], data: dict[str, Any], err: ValidationError) -> BaseModel:
148
+ """Drop each NON-required top field the ValidationError blames (a partial nested block, a constraint-violating
149
+ scalar) so it falls back to its default, then rebuild ONCE. A required field cannot be omitted, so if the
150
+ error is (also) on a required field the rebuild re-raises -- the caller's graceful-degrade contract then owns
151
+ it. Only used on the last, lenient attempt."""
152
+ pruned = False
153
+ for e in err.errors():
154
+ loc = e.get("loc") or ()
155
+ if not loc:
156
+ continue
157
+ top = loc[0]
158
+ f = schema.model_fields.get(top) # type: ignore[arg-type]
159
+ if f is not None and not f.is_required() and top in data:
160
+ del data[top]
161
+ pruned = True
162
+ if not pruned:
163
+ raise err # nothing omittable (the failure is on required data) -> genuine, propagate
164
+ return schema(**data)
165
+
166
+
167
+ def parse_tagged(text: str, schema: type[BaseModel], *, lenient: bool = False,
168
+ fields: set[str] | None = None) -> BaseModel:
169
+ """Parse XML-tagged free text into `schema`. An absent tag is omitted so the field's default/optional applies;
170
+ Pydantic does the final validation. STRICT (default): a missing required field or a malformed/partial nested
171
+ block raises ValidationError (which `build_tag_structured` retries -- the re-ask). LENIENT (the last attempt,
172
+ TAGPARSE-INGEST-1a): omit-to-default -- a NON-required field whose value fails to validate (an unbuildable
173
+ nested block, a constraint-violating scalar, an invalid list item) is dropped to its default rather than
174
+ failing the whole extraction; a missing REQUIRED field still raises."""
175
+ data: dict[str, Any] = {}
176
+ for name, field in schema.model_fields.items():
177
+ if fields is not None and name not in fields:
178
+ continue
179
+ kind, _sub, _ = _classify(field.annotation)
180
+ body = _extract(text, name)
181
+ if body is None:
182
+ # ISSUE-0020: a nested single model may be FULLY FLATTENED -- the model emits its sub-fields as siblings
183
+ # with NO <name> wrapper at all (the temporal_bound miss: <temporal_duration>/<temporal_kind> flat, no
184
+ # <bounded_by>). Unlike the wrapper-present flatten (handled in `_field_value` via full_text), that case
185
+ # is invisible here because `_extract(text, name)` is None. Recover it from the full text by the sub-
186
+ # model's (unique) sub-field names; every other absent tag is genuinely absent -> omit.
187
+ if kind == "nested" and any((_extract(text, sn) or "").strip() for sn in _sub.model_fields): # type: ignore[union-attr]
188
+ try:
189
+ val = parse_tagged(text, _sub, lenient=lenient) # type: ignore[arg-type]
190
+ except ValidationError:
191
+ if not lenient:
192
+ raise
193
+ val = None
194
+ if val is not None:
195
+ data[name] = val
196
+ continue
197
+ if kind in ("scalar", "list_scalar") and not body.strip():
198
+ continue # an empty <tag></tag> means "not provided" -> omit (never coerce "" to an enum OTHER)
199
+ try:
200
+ val = _field_value(body, field, lenient, full_text=text)
201
+ except ValidationError:
202
+ if not lenient: # strict: let the re-ask handle it; lenient: omit this field (default applies)
203
+ raise
204
+ continue
205
+ if kind == "nested" and val is None:
206
+ continue # all-empty nested block -> absent (not a hollow model)
207
+ data[name] = val
208
+ try:
209
+ return schema(**data)
210
+ except ValidationError as exc:
211
+ if not lenient:
212
+ raise
213
+ return _prune_invalid_optionals(schema, data, exc)
214
+
215
+
216
+ class _TagStructuredRunnable:
217
+ """A `build_structured`-shaped runnable that drives structured output client-side (free-text + tag parse)."""
218
+
219
+ def __init__(
220
+ self, model_id: str, schema: type[BaseModel], *, temperature: float, max_tokens: int | None, retries: int,
221
+ label: str | None = None, fields: set[str] | None = None,
222
+ ) -> None:
223
+ self._model_id = model_id
224
+ self._schema = schema
225
+ self._temperature = temperature
226
+ self._max_tokens = max_tokens
227
+ self._retries = retries
228
+ self._label = label # ADR-0058: stage/call-site name for the deadline warning (threaded to astream_text)
229
+ self._fields = fields # TAGPARSE-INGEST-1b: restrict this pass to a field subset (per-group extraction)
230
+
231
+ def _full_prompt(self, prompt: Any) -> Any:
232
+ # `prompt` is a plain string or a LangChain message sequence (e.g. [SystemMessage, HumanMessage]); append
233
+ # the tag instructions as a trailing human turn either way.
234
+ instr = tag_instructions(self._schema, fields=self._fields)
235
+ return f"{prompt}\n\n{instr}" if isinstance(prompt, str) else [*prompt, ("human", instr)]
236
+
237
+ def invoke(self, prompt: Any, config: Any = None) -> BaseModel: # config accepted for runnable-compat, unused
238
+ # Drop-in for build_structured. The re-ask is STRICT (so a malformed/partial answer is re-asked); the
239
+ # FINAL attempt is LENIENT (omit-to-default) so a persistently-partial nested block degrades rather than
240
+ # failing the whole extraction (TAGPARSE-INGEST-1a).
241
+ full = self._full_prompt(prompt)
242
+ attempts = self._retries + 1
243
+ last: Exception | None = None
244
+ for i in range(attempts):
245
+ text = str(build_model(
246
+ self._model_id, temperature=self._temperature, max_tokens=self._max_tokens).invoke(full).content)
247
+ try:
248
+ return parse_tagged(text, self._schema, lenient=(i == attempts - 1), fields=self._fields)
249
+ except ValidationError as exc: # malformed/incomplete -> re-ask, bounded
250
+ last = exc
251
+ raise last # type: ignore[misc]
252
+
253
+ async def ainvoke(self, prompt: Any, config: Any = None) -> BaseModel:
254
+ # ASYNC-A3 (ADR-0057): the async tag-parse structured path -- stream the free-text answer (idle-drip
255
+ # detection + true wall-clock deadline via astream_text), then parse the light tags client-side, with the
256
+ # same bounded re-ask on a ValidationError.
257
+ full = self._full_prompt(prompt)
258
+ attempts = self._retries + 1
259
+ last: Exception | None = None
260
+ for i in range(attempts):
261
+ text = await astream_text(
262
+ self._model_id, full, temperature=self._temperature, max_tokens=self._max_tokens,
263
+ label=self._label)
264
+ try:
265
+ return parse_tagged(text, self._schema, lenient=(i == attempts - 1), fields=self._fields)
266
+ except ValidationError as exc:
267
+ last = exc
268
+ raise last # type: ignore[misc]
269
+
270
+
271
+ def build_tag_structured(
272
+ model_id: str, schema: type[BaseModel], *, include_raw: bool = False, temperature: float = 0.0,
273
+ max_tokens: int | None = 2048, retries: int = 1, label: str | None = None, fields: set[str] | None = None,
274
+ ) -> _TagStructuredRunnable:
275
+ """Drop-in for `models.seam.build_structured`: returns a runnable whose `.invoke(prompt)` yields a validated
276
+ `schema` instance -- but via CLIENT-SIDE tag parsing (no server guided decoding), so it works on any model/
277
+ provider. `max_tokens` defaults to a generous cap (free-text terminates on its own). `label` (ADR-0058) names
278
+ the stage/call-site in the deadline warning. `fields` (TAGPARSE-INGEST-1b) restricts emission+parsing to a
279
+ subset of `schema`'s fields -- for per-group ingestion passes over one big schema. `include_raw` is accepted
280
+ for signature-compat but not supported (no query-side caller uses it)."""
281
+ if include_raw:
282
+ raise NotImplementedError("tag_structured: include_raw is not supported")
283
+ return _TagStructuredRunnable(
284
+ model_id, schema, temperature=temperature, max_tokens=max_tokens, retries=retries, label=label,
285
+ fields=fields)
@@ -0,0 +1,179 @@
1
+ """Engine LLM tracing to Langfuse (engine issue 0017).
2
+
3
+ The engine owns its model calls, so it owns their instrumentation: one Langfuse **generation** per model call,
4
+ carrying model id, token usage, latency, and the semantic metadata only the engine holds (the ADR-0058 `label`
5
+ plus role/stage and any caller correlation id). The product queries Langfuse directly -- this module is an
6
+ EMISSION contract, never a read API.
7
+
8
+ Gating (two independent conditions, BOTH required to emit):
9
+ 1. `RAG_TRACE_LEVEL` in {generations, verbose} (default `off`).
10
+ 2. `LANGFUSE_*` credentials actually configured.
11
+ Activation hinges on real CONFIGURATION, never on importability -- langfuse (and its transitive opentelemetry-sdk)
12
+ being installed says nothing about whether the operator wants telemetry (the FastMCP/OTel lesson, restated by the
13
+ issue). `generations` emits model+usage+latency+metadata; `verbose` also captures the input prompt and output.
14
+
15
+ Tracing must NEVER break a model call: every langfuse touch is wrapped and degrades to a no-op on any error.
16
+ """
17
+
18
+ from __future__ import annotations
19
+
20
+ import os
21
+ from contextlib import contextmanager
22
+ from typing import Any, Iterator, Optional
23
+
24
+ _LEVELS = ("off", "generations", "verbose")
25
+
26
+
27
+ def trace_level() -> str:
28
+ lvl = os.environ.get("RAG_TRACE_LEVEL", "off").strip().lower()
29
+ return lvl if lvl in _LEVELS else "off"
30
+
31
+
32
+ def _configured() -> bool:
33
+ """Gate on REAL config -- the credentials -- not on `import langfuse` succeeding."""
34
+ return bool(os.environ.get("LANGFUSE_PUBLIC_KEY") and os.environ.get("LANGFUSE_SECRET_KEY"))
35
+
36
+
37
+ def tracing_on() -> bool:
38
+ return trace_level() != "off" and _configured()
39
+
40
+
41
+ _client: Any = None
42
+ _client_tried = False
43
+
44
+
45
+ def _get_client() -> Any:
46
+ """Lazy process-lifetime Langfuse client, constructed only when tracing is on AND configured. Constructing it
47
+ initializes langfuse's OWN OpenTelemetry provider -- we never touch the engine's dormant observability seam."""
48
+ global _client, _client_tried
49
+ if _client is not None or _client_tried:
50
+ return _client
51
+ _client_tried = True
52
+ if not tracing_on():
53
+ return None
54
+ try:
55
+ from langfuse import Langfuse
56
+ _client = Langfuse() # reads LANGFUSE_PUBLIC_KEY / SECRET_KEY / HOST from env
57
+ except Exception: # noqa: BLE001 - tracing must never break the engine
58
+ _client = None
59
+ return _client
60
+
61
+
62
+ def start_generation(*, model: str, input: Any = None, label: Optional[str] = None,
63
+ role: Optional[str] = None, stage: Optional[str] = None,
64
+ metadata: Optional[dict[str, Any]] = None) -> Any:
65
+ """Open a Langfuse generation AT THE CALL START and return the observation (or None when tracing is off).
66
+ Its `start_time` is the moment this is called, so pairing it with `finish_generation` after the call makes
67
+ the span's OWN duration the real wall-clock latency. This matters because langfuse v4 has no way to
68
+ back-date a start: a post-hoc emit (open + immediately end) reports a ~0s duration and the real figure
69
+ survives only in metadata -- the defect engine issue 0048 hit (Langfuse's latency column read 0 for
70
+ `astream_text`/`span-relevance`). `input` is captured only at `verbose`; label/role/stage go into metadata.
71
+ Document/job grouping comes from the ambient `traced_run`."""
72
+ lf = _get_client()
73
+ if lf is None:
74
+ return None
75
+ verbose = trace_level() == "verbose"
76
+ md = {"label": label, "role": role, "stage": stage, **(metadata or {})}
77
+ md = {k: v for k, v in md.items() if v is not None}
78
+ try:
79
+ return lf.start_observation(
80
+ name=label or stage or "llm", as_type="generation",
81
+ input=input if verbose else None, model=model, metadata=md or None,
82
+ )
83
+ except Exception: # noqa: BLE001 - never let tracing break a model call
84
+ return None
85
+
86
+
87
+ def finish_generation(gen: Any, *, output: Any = None, usage: Optional[dict[str, int]] = None,
88
+ cost: Optional[float] = None, latency_ms: Optional[float] = None,
89
+ completion_start_time: Any = None,
90
+ metadata: Optional[dict[str, Any]] = None) -> None:
91
+ """Attach a completed call's results to a generation opened by `start_generation` and END it (end_time =
92
+ now), so the observation's duration is the true latency. No-op when `gen` is None (tracing off). `output`
93
+ captured only at `verbose`; `usage` = {"input": n, "output": n}; `cost` is the provider's ACTUAL total USD
94
+ (OpenRouter pass-through, not an engine price table) -> `cost_details` (None when the backend omits it, e.g.
95
+ self-hosted vLLM, so Langfuse prices from its own table). `completion_start_time` (time to first token) lets
96
+ Langfuse split queue+prefill from decode -- the attribution engine issue 0048 asked for. `latency_ms` is
97
+ also kept in metadata (redundant with the now-correct span duration, but exact)."""
98
+ if gen is None:
99
+ return
100
+ verbose = trace_level() == "verbose"
101
+ late = {"latency_ms": latency_ms, **(metadata or {})}
102
+ late = {k: v for k, v in late.items() if v is not None}
103
+ try:
104
+ gen.update(output=output if verbose else None, usage_details=usage or None,
105
+ cost_details=({"total": cost} if cost is not None else None),
106
+ completion_start_time=completion_start_time, metadata=late or None)
107
+ gen.end()
108
+ except Exception: # noqa: BLE001 - never let tracing break a model call
109
+ pass
110
+
111
+
112
+ def record_generation(*, model: str, input: Any = None, output: Any = None,
113
+ usage: Optional[dict[str, int]] = None, cost: Optional[float] = None,
114
+ latency_ms: Optional[float] = None, label: Optional[str] = None,
115
+ role: Optional[str] = None, stage: Optional[str] = None,
116
+ metadata: Optional[dict[str, Any]] = None) -> None:
117
+ """Post-hoc convenience: open + immediately end a generation. The measured span duration is ~0 (the real
118
+ latency survives only in metadata) -- PREFER `start_generation`/`finish_generation` around the call so
119
+ Langfuse's own latency is correct (issue 0048). Kept for a caller that genuinely has only post-call data."""
120
+ gen = start_generation(model=model, input=input, label=label, role=role, stage=stage, metadata=metadata)
121
+ finish_generation(gen, output=output, usage=usage, cost=cost, latency_ms=latency_ms)
122
+
123
+
124
+ @contextmanager
125
+ def traced_step(name: str, *, metadata: Optional[dict[str, Any]] = None) -> Iterator[None]:
126
+ """Time a NON-generation sub-step (e.g. retrieval: ArcadeDB + embedding + rerank) as its OWN Langfuse span,
127
+ so its duration is separable from the generation in the same trace -- the retrieval/generation split engine
128
+ issue 0048 asked for. No-op unless tracing is on. (Distinct from `subgraphs.observability.business_span`,
129
+ which is an OTel-ambient span that no-ops under a Langfuse-only setup -- this one emits to Langfuse.)"""
130
+ lf = _get_client()
131
+ if lf is None:
132
+ yield
133
+ return
134
+ obs = None
135
+ try:
136
+ obs = lf.start_observation(name=name, as_type="span", metadata=metadata or None)
137
+ except Exception: # noqa: BLE001 - never let tracing break the step
138
+ obs = None
139
+ try:
140
+ yield
141
+ finally:
142
+ if obs is not None:
143
+ try:
144
+ obs.end()
145
+ except Exception: # noqa: BLE001
146
+ pass
147
+
148
+
149
+ @contextmanager
150
+ def traced_run(*, document_id: Optional[str] = None, job_id: Optional[str] = None,
151
+ name: Optional[str] = None, metadata: Optional[dict[str, Any]] = None) -> Iterator[None]:
152
+ """Group every generation emitted inside the block under one trace/session -- the caller's correlation id
153
+ (job/document). A no-op unless tracing is on. Flushes on exit so a short-lived run's spans are sent."""
154
+ if not tracing_on():
155
+ yield
156
+ return
157
+ cm = None
158
+ try:
159
+ from langfuse import propagate_attributes
160
+ md = {"document_id": document_id, **(metadata or {})}
161
+ md = {k: v for k, v in md.items() if v is not None}
162
+ cm = propagate_attributes(session_id=job_id or document_id, metadata=md or None, trace_name=name)
163
+ cm.__enter__()
164
+ except Exception: # noqa: BLE001
165
+ cm = None
166
+ try:
167
+ yield
168
+ finally:
169
+ if cm is not None:
170
+ try:
171
+ cm.__exit__(None, None, None)
172
+ except Exception: # noqa: BLE001
173
+ pass
174
+ lf = _get_client()
175
+ if lf is not None:
176
+ try:
177
+ lf.flush()
178
+ except Exception: # noqa: BLE001
179
+ pass
@@ -0,0 +1,102 @@
1
+ """In-band model usage accounting (engine issue 0042): total round trips, tokens, OpenRouter's real per-call
2
+ cost, and latency, returned to the caller WITHOUT a Langfuse round-trip and WITHOUT tracing being on.
3
+
4
+ A caller opens a `usage_scope()` around an operation; the model seam records each call into the active scope.
5
+ The scope is ambient via a `ContextVar`, so no capability signature or graph-state changes. Usage is a property
6
+ of the CALL, captured whenever a scope is active — independent of `RAG_TRACE_LEVEL` (which defaults to `off`).
7
+
8
+ Design notes (issue 0042, per RuleWright):
9
+ - `calls_without_cost` is a DISTINCT counter (top-level AND per model): a call the backend/model returned no
10
+ cost for is UNKNOWN, never folded into `cost_usd` as a $0 that reads as free. A fully-unpriced run therefore
11
+ has `cost_usd == 0.0` with `calls_without_cost == calls` — a signal never to print the total as money.
12
+ - NESTING IS ADDITIVE: a call records into EVERY active scope on the stack, so a task-level outer scope totals
13
+ everything while inner per-operation scopes attribute their own slice. (Both are tested.)
14
+ - THREAD-SAFE: the accumulator is shared across the asyncio tasks and worker threads a scope spans (the engine
15
+ copies the context across its `to_thread` / dedicated-executor boundaries — ISSUE-0018), and every add is
16
+ locked, so concurrent model calls (the ingest/query gather + executor fan-out) accumulate correctly.
17
+ """
18
+
19
+ from __future__ import annotations
20
+
21
+ import contextvars
22
+ import threading
23
+ from contextlib import contextmanager
24
+ from dataclasses import dataclass, field
25
+ from typing import Iterator, Optional
26
+
27
+
28
+ @dataclass
29
+ class ModelUsage:
30
+ """Per-model totals within a scope (`cost_usd` sums KNOWN per-call costs only)."""
31
+
32
+ calls: int = 0
33
+ input_tokens: int = 0
34
+ output_tokens: int = 0
35
+ cost_usd: float = 0.0
36
+ calls_without_cost: int = 0 # calls the backend/model returned NO cost for -> unknown, NOT free
37
+ latency_ms_total: float = 0.0 # sum of per-call wall-clock ms; avg = latency_ms_total / calls
38
+
39
+
40
+ @dataclass
41
+ class UsageTotals:
42
+ """The usage accumulated within one `usage_scope()`: top-level totals + a per-model breakdown."""
43
+
44
+ calls: int = 0
45
+ input_tokens: int = 0
46
+ output_tokens: int = 0
47
+ cost_usd: float = 0.0
48
+ calls_without_cost: int = 0
49
+ latency_ms_total: float = 0.0
50
+ by_model: dict[str, ModelUsage] = field(default_factory=dict)
51
+ _lock: threading.Lock = field(default_factory=threading.Lock, repr=False, compare=False)
52
+
53
+ def _add(self, model: str, input_tokens: int, output_tokens: int,
54
+ cost: Optional[float], latency_ms: Optional[float]) -> None:
55
+ with self._lock:
56
+ per = self.by_model.get(model)
57
+ if per is None:
58
+ per = ModelUsage()
59
+ self.by_model[model] = per
60
+ for tgt in (self, per): # same shared counters on the top-level totals and this model's row
61
+ tgt.calls += 1
62
+ tgt.input_tokens += int(input_tokens or 0)
63
+ tgt.output_tokens += int(output_tokens or 0)
64
+ tgt.latency_ms_total += float(latency_ms or 0.0)
65
+ if cost is None:
66
+ tgt.calls_without_cost += 1
67
+ else:
68
+ tgt.cost_usd += float(cost)
69
+
70
+
71
+ # The STACK of active scopes (innermost last). A tuple set per scope-enter, so parallel contexts are isolated
72
+ # and copy_context() carries the whole stack (same UsageTotals objects) into a worker thread / executor.
73
+ _STACK: contextvars.ContextVar[tuple[UsageTotals, ...]] = contextvars.ContextVar("rag_usage_stack", default=())
74
+
75
+
76
+ @contextmanager
77
+ def usage_scope() -> Iterator[UsageTotals]:
78
+ """Accumulate model usage for the operation run inside the block; returns the totals to read after it.
79
+
80
+ Nesting is additive: a call made inside an inner scope is counted by the inner scope AND every enclosing
81
+ scope, so a task-level outer scope totals everything while inner per-operation scopes attribute their slice."""
82
+ totals = UsageTotals()
83
+ token = _STACK.set(_STACK.get() + (totals,))
84
+ try:
85
+ yield totals
86
+ finally:
87
+ _STACK.reset(token)
88
+
89
+
90
+ def usage_capturing() -> bool:
91
+ """True when at least one usage scope is active — the seam checks this to decide whether to capture usage
92
+ (e.g. request `stream_usage`) on a call that would otherwise not need it."""
93
+ return bool(_STACK.get())
94
+
95
+
96
+ def record_usage(model: str, *, input_tokens: int = 0, output_tokens: int = 0,
97
+ cost: Optional[float] = None, latency_ms: Optional[float] = None) -> None:
98
+ """Record one model call into every active scope (a no-op when none is active, so it is safe to call on
99
+ every model call regardless of tracing). `cost=None` means the backend surfaced no cost (counted as
100
+ `calls_without_cost`, never as $0). `model` is the engine model-id string, keyed consistently across paths."""
101
+ for totals in _STACK.get():
102
+ totals._add(model, input_tokens, output_tokens, cost, latency_ms)
@@ -0,0 +1,11 @@
1
+ """OKF (Open Knowledge Format) bundle compile pipeline (FR-K.1-K.4, T46).
2
+
3
+ Compiles the ingested chunk corpus into an OKF v0.1 conformant bundle for the embedding-free
4
+ knowledge-navigation path (FR-K). Not a re-chunk: bodies come from the T40 chunk-text sidecar and
5
+ `chunk_id` identity is unchanged (FR-S.2). Three parts:
6
+
7
+ - `enrich` : the one model-bearing step (category + one-line description per clause), gated and
8
+ concurrent. Uses a cheap model for this simple task only (ADR-0023).
9
+ - `compile` : deterministic bundle write (tree, frontmatter, index.md, content-hash gate).
10
+ - `lint` : deterministic conformance linter (frontmatter, type, links, coverage).
11
+ """