rag-wright 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (184) hide show
  1. rag_wright/__init__.py +13 -0
  2. rag_wright/api/__init__.py +33 -0
  3. rag_wright/api/config.py +59 -0
  4. rag_wright/api/discover.py +70 -0
  5. rag_wright/api/documents.py +39 -0
  6. rag_wright/api/ids.py +31 -0
  7. rag_wright/api/invoke.py +99 -0
  8. rag_wright/api/kg.py +61 -0
  9. rag_wright/api/mcp.py +94 -0
  10. rag_wright/api/usage.py +30 -0
  11. rag_wright/api/workspace.py +85 -0
  12. rag_wright/capabilities/__init__.py +8 -0
  13. rag_wright/capabilities/answer_generator.py +427 -0
  14. rag_wright/capabilities/ard.py +286 -0
  15. rag_wright/capabilities/assertion_extraction.py +79 -0
  16. rag_wright/capabilities/chunk_read.py +58 -0
  17. rag_wright/capabilities/chunk_write.py +163 -0
  18. rag_wright/capabilities/claim_extraction.py +153 -0
  19. rag_wright/capabilities/clause_exception_linking.py +117 -0
  20. rag_wright/capabilities/compliance_judgment.py +322 -0
  21. rag_wright/capabilities/compliance_store.py +87 -0
  22. rag_wright/capabilities/contract_kg_serve.py +156 -0
  23. rag_wright/capabilities/contract_kg_store.py +251 -0
  24. rag_wright/capabilities/dg_extraction.py +585 -0
  25. rag_wright/capabilities/disambiguation.py +163 -0
  26. rag_wright/capabilities/document_parse.py +87 -0
  27. rag_wright/capabilities/document_scope.py +49 -0
  28. rag_wright/capabilities/embedding.py +164 -0
  29. rag_wright/capabilities/embedding_profiles.py +43 -0
  30. rag_wright/capabilities/entity_resolution.py +154 -0
  31. rag_wright/capabilities/fusion.py +64 -0
  32. rag_wright/capabilities/graph_extraction.py +243 -0
  33. rag_wright/capabilities/graph_query.py +73 -0
  34. rag_wright/capabilities/graph_storage.py +111 -0
  35. rag_wright/capabilities/highlight_serve.py +142 -0
  36. rag_wright/capabilities/hybrid_search.py +65 -0
  37. rag_wright/capabilities/invoke.py +31 -0
  38. rag_wright/capabilities/jev_decision.py +38 -0
  39. rag_wright/capabilities/manifests.py +872 -0
  40. rag_wright/capabilities/okf_navigate.py +456 -0
  41. rag_wright/capabilities/parsing.py +286 -0
  42. rag_wright/capabilities/property_boosted_retrieval.py +125 -0
  43. rag_wright/capabilities/query_function_classifier.py +94 -0
  44. rag_wright/capabilities/query_understanding.py +109 -0
  45. rag_wright/capabilities/registry.py +262 -0
  46. rag_wright/capabilities/remote_encoders.py +94 -0
  47. rag_wright/capabilities/requirement_extraction.py +247 -0
  48. rag_wright/capabilities/reranking.py +123 -0
  49. rag_wright/capabilities/retrieval_core.py +126 -0
  50. rag_wright/capabilities/rlm_chunking.py +808 -0
  51. rag_wright/capabilities/rlm_synthesis.py +316 -0
  52. rag_wright/capabilities/scan_quality.py +136 -0
  53. rag_wright/capabilities/span_relevance_judgment.py +191 -0
  54. rag_wright/capabilities/vision_to_text.py +85 -0
  55. rag_wright/capabilities/vlm_ocr.py +85 -0
  56. rag_wright/contracts/__init__.py +6 -0
  57. rag_wright/contracts/chunk.py +79 -0
  58. rag_wright/contracts/compliance.py +303 -0
  59. rag_wright/contracts/contract_meta.py +27 -0
  60. rag_wright/contracts/extraction.py +130 -0
  61. rag_wright/contracts/function.py +167 -0
  62. rag_wright/contracts/function_routing.py +91 -0
  63. rag_wright/contracts/highlight.py +74 -0
  64. rag_wright/contracts/identifiers.py +153 -0
  65. rag_wright/contracts/jurisdiction.py +96 -0
  66. rag_wright/contracts/ontology.py +142 -0
  67. rag_wright/contracts/property.py +201 -0
  68. rag_wright/contracts/provenance.py +78 -0
  69. rag_wright/contracts/query_intent.py +53 -0
  70. rag_wright/contracts/span.py +76 -0
  71. rag_wright/contracts/value_match.py +84 -0
  72. rag_wright/corpus/__init__.py +0 -0
  73. rag_wright/corpus/canonicalize.py +116 -0
  74. rag_wright/corpus/cuad.py +153 -0
  75. rag_wright/corpus/cuad_ingestion.py +72 -0
  76. rag_wright/corpus/document_parser.py +299 -0
  77. rag_wright/corpus/edgar.py +231 -0
  78. rag_wright/corpus/gcs_ingestion.py +120 -0
  79. rag_wright/corpus/http.py +110 -0
  80. rag_wright/corpus/selection.py +152 -0
  81. rag_wright/mcp/__init__.py +11 -0
  82. rag_wright/mcp/compliance_server.py +299 -0
  83. rag_wright/mcp/intra_document_qa_server.py +170 -0
  84. rag_wright/mcp/relational_qa_server.py +171 -0
  85. rag_wright/mcp/session_store.py +64 -0
  86. rag_wright/mcp/typed_property_retrieval_server.py +191 -0
  87. rag_wright/models/__init__.py +8 -0
  88. rag_wright/models/profiles.py +331 -0
  89. rag_wright/models/seam.py +497 -0
  90. rag_wright/models/tag_structured.py +285 -0
  91. rag_wright/models/tracing.py +179 -0
  92. rag_wright/models/usage.py +102 -0
  93. rag_wright/okf/__init__.py +11 -0
  94. rag_wright/okf/compile.py +292 -0
  95. rag_wright/okf/document.py +47 -0
  96. rag_wright/okf/enrich.py +176 -0
  97. rag_wright/okf/links.py +190 -0
  98. rag_wright/okf/lint.py +105 -0
  99. rag_wright/ontology/__init__.py +6 -0
  100. rag_wright/ontology/_generated_template_meta.py +60 -0
  101. rag_wright/ontology/_generated_vocab.py +52 -0
  102. rag_wright/ontology/clause_template.py +964 -0
  103. rag_wright/ontology/codegen.py +84 -0
  104. rag_wright/ontology/compliance_bridge.ttl +186 -0
  105. rag_wright/ontology/contract_bridge.ttl +2685 -0
  106. rag_wright/ontology/contract_taxonomy.py +24 -0
  107. rag_wright/ontology/derive.py +58 -0
  108. rag_wright/ontology/loader.py +435 -0
  109. rag_wright/ontology/packs/ftc_16cfr255.ttl +29 -0
  110. rag_wright/ontology/registry.py +87 -0
  111. rag_wright/ontology/template_introspect.py +100 -0
  112. rag_wright/py.typed +0 -0
  113. rag_wright/reference/__init__.py +2 -0
  114. rag_wright/reference/compliance.py +41 -0
  115. rag_wright/reference/contract_seam.py +123 -0
  116. rag_wright/skills/__init__.py +7 -0
  117. rag_wright/skills/claim_extraction/SKILL.md +47 -0
  118. rag_wright/skills/claim_extraction/__init__.py +1 -0
  119. rag_wright/skills/claim_extraction/template.py +50 -0
  120. rag_wright/skills/compliance_judgment/SKILL.md +59 -0
  121. rag_wright/skills/corpus_ingest/SKILL.md +106 -0
  122. rag_wright/skills/extraction_semantic_judge/SKILL.md +51 -0
  123. rag_wright/skills/extraction_semantic_judge/__init__.py +1 -0
  124. rag_wright/skills/generation/SKILL.md +64 -0
  125. rag_wright/skills/generation/__init__.py +1 -0
  126. rag_wright/skills/generic_compliance_judgment/SKILL.md +58 -0
  127. rag_wright/skills/okf_navigate/SKILL.md +137 -0
  128. rag_wright/skills/requirement_extraction/SKILL.md +47 -0
  129. rag_wright/skills/requirement_extraction/__init__.py +1 -0
  130. rag_wright/skills/requirement_extraction/template.py +50 -0
  131. rag_wright/skills/rlm/SKILL.md +186 -0
  132. rag_wright/skills/rlm/__init__.py +31 -0
  133. rag_wright/skills/rlm/agent.py +292 -0
  134. rag_wright/skills/span_relevance_judgment/SKILL.md +67 -0
  135. rag_wright/skills/vision_to_text/SKILL.md +36 -0
  136. rag_wright/skills/vision_to_text/__init__.py +1 -0
  137. rag_wright/spans/__init__.py +1 -0
  138. rag_wright/spans/boundary.py +78 -0
  139. rag_wright/spans/clause_function_classifier.py +490 -0
  140. rag_wright/spans/clause_kg_extractor.py +337 -0
  141. rag_wright/spans/cuad_labels.py +81 -0
  142. rag_wright/spans/dim_classifier.py +158 -0
  143. rag_wright/spans/dim_fleet.json +411 -0
  144. rag_wright/spans/function_classifier.py +77 -0
  145. rag_wright/spans/function_families.py +62 -0
  146. rag_wright/spans/hybrid_classifier.py +103 -0
  147. rag_wright/spans/legalbert_classifier.py +83 -0
  148. rag_wright/spans/model_capabilities.py +107 -0
  149. rag_wright/spans/new_function_labels.py +111 -0
  150. rag_wright/spans/page_map.py +68 -0
  151. rag_wright/spans/property_extractor.py +365 -0
  152. rag_wright/spans/property_grounding.py +182 -0
  153. rag_wright/spans/reclassify.py +77 -0
  154. rag_wright/spans/scarce_function_labels.py +105 -0
  155. rag_wright/spans/segment.py +341 -0
  156. rag_wright/spans/semantic_judge.py +197 -0
  157. rag_wright/spans/symbolic_validation.py +131 -0
  158. rag_wright/spans/tag_clause_extractor.py +182 -0
  159. rag_wright/store/__init__.py +6 -0
  160. rag_wright/store/arcadedb.py +1135 -0
  161. rag_wright/store/chunk_text.py +66 -0
  162. rag_wright/store/seam.py +213 -0
  163. rag_wright/subgraphs/__init__.py +0 -0
  164. rag_wright/subgraphs/async_ingestion.py +204 -0
  165. rag_wright/subgraphs/compliance_check.py +1042 -0
  166. rag_wright/subgraphs/compliance_ingestion.py +306 -0
  167. rag_wright/subgraphs/contract_ingestion_pipeline.py +999 -0
  168. rag_wright/subgraphs/graph_extraction.py +102 -0
  169. rag_wright/subgraphs/intra_document_qa.py +328 -0
  170. rag_wright/subgraphs/observability.py +140 -0
  171. rag_wright/subgraphs/query_constraint_extraction.py +73 -0
  172. rag_wright/subgraphs/relational_qa.py +165 -0
  173. rag_wright/subgraphs/requirement_extraction.py +137 -0
  174. rag_wright/subgraphs/scaffold.py +65 -0
  175. rag_wright/subgraphs/semantic_chunking.py +183 -0
  176. rag_wright/subgraphs/typed_clause_extraction.py +172 -0
  177. rag_wright/subgraphs/typed_property_retrieval.py +278 -0
  178. rag_wright/util/__init__.py +1 -0
  179. rag_wright/util/concurrent.py +153 -0
  180. rag_wright/util/spacy_model.py +45 -0
  181. rag_wright-0.1.0.dist-info/METADATA +168 -0
  182. rag_wright-0.1.0.dist-info/RECORD +184 -0
  183. rag_wright-0.1.0.dist-info/WHEEL +4 -0
  184. rag_wright-0.1.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,292 @@
1
+ """OKF bundle compile (FR-K.1/K.4, T46): deterministic bundle write from sidecar text + enrichment.
2
+
3
+ Given the chunk texts (from the T40 sidecar) and the per-clause enrichment (category + description), write
4
+ an OKF v0.1 conformant bundle: one markdown file per clause (non-empty `type` frontmatter, body byte-faithful
5
+ to the sidecar text), organized `root -> <category>/ -> <clause>.md`, with an `index.md` per directory and the
6
+ bundle root stamped with `okf_version` and the compile-recipe version. No re-chunk; `chunk_id` is unchanged.
7
+
8
+ Deterministic (no model call) and content-hash gated: recompiling an unchanged corpus under the same recipe
9
+ does effectively no work. `recompile_category` rewrites one subtree (T51's fast-iteration path).
10
+
11
+ Registered under the canonical slug `okf_compile` (category 3, a foundation derivation: internal registry
12
+ entry, no ARD manifest -- same shape as `ontology_registry_derivation`).
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ import hashlib
18
+ import json
19
+ import re
20
+ from pathlib import Path
21
+
22
+ from pydantic import BaseModel
23
+
24
+ from rag_wright.capabilities.registry import CapabilityRegistry
25
+ from rag_wright.okf.document import parse_okf, serialize_okf
26
+ from rag_wright.okf.enrich import EnrichedClause
27
+
28
+ OKF_VERSION = "0.1"
29
+ RECIPE_VERSION = "okf-acord-v1"
30
+ CLAUSE_TYPE = "Clause"
31
+ UNCATEGORIZED = "_uncategorized"
32
+ _INDEX = "index.md"
33
+ _MANIFEST = ".okf_manifest.json"
34
+
35
+
36
+ class OkfBundleManifest(BaseModel):
37
+ """The compile result and the `okf_compile` capability contract: what the bundle root records."""
38
+
39
+ okf_version: str
40
+ recipe_version: str
41
+ n_clauses: int
42
+ n_categorized: int
43
+ categories: dict[str, int] # category label -> clause-file count (UNCATEGORIZED for abstentions)
44
+ content_hash: str
45
+
46
+
47
+ def _slug(label: str) -> str:
48
+ """Filesystem-safe directory slug for a category label (`IP Ownership/License` -> `ip-ownership-license`)."""
49
+ s = re.sub(r"[^a-z0-9]+", "-", label.lower()).strip("-")
50
+ return s or UNCATEGORIZED
51
+
52
+
53
+ def _source_doc_id(chunk_id: str) -> str:
54
+ return chunk_id.rsplit(":", 2)[0] # <source_doc_id>:<chunk_index>:<content_hash>
55
+
56
+
57
+ def _dir_for(e: EnrichedClause) -> str:
58
+ return _slug(e.category) if e.categorized else UNCATEGORIZED
59
+
60
+
61
+ def _concept_document(chunk_id: str, text: str, e: EnrichedClause) -> str:
62
+ frontmatter = {
63
+ "type": CLAUSE_TYPE,
64
+ "title": _source_doc_id(chunk_id),
65
+ "description": e.description,
66
+ "tags": [e.category] if e.categorized else [],
67
+ "category": e.category if e.categorized else UNCATEGORIZED,
68
+ "chunk_id": chunk_id,
69
+ "source_doc_id": _source_doc_id(chunk_id),
70
+ }
71
+ return serialize_okf(frontmatter, text)
72
+
73
+
74
+ def _bundle_hash(
75
+ texts: dict[str, str], enrichment: dict[str, EnrichedClause], recipe_version: str
76
+ ) -> str:
77
+ """A content hash over (chunk_id, category, description) plus the recipe version. chunk_id already
78
+ embeds the text content hash, so an unchanged clause + unchanged enrichment + unchanged recipe hashes
79
+ the same, and the compile gate skips."""
80
+ h = hashlib.sha256()
81
+ h.update(recipe_version.encode("utf-8"))
82
+ for cid in sorted(texts):
83
+ e = enrichment[cid]
84
+ h.update(f"\x00{cid}\x00{e.categorized}\x00{e.category}\x00{e.description}".encode("utf-8"))
85
+ return h.hexdigest()
86
+
87
+
88
+ def _index_text(entries: list[tuple[str, str, str]], *, heading: str) -> str:
89
+ """Render an index.md section: `* [title](link) - description`, sorted by title."""
90
+ lines = [f"# {heading}", ""]
91
+ for title, link, desc in sorted(entries, key=lambda e: e[0].lower()):
92
+ suffix = f" - {desc}" if desc else ""
93
+ lines.append(f"* [{title}]({link}){suffix}")
94
+ return "\n".join(lines) + "\n"
95
+
96
+
97
+ def _write_category_index(category_dir: Path) -> None:
98
+ entries: list[tuple[str, str, str]] = []
99
+ for md in sorted(category_dir.glob("*.md")):
100
+ if md.name == _INDEX:
101
+ continue
102
+ fm, _ = parse_okf(md.read_text(encoding="utf-8"))
103
+ entries.append((str(fm.get("title") or md.stem), md.name, str(fm.get("description") or "")))
104
+ if entries:
105
+ (category_dir / _INDEX).write_text(_index_text(entries, heading="Clauses"), encoding="utf-8")
106
+
107
+
108
+ def _write_root_index(root: Path, *, okf_version: str, recipe_version: str) -> None:
109
+ entries: list[tuple[str, str, str]] = []
110
+ for sub in sorted(p for p in root.iterdir() if p.is_dir()):
111
+ n = len([m for m in sub.glob("*.md") if m.name != _INDEX])
112
+ entries.append((sub.name, f"{sub.name}/{_INDEX}", f"{n} clauses"))
113
+ frontmatter = {"okf_version": okf_version, "compile_recipe_version": recipe_version}
114
+ body = _index_text(entries, heading="Categories")
115
+ (root / _INDEX).write_text(serialize_okf(frontmatter, body), encoding="utf-8")
116
+
117
+
118
+ def compile_bundle(
119
+ texts: dict[str, str],
120
+ enrichment: dict[str, EnrichedClause],
121
+ out_root: Path,
122
+ *,
123
+ recipe_version: str = RECIPE_VERSION,
124
+ okf_version: str = OKF_VERSION,
125
+ ) -> OkfBundleManifest:
126
+ """Compile the whole bundle. Content-hash gated: an unchanged corpus+recipe returns without rewriting."""
127
+ out_root = Path(out_root)
128
+ content_hash = _bundle_hash(texts, enrichment, recipe_version)
129
+ manifest_path = out_root / _MANIFEST
130
+ if manifest_path.exists():
131
+ prior = OkfBundleManifest.model_validate_json(manifest_path.read_text(encoding="utf-8"))
132
+ if prior.content_hash == content_hash and prior.recipe_version == recipe_version:
133
+ return prior # unchanged corpus + recipe -> effectively no work
134
+
135
+ out_root.mkdir(parents=True, exist_ok=True)
136
+ categories: dict[str, int] = {}
137
+ touched_dirs: set[Path] = set()
138
+ for chunk_id, text in sorted(texts.items()):
139
+ e = enrichment[chunk_id]
140
+ category_dir = out_root / _dir_for(e)
141
+ category_dir.mkdir(parents=True, exist_ok=True)
142
+ (category_dir / f"{_source_doc_id(chunk_id)}.md").write_text(
143
+ _concept_document(chunk_id, text, e), encoding="utf-8"
144
+ )
145
+ label = e.category if e.categorized else UNCATEGORIZED
146
+ categories[label] = categories.get(label, 0) + 1
147
+ touched_dirs.add(category_dir)
148
+
149
+ for category_dir in sorted(touched_dirs):
150
+ _write_category_index(category_dir)
151
+ _write_root_index(out_root, okf_version=okf_version, recipe_version=recipe_version)
152
+
153
+ manifest = OkfBundleManifest(
154
+ okf_version=okf_version,
155
+ recipe_version=recipe_version,
156
+ n_clauses=len(texts),
157
+ n_categorized=sum(1 for e in enrichment.values() if e.categorized),
158
+ categories=dict(sorted(categories.items())),
159
+ content_hash=content_hash,
160
+ )
161
+ manifest_path.write_text(manifest.model_dump_json(indent=2), encoding="utf-8")
162
+ return manifest
163
+
164
+
165
+ def recompile_category(
166
+ out_root: Path,
167
+ category_label: str,
168
+ texts: dict[str, str],
169
+ enrichment: dict[str, EnrichedClause],
170
+ *,
171
+ recipe_version: str = RECIPE_VERSION,
172
+ okf_version: str = OKF_VERSION,
173
+ ) -> None:
174
+ """Rewrite a single category subtree (its clause files + index) and refresh the root index only.
175
+
176
+ T51's fast-iteration path: a signpost-recipe change for one category costs one subtree, not a full
177
+ rebuild. `texts`/`enrichment` hold just that category's clauses.
178
+ """
179
+ out_root = Path(out_root)
180
+ category_dir = out_root / _slug(category_label)
181
+ category_dir.mkdir(parents=True, exist_ok=True)
182
+ for chunk_id, text in sorted(texts.items()):
183
+ (category_dir / f"{_source_doc_id(chunk_id)}.md").write_text(
184
+ _concept_document(chunk_id, text, enrichment[chunk_id]), encoding="utf-8"
185
+ )
186
+ _write_category_index(category_dir)
187
+ _write_root_index(out_root, okf_version=okf_version, recipe_version=recipe_version)
188
+
189
+
190
+ def register_okf_compile(registry: CapabilityRegistry) -> None:
191
+ """Register `okf_compile` (FR-K.1-K.4): an in-process `function`, category-3 (no ARD manifest)."""
192
+ registry.register(
193
+ "okf_compile",
194
+ contract=OkfBundleManifest,
195
+ kind="function",
196
+ display_name="OKF bundle compile (chunk-only, signpost-enriched)",
197
+ )
198
+
199
+
200
+ # --- ACORD compile driver (the RAC verify entrypoint: uv run python -m rag_wright.okf.compile) ----
201
+
202
+ _ACORD_SIDECAR = Path("data/acord/chunk_text")
203
+ _ACORD_OUT = Path("data/acord/okf/bundle")
204
+ _ACORD_ENRICH_CACHE = Path("data/acord/okf/enrichment.json")
205
+
206
+
207
+ def _load_sidecar_texts(sidecar_root: Path) -> dict[str, str]:
208
+ """All ingested chunk texts as {chunk_id: text} by reading the T40 sidecar (one JSON file per doc)."""
209
+ texts: dict[str, str] = {}
210
+ for path in sorted(sidecar_root.glob("*.json")):
211
+ texts.update(json.loads(path.read_text(encoding="utf-8")))
212
+ return texts
213
+
214
+
215
+ def _fallback_enrichment(chunk_id: str, text: str) -> EnrichedClause:
216
+ """Deterministic enrichment for a clause the model could not classify: uncategorized, first-sentence desc.
217
+
218
+ Guarantees every ingested clause gets a bundle file (RAC-46) even when the cheap model persistently fails
219
+ on it. These land in `_uncategorized/` and their count is reported, so the fallback is visible, not silent.
220
+ """
221
+ first = " ".join(text.split())[:120]
222
+ return EnrichedClause(chunk_id=chunk_id, category="", description=first or "(no description)", categorized=False)
223
+
224
+
225
+ def _load_env() -> None:
226
+ """Load `.env` into the environment for the CLI run (the seam reads OPENROUTER_* from os.environ).
227
+
228
+ Minimal parser mirroring conftest, so no new dependency and no reliance on the pytest env hook.
229
+ """
230
+ import os
231
+
232
+ env = Path(".env")
233
+ if not env.exists():
234
+ return
235
+ for line in env.read_text().splitlines():
236
+ line = line.strip()
237
+ if not line or line.startswith("#") or "=" not in line:
238
+ continue
239
+ key, value = line.split("=", 1)
240
+ os.environ.setdefault(key.strip(), value.strip())
241
+
242
+
243
+ def main() -> None:
244
+ import asyncio
245
+
246
+ from rag_wright.okf.enrich import (
247
+ SeamClassifier,
248
+ enrich_all,
249
+ load_enrich_cache,
250
+ save_enrich_cache,
251
+ )
252
+ from rag_wright.okf.lint import lint_bundle
253
+
254
+ _load_env()
255
+ texts = _load_sidecar_texts(_ACORD_SIDECAR)
256
+ print(f"[okf_compile] {len(texts)} ingested clauses from {_ACORD_SIDECAR}")
257
+
258
+ cache = load_enrich_cache(_ACORD_ENRICH_CACHE, RECIPE_VERSION)
259
+ enrichment: dict[str, EnrichedClause] = cache
260
+ if len(enrichment) < len(texts):
261
+ SeamClassifier()(next(iter(texts.values()))) # preflight: fail loud on a systemic error (bad key/model)
262
+ for attempt in range(1, 8): # resumable: each pass gates on the cache, retries only what is still missing
263
+ before = len(enrichment)
264
+ enrichment = asyncio.run(enrich_all(texts, SeamClassifier(), concurrency=12, cache=enrichment))
265
+ save_enrich_cache(_ACORD_ENRICH_CACHE, RECIPE_VERSION, enrichment) # LLM results only (no fallbacks)
266
+ print(f"[okf_compile] enriched {len(enrichment)}/{len(texts)} (pass {attempt})")
267
+ if len(enrichment) == len(texts) or len(enrichment) == before:
268
+ break # done, or a pass made no progress -> the residue goes to the deterministic fallback
269
+
270
+ missing = [cid for cid in texts if cid not in enrichment]
271
+ if missing:
272
+ if len(enrichment) < len(texts) // 2: # a majority failing is systemic, not a stubborn tail
273
+ raise RuntimeError(
274
+ f"only {len(enrichment)}/{len(texts)} clauses enriched by the model; aborting (systemic issue)"
275
+ )
276
+ print(f"[okf_compile] {len(missing)} clauses fell back to deterministic (uncategorized) enrichment")
277
+ for cid in missing: # every ingested clause must get a bundle file (RAC-46); fallbacks are not cached
278
+ enrichment[cid] = _fallback_enrichment(cid, texts[cid])
279
+
280
+ manifest = compile_bundle(texts, enrichment, _ACORD_OUT)
281
+ print(f"[okf_compile] bundle: {manifest.n_clauses} clauses, {manifest.n_categorized} categorized")
282
+ for label, n in manifest.categories.items():
283
+ print(f" {n:4d} {label}")
284
+ report = lint_bundle(_ACORD_OUT)
285
+ print(f"[okf_compile] lint: passes={report.passes} | orphan_rate={report.orphan_rate:.3f} | "
286
+ f"description_coverage={report.description_coverage:.3f} | "
287
+ f"broken_link_ratio={report.broken_link_ratio:.3f}")
288
+ print(f"[okf_compile] wrote {_ACORD_OUT}")
289
+
290
+
291
+ if __name__ == "__main__":
292
+ main()
@@ -0,0 +1,47 @@
1
+ """OKF concept-document (de)serialization: YAML frontmatter + markdown body.
2
+
3
+ Mirrors the OKF reference `OKFDocument` (knowledge-catalog/okf) on our stack: a document is a `---`
4
+ delimited YAML frontmatter block followed by a markdown body. `serialize_okf` preserves key order and
5
+ appends the body verbatim (so a chunk body stays byte-faithful to its sidecar text); `parse_okf`
6
+ round-trips it. YAML handles the escaping of descriptions that carry colons, quotes, or brackets, so
7
+ the messy-real-text failure mode does not reach the frontmatter.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ from typing import Any
13
+
14
+ import yaml
15
+
16
+ _DELIM = "---"
17
+
18
+
19
+ class OkfParseError(ValueError):
20
+ """A concept document whose frontmatter is unterminated or is not a YAML mapping."""
21
+
22
+
23
+ def serialize_okf(frontmatter: dict[str, Any], body: str) -> str:
24
+ """Serialize a concept document: `---` frontmatter (key order preserved) then the body verbatim."""
25
+ fm_text = yaml.safe_dump(frontmatter, sort_keys=False, allow_unicode=True).rstrip()
26
+ body = body if body.endswith("\n") else body + "\n"
27
+ return f"{_DELIM}\n{fm_text}\n{_DELIM}\n\n{body}"
28
+
29
+
30
+ def parse_okf(text: str) -> tuple[dict[str, Any], str]:
31
+ """Parse a concept document into (frontmatter, body). No frontmatter -> ({}, whole text)."""
32
+ lines = text.splitlines()
33
+ if not lines or lines[0].strip() != _DELIM:
34
+ return {}, text
35
+ end = next((i for i in range(1, len(lines)) if lines[i].strip() == _DELIM), None)
36
+ if end is None:
37
+ raise OkfParseError("unterminated YAML frontmatter block")
38
+ try:
39
+ fm = yaml.safe_load("\n".join(lines[1:end])) or {}
40
+ except yaml.YAMLError as e:
41
+ raise OkfParseError(f"invalid YAML in frontmatter: {e}") from e
42
+ if not isinstance(fm, dict):
43
+ raise OkfParseError("frontmatter must be a YAML mapping")
44
+ body = "\n".join(lines[end + 1 :])
45
+ if body.startswith("\n"):
46
+ body = body[1:]
47
+ return fm, body
@@ -0,0 +1,176 @@
1
+ """OKF signpost enrichment (FR-K.2, T46): category + one-line description per clause.
2
+
3
+ The one model-bearing step of the compile. For each clause it produces the two signposts a traversal
4
+ filters on without reading a body: the category (drives the directory tree and `tags`) and a one-line
5
+ description (the `index.md` entry text). ACORD ships no clause categories and its stored "summary" is the
6
+ full clause text, so both are manufactured here by a corpus-appropriate classifier through the
7
+ model-profile seam, NOT by `graph_extraction`'s fixed CUAD ontology (ADR-0022: 41-CUAD covers only 67% of
8
+ ACORD gold; a direct ACORD-9 classifier agrees 91.6%).
9
+
10
+ Model: a cheap model for this simple task ONLY (the `OKF_ENRICHMENT` profile role, ADR-0023). Everything
11
+ else in the system stays on its DeepSeek/Gemma role. The call is content-hash gated (keyed by `chunk_id`,
12
+ which embeds the
13
+ content hash) and concurrent (async + semaphore), per the parallel-LLM rule.
14
+ """
15
+
16
+ from __future__ import annotations
17
+
18
+ import asyncio
19
+ import json
20
+ from enum import Enum
21
+ from pathlib import Path
22
+ from typing import Optional, Protocol, runtime_checkable
23
+
24
+ from pydantic import BaseModel
25
+
26
+ from rag_wright.models.profiles import ModelRole, model_for
27
+ from rag_wright.models.seam import build_structured
28
+
29
+ # ACORD's own 9 attorney categories (the corpus-appropriate label set; ADR-0022). A different corpus
30
+ # supplies its own list.
31
+ ACORD_CATEGORIES = [
32
+ "Limitation of Liability",
33
+ "Indemnification",
34
+ "Restrictive Covenants",
35
+ "Governing Law",
36
+ "Affirmative Covenants",
37
+ "IP Ownership/License",
38
+ "Term",
39
+ "Liquidated Damages",
40
+ "third party beneficiary clause",
41
+ ]
42
+ _NONE = "None of these"
43
+
44
+
45
+ class AcordCat(str, Enum):
46
+ LOL = "Limitation of Liability"
47
+ INDEMN = "Indemnification"
48
+ RESTRICT = "Restrictive Covenants"
49
+ GOVLAW = "Governing Law"
50
+ AFFIRM = "Affirmative Covenants"
51
+ IP = "IP Ownership/License"
52
+ TERM = "Term"
53
+ LIQDAM = "Liquidated Damages"
54
+ TPB = "third party beneficiary clause"
55
+ NONE = _NONE
56
+
57
+
58
+ class ClauseEnrichment(BaseModel):
59
+ """The structured classifier output: one category (or 'None of these') and a one-line description."""
60
+
61
+ category: AcordCat
62
+ description: str
63
+
64
+
65
+ class EnrichedClause(BaseModel):
66
+ """One clause's signposts, keyed by chunk_id. `categorized` is False when the classifier abstained."""
67
+
68
+ chunk_id: str
69
+ category: str
70
+ description: str
71
+ categorized: bool
72
+
73
+
74
+ @runtime_checkable
75
+ class ClauseClassifier(Protocol):
76
+ """Text -> {category, description}. The seam a test stubs so the compile is exercised without a model."""
77
+
78
+ def __call__(self, text: str) -> ClauseEnrichment: ...
79
+
80
+
81
+ def enrichment_prompt(text: str, categories: list[str] = ACORD_CATEGORIES) -> str:
82
+ """The classify-and-describe prompt (shared by the real classifier and the bench, so they match)."""
83
+ return (
84
+ "You are enriching a contract clause for a knowledge index.\n"
85
+ "1) Classify it into exactly ONE category (or 'None of these' if none fit).\n"
86
+ "2) Write a ONE-LINE description (<= 15 words) that a lawyer could use to tell this clause apart "
87
+ "from others of the same category. Describe what THIS clause specifically says, not the category.\n\n"
88
+ "Categories:\n- " + "\n- ".join(categories) + "\n\n"
89
+ "Clause:\n" + text[:2000]
90
+ )
91
+
92
+
93
+ class SeamClassifier:
94
+ """The real classifier: structured output through the model-profile seam on the OKF_ENRICHMENT role.
95
+
96
+ Retries a few times because a cheap model occasionally emits no valid tool call, which
97
+ `with_structured_output` surfaces as a `None` return (not an exception); a bare `None` is a transient
98
+ miss, so retry, and only raise when it persists (the caller skips a persistent failure and re-gates it).
99
+ """
100
+
101
+ def __init__(
102
+ self, model_id: Optional[str] = None, categories: list[str] = ACORD_CATEGORIES, *, retries: int = 3
103
+ ) -> None:
104
+ self._runnable = build_structured(model_id or model_for(ModelRole.OKF_ENRICHMENT), ClauseEnrichment)
105
+ self._categories = categories
106
+ self._retries = retries
107
+
108
+ def __call__(self, text: str) -> ClauseEnrichment:
109
+ prompt = enrichment_prompt(text, self._categories)
110
+ last_error: Exception | None = None
111
+ for _ in range(self._retries):
112
+ try:
113
+ v = self._runnable.invoke(prompt)
114
+ except Exception as e: # noqa: BLE001 - transient provider/parse error; retry
115
+ last_error = e
116
+ continue
117
+ if v is not None:
118
+ return v
119
+ raise last_error or ValueError("no structured output after retries")
120
+
121
+
122
+ async def enrich_all(
123
+ texts: dict[str, str],
124
+ classify: ClauseClassifier,
125
+ *,
126
+ concurrency: int = 8,
127
+ cache: Optional[dict[str, EnrichedClause]] = None,
128
+ ) -> dict[str, EnrichedClause]:
129
+ """Enrich every chunk, reusing `cache` (content-hash gated by chunk_id) and classifying only the rest.
130
+
131
+ Concurrent (async + semaphore) over the un-cached clauses. Determinism holds: the returned map is keyed
132
+ by chunk_id regardless of completion order.
133
+ """
134
+ cache = cache or {}
135
+ todo = {cid: text for cid, text in texts.items() if cid not in cache}
136
+ sem = asyncio.Semaphore(concurrency)
137
+
138
+ async def one(chunk_id: str, text: str) -> tuple[str, Optional[EnrichedClause]]:
139
+ async with sem:
140
+ try:
141
+ v = await asyncio.to_thread(classify, text)
142
+ except Exception: # noqa: BLE001 - a rate-limited/failed call is skipped, not fatal;
143
+ return chunk_id, None # it stays un-cached so a gated re-run retries it
144
+ if v is None: # a classifier that yields no structured output is a skip, not a crash
145
+ return chunk_id, None
146
+ categorized = v.category != AcordCat.NONE
147
+ return chunk_id, EnrichedClause(
148
+ chunk_id=chunk_id,
149
+ category=v.category.value if categorized else "",
150
+ description=v.description,
151
+ categorized=categorized,
152
+ )
153
+
154
+ fresh = {cid: ec for cid, ec in await asyncio.gather(*(one(c, t) for c, t in todo.items())) if ec}
155
+ merged = {**cache, **fresh}
156
+ return {cid: merged[cid] for cid in texts if cid in merged} # excludes clauses that failed
157
+
158
+
159
+ def load_enrich_cache(path: Path, recipe_version: str) -> dict[str, EnrichedClause]:
160
+ """Load the enrichment cache if it matches `recipe_version`; a recipe change invalidates it."""
161
+ if not path.exists():
162
+ return {}
163
+ data = json.loads(path.read_text(encoding="utf-8"))
164
+ if data.get("recipe_version") != recipe_version:
165
+ return {}
166
+ return {cid: EnrichedClause.model_validate(e) for cid, e in data.get("entries", {}).items()}
167
+
168
+
169
+ def save_enrich_cache(path: Path, recipe_version: str, entries: dict[str, EnrichedClause]) -> None:
170
+ """Persist the enrichment cache keyed by recipe_version (the content-hash gate's durable half)."""
171
+ path.parent.mkdir(parents=True, exist_ok=True)
172
+ payload = {
173
+ "recipe_version": recipe_version,
174
+ "entries": {cid: e.model_dump() for cid, e in entries.items()},
175
+ }
176
+ path.write_text(json.dumps(payload, ensure_ascii=False, indent=2), encoding="utf-8")