rag-wright 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- rag_wright/__init__.py +13 -0
- rag_wright/api/__init__.py +33 -0
- rag_wright/api/config.py +59 -0
- rag_wright/api/discover.py +70 -0
- rag_wright/api/documents.py +39 -0
- rag_wright/api/ids.py +31 -0
- rag_wright/api/invoke.py +99 -0
- rag_wright/api/kg.py +61 -0
- rag_wright/api/mcp.py +94 -0
- rag_wright/api/usage.py +30 -0
- rag_wright/api/workspace.py +85 -0
- rag_wright/capabilities/__init__.py +8 -0
- rag_wright/capabilities/answer_generator.py +427 -0
- rag_wright/capabilities/ard.py +286 -0
- rag_wright/capabilities/assertion_extraction.py +79 -0
- rag_wright/capabilities/chunk_read.py +58 -0
- rag_wright/capabilities/chunk_write.py +163 -0
- rag_wright/capabilities/claim_extraction.py +153 -0
- rag_wright/capabilities/clause_exception_linking.py +117 -0
- rag_wright/capabilities/compliance_judgment.py +322 -0
- rag_wright/capabilities/compliance_store.py +87 -0
- rag_wright/capabilities/contract_kg_serve.py +156 -0
- rag_wright/capabilities/contract_kg_store.py +251 -0
- rag_wright/capabilities/dg_extraction.py +585 -0
- rag_wright/capabilities/disambiguation.py +163 -0
- rag_wright/capabilities/document_parse.py +87 -0
- rag_wright/capabilities/document_scope.py +49 -0
- rag_wright/capabilities/embedding.py +164 -0
- rag_wright/capabilities/embedding_profiles.py +43 -0
- rag_wright/capabilities/entity_resolution.py +154 -0
- rag_wright/capabilities/fusion.py +64 -0
- rag_wright/capabilities/graph_extraction.py +243 -0
- rag_wright/capabilities/graph_query.py +73 -0
- rag_wright/capabilities/graph_storage.py +111 -0
- rag_wright/capabilities/highlight_serve.py +142 -0
- rag_wright/capabilities/hybrid_search.py +65 -0
- rag_wright/capabilities/invoke.py +31 -0
- rag_wright/capabilities/jev_decision.py +38 -0
- rag_wright/capabilities/manifests.py +872 -0
- rag_wright/capabilities/okf_navigate.py +456 -0
- rag_wright/capabilities/parsing.py +286 -0
- rag_wright/capabilities/property_boosted_retrieval.py +125 -0
- rag_wright/capabilities/query_function_classifier.py +94 -0
- rag_wright/capabilities/query_understanding.py +109 -0
- rag_wright/capabilities/registry.py +262 -0
- rag_wright/capabilities/remote_encoders.py +94 -0
- rag_wright/capabilities/requirement_extraction.py +247 -0
- rag_wright/capabilities/reranking.py +123 -0
- rag_wright/capabilities/retrieval_core.py +126 -0
- rag_wright/capabilities/rlm_chunking.py +808 -0
- rag_wright/capabilities/rlm_synthesis.py +316 -0
- rag_wright/capabilities/scan_quality.py +136 -0
- rag_wright/capabilities/span_relevance_judgment.py +191 -0
- rag_wright/capabilities/vision_to_text.py +85 -0
- rag_wright/capabilities/vlm_ocr.py +85 -0
- rag_wright/contracts/__init__.py +6 -0
- rag_wright/contracts/chunk.py +79 -0
- rag_wright/contracts/compliance.py +303 -0
- rag_wright/contracts/contract_meta.py +27 -0
- rag_wright/contracts/extraction.py +130 -0
- rag_wright/contracts/function.py +167 -0
- rag_wright/contracts/function_routing.py +91 -0
- rag_wright/contracts/highlight.py +74 -0
- rag_wright/contracts/identifiers.py +153 -0
- rag_wright/contracts/jurisdiction.py +96 -0
- rag_wright/contracts/ontology.py +142 -0
- rag_wright/contracts/property.py +201 -0
- rag_wright/contracts/provenance.py +78 -0
- rag_wright/contracts/query_intent.py +53 -0
- rag_wright/contracts/span.py +76 -0
- rag_wright/contracts/value_match.py +84 -0
- rag_wright/corpus/__init__.py +0 -0
- rag_wright/corpus/canonicalize.py +116 -0
- rag_wright/corpus/cuad.py +153 -0
- rag_wright/corpus/cuad_ingestion.py +72 -0
- rag_wright/corpus/document_parser.py +299 -0
- rag_wright/corpus/edgar.py +231 -0
- rag_wright/corpus/gcs_ingestion.py +120 -0
- rag_wright/corpus/http.py +110 -0
- rag_wright/corpus/selection.py +152 -0
- rag_wright/mcp/__init__.py +11 -0
- rag_wright/mcp/compliance_server.py +299 -0
- rag_wright/mcp/intra_document_qa_server.py +170 -0
- rag_wright/mcp/relational_qa_server.py +171 -0
- rag_wright/mcp/session_store.py +64 -0
- rag_wright/mcp/typed_property_retrieval_server.py +191 -0
- rag_wright/models/__init__.py +8 -0
- rag_wright/models/profiles.py +331 -0
- rag_wright/models/seam.py +497 -0
- rag_wright/models/tag_structured.py +285 -0
- rag_wright/models/tracing.py +179 -0
- rag_wright/models/usage.py +102 -0
- rag_wright/okf/__init__.py +11 -0
- rag_wright/okf/compile.py +292 -0
- rag_wright/okf/document.py +47 -0
- rag_wright/okf/enrich.py +176 -0
- rag_wright/okf/links.py +190 -0
- rag_wright/okf/lint.py +105 -0
- rag_wright/ontology/__init__.py +6 -0
- rag_wright/ontology/_generated_template_meta.py +60 -0
- rag_wright/ontology/_generated_vocab.py +52 -0
- rag_wright/ontology/clause_template.py +964 -0
- rag_wright/ontology/codegen.py +84 -0
- rag_wright/ontology/compliance_bridge.ttl +186 -0
- rag_wright/ontology/contract_bridge.ttl +2685 -0
- rag_wright/ontology/contract_taxonomy.py +24 -0
- rag_wright/ontology/derive.py +58 -0
- rag_wright/ontology/loader.py +435 -0
- rag_wright/ontology/packs/ftc_16cfr255.ttl +29 -0
- rag_wright/ontology/registry.py +87 -0
- rag_wright/ontology/template_introspect.py +100 -0
- rag_wright/py.typed +0 -0
- rag_wright/reference/__init__.py +2 -0
- rag_wright/reference/compliance.py +41 -0
- rag_wright/reference/contract_seam.py +123 -0
- rag_wright/skills/__init__.py +7 -0
- rag_wright/skills/claim_extraction/SKILL.md +47 -0
- rag_wright/skills/claim_extraction/__init__.py +1 -0
- rag_wright/skills/claim_extraction/template.py +50 -0
- rag_wright/skills/compliance_judgment/SKILL.md +59 -0
- rag_wright/skills/corpus_ingest/SKILL.md +106 -0
- rag_wright/skills/extraction_semantic_judge/SKILL.md +51 -0
- rag_wright/skills/extraction_semantic_judge/__init__.py +1 -0
- rag_wright/skills/generation/SKILL.md +64 -0
- rag_wright/skills/generation/__init__.py +1 -0
- rag_wright/skills/generic_compliance_judgment/SKILL.md +58 -0
- rag_wright/skills/okf_navigate/SKILL.md +137 -0
- rag_wright/skills/requirement_extraction/SKILL.md +47 -0
- rag_wright/skills/requirement_extraction/__init__.py +1 -0
- rag_wright/skills/requirement_extraction/template.py +50 -0
- rag_wright/skills/rlm/SKILL.md +186 -0
- rag_wright/skills/rlm/__init__.py +31 -0
- rag_wright/skills/rlm/agent.py +292 -0
- rag_wright/skills/span_relevance_judgment/SKILL.md +67 -0
- rag_wright/skills/vision_to_text/SKILL.md +36 -0
- rag_wright/skills/vision_to_text/__init__.py +1 -0
- rag_wright/spans/__init__.py +1 -0
- rag_wright/spans/boundary.py +78 -0
- rag_wright/spans/clause_function_classifier.py +490 -0
- rag_wright/spans/clause_kg_extractor.py +337 -0
- rag_wright/spans/cuad_labels.py +81 -0
- rag_wright/spans/dim_classifier.py +158 -0
- rag_wright/spans/dim_fleet.json +411 -0
- rag_wright/spans/function_classifier.py +77 -0
- rag_wright/spans/function_families.py +62 -0
- rag_wright/spans/hybrid_classifier.py +103 -0
- rag_wright/spans/legalbert_classifier.py +83 -0
- rag_wright/spans/model_capabilities.py +107 -0
- rag_wright/spans/new_function_labels.py +111 -0
- rag_wright/spans/page_map.py +68 -0
- rag_wright/spans/property_extractor.py +365 -0
- rag_wright/spans/property_grounding.py +182 -0
- rag_wright/spans/reclassify.py +77 -0
- rag_wright/spans/scarce_function_labels.py +105 -0
- rag_wright/spans/segment.py +341 -0
- rag_wright/spans/semantic_judge.py +197 -0
- rag_wright/spans/symbolic_validation.py +131 -0
- rag_wright/spans/tag_clause_extractor.py +182 -0
- rag_wright/store/__init__.py +6 -0
- rag_wright/store/arcadedb.py +1135 -0
- rag_wright/store/chunk_text.py +66 -0
- rag_wright/store/seam.py +213 -0
- rag_wright/subgraphs/__init__.py +0 -0
- rag_wright/subgraphs/async_ingestion.py +204 -0
- rag_wright/subgraphs/compliance_check.py +1042 -0
- rag_wright/subgraphs/compliance_ingestion.py +306 -0
- rag_wright/subgraphs/contract_ingestion_pipeline.py +999 -0
- rag_wright/subgraphs/graph_extraction.py +102 -0
- rag_wright/subgraphs/intra_document_qa.py +328 -0
- rag_wright/subgraphs/observability.py +140 -0
- rag_wright/subgraphs/query_constraint_extraction.py +73 -0
- rag_wright/subgraphs/relational_qa.py +165 -0
- rag_wright/subgraphs/requirement_extraction.py +137 -0
- rag_wright/subgraphs/scaffold.py +65 -0
- rag_wright/subgraphs/semantic_chunking.py +183 -0
- rag_wright/subgraphs/typed_clause_extraction.py +172 -0
- rag_wright/subgraphs/typed_property_retrieval.py +278 -0
- rag_wright/util/__init__.py +1 -0
- rag_wright/util/concurrent.py +153 -0
- rag_wright/util/spacy_model.py +45 -0
- rag_wright-0.1.0.dist-info/METADATA +168 -0
- rag_wright-0.1.0.dist-info/RECORD +184 -0
- rag_wright-0.1.0.dist-info/WHEEL +4 -0
- rag_wright-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,292 @@
|
|
|
1
|
+
"""OKF bundle compile (FR-K.1/K.4, T46): deterministic bundle write from sidecar text + enrichment.
|
|
2
|
+
|
|
3
|
+
Given the chunk texts (from the T40 sidecar) and the per-clause enrichment (category + description), write
|
|
4
|
+
an OKF v0.1 conformant bundle: one markdown file per clause (non-empty `type` frontmatter, body byte-faithful
|
|
5
|
+
to the sidecar text), organized `root -> <category>/ -> <clause>.md`, with an `index.md` per directory and the
|
|
6
|
+
bundle root stamped with `okf_version` and the compile-recipe version. No re-chunk; `chunk_id` is unchanged.
|
|
7
|
+
|
|
8
|
+
Deterministic (no model call) and content-hash gated: recompiling an unchanged corpus under the same recipe
|
|
9
|
+
does effectively no work. `recompile_category` rewrites one subtree (T51's fast-iteration path).
|
|
10
|
+
|
|
11
|
+
Registered under the canonical slug `okf_compile` (category 3, a foundation derivation: internal registry
|
|
12
|
+
entry, no ARD manifest -- same shape as `ontology_registry_derivation`).
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import hashlib
|
|
18
|
+
import json
|
|
19
|
+
import re
|
|
20
|
+
from pathlib import Path
|
|
21
|
+
|
|
22
|
+
from pydantic import BaseModel
|
|
23
|
+
|
|
24
|
+
from rag_wright.capabilities.registry import CapabilityRegistry
|
|
25
|
+
from rag_wright.okf.document import parse_okf, serialize_okf
|
|
26
|
+
from rag_wright.okf.enrich import EnrichedClause
|
|
27
|
+
|
|
28
|
+
OKF_VERSION = "0.1"
|
|
29
|
+
RECIPE_VERSION = "okf-acord-v1"
|
|
30
|
+
CLAUSE_TYPE = "Clause"
|
|
31
|
+
UNCATEGORIZED = "_uncategorized"
|
|
32
|
+
_INDEX = "index.md"
|
|
33
|
+
_MANIFEST = ".okf_manifest.json"
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
class OkfBundleManifest(BaseModel):
|
|
37
|
+
"""The compile result and the `okf_compile` capability contract: what the bundle root records."""
|
|
38
|
+
|
|
39
|
+
okf_version: str
|
|
40
|
+
recipe_version: str
|
|
41
|
+
n_clauses: int
|
|
42
|
+
n_categorized: int
|
|
43
|
+
categories: dict[str, int] # category label -> clause-file count (UNCATEGORIZED for abstentions)
|
|
44
|
+
content_hash: str
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def _slug(label: str) -> str:
|
|
48
|
+
"""Filesystem-safe directory slug for a category label (`IP Ownership/License` -> `ip-ownership-license`)."""
|
|
49
|
+
s = re.sub(r"[^a-z0-9]+", "-", label.lower()).strip("-")
|
|
50
|
+
return s or UNCATEGORIZED
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def _source_doc_id(chunk_id: str) -> str:
|
|
54
|
+
return chunk_id.rsplit(":", 2)[0] # <source_doc_id>:<chunk_index>:<content_hash>
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def _dir_for(e: EnrichedClause) -> str:
|
|
58
|
+
return _slug(e.category) if e.categorized else UNCATEGORIZED
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def _concept_document(chunk_id: str, text: str, e: EnrichedClause) -> str:
|
|
62
|
+
frontmatter = {
|
|
63
|
+
"type": CLAUSE_TYPE,
|
|
64
|
+
"title": _source_doc_id(chunk_id),
|
|
65
|
+
"description": e.description,
|
|
66
|
+
"tags": [e.category] if e.categorized else [],
|
|
67
|
+
"category": e.category if e.categorized else UNCATEGORIZED,
|
|
68
|
+
"chunk_id": chunk_id,
|
|
69
|
+
"source_doc_id": _source_doc_id(chunk_id),
|
|
70
|
+
}
|
|
71
|
+
return serialize_okf(frontmatter, text)
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def _bundle_hash(
|
|
75
|
+
texts: dict[str, str], enrichment: dict[str, EnrichedClause], recipe_version: str
|
|
76
|
+
) -> str:
|
|
77
|
+
"""A content hash over (chunk_id, category, description) plus the recipe version. chunk_id already
|
|
78
|
+
embeds the text content hash, so an unchanged clause + unchanged enrichment + unchanged recipe hashes
|
|
79
|
+
the same, and the compile gate skips."""
|
|
80
|
+
h = hashlib.sha256()
|
|
81
|
+
h.update(recipe_version.encode("utf-8"))
|
|
82
|
+
for cid in sorted(texts):
|
|
83
|
+
e = enrichment[cid]
|
|
84
|
+
h.update(f"\x00{cid}\x00{e.categorized}\x00{e.category}\x00{e.description}".encode("utf-8"))
|
|
85
|
+
return h.hexdigest()
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def _index_text(entries: list[tuple[str, str, str]], *, heading: str) -> str:
|
|
89
|
+
"""Render an index.md section: `* [title](link) - description`, sorted by title."""
|
|
90
|
+
lines = [f"# {heading}", ""]
|
|
91
|
+
for title, link, desc in sorted(entries, key=lambda e: e[0].lower()):
|
|
92
|
+
suffix = f" - {desc}" if desc else ""
|
|
93
|
+
lines.append(f"* [{title}]({link}){suffix}")
|
|
94
|
+
return "\n".join(lines) + "\n"
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def _write_category_index(category_dir: Path) -> None:
|
|
98
|
+
entries: list[tuple[str, str, str]] = []
|
|
99
|
+
for md in sorted(category_dir.glob("*.md")):
|
|
100
|
+
if md.name == _INDEX:
|
|
101
|
+
continue
|
|
102
|
+
fm, _ = parse_okf(md.read_text(encoding="utf-8"))
|
|
103
|
+
entries.append((str(fm.get("title") or md.stem), md.name, str(fm.get("description") or "")))
|
|
104
|
+
if entries:
|
|
105
|
+
(category_dir / _INDEX).write_text(_index_text(entries, heading="Clauses"), encoding="utf-8")
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def _write_root_index(root: Path, *, okf_version: str, recipe_version: str) -> None:
|
|
109
|
+
entries: list[tuple[str, str, str]] = []
|
|
110
|
+
for sub in sorted(p for p in root.iterdir() if p.is_dir()):
|
|
111
|
+
n = len([m for m in sub.glob("*.md") if m.name != _INDEX])
|
|
112
|
+
entries.append((sub.name, f"{sub.name}/{_INDEX}", f"{n} clauses"))
|
|
113
|
+
frontmatter = {"okf_version": okf_version, "compile_recipe_version": recipe_version}
|
|
114
|
+
body = _index_text(entries, heading="Categories")
|
|
115
|
+
(root / _INDEX).write_text(serialize_okf(frontmatter, body), encoding="utf-8")
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def compile_bundle(
|
|
119
|
+
texts: dict[str, str],
|
|
120
|
+
enrichment: dict[str, EnrichedClause],
|
|
121
|
+
out_root: Path,
|
|
122
|
+
*,
|
|
123
|
+
recipe_version: str = RECIPE_VERSION,
|
|
124
|
+
okf_version: str = OKF_VERSION,
|
|
125
|
+
) -> OkfBundleManifest:
|
|
126
|
+
"""Compile the whole bundle. Content-hash gated: an unchanged corpus+recipe returns without rewriting."""
|
|
127
|
+
out_root = Path(out_root)
|
|
128
|
+
content_hash = _bundle_hash(texts, enrichment, recipe_version)
|
|
129
|
+
manifest_path = out_root / _MANIFEST
|
|
130
|
+
if manifest_path.exists():
|
|
131
|
+
prior = OkfBundleManifest.model_validate_json(manifest_path.read_text(encoding="utf-8"))
|
|
132
|
+
if prior.content_hash == content_hash and prior.recipe_version == recipe_version:
|
|
133
|
+
return prior # unchanged corpus + recipe -> effectively no work
|
|
134
|
+
|
|
135
|
+
out_root.mkdir(parents=True, exist_ok=True)
|
|
136
|
+
categories: dict[str, int] = {}
|
|
137
|
+
touched_dirs: set[Path] = set()
|
|
138
|
+
for chunk_id, text in sorted(texts.items()):
|
|
139
|
+
e = enrichment[chunk_id]
|
|
140
|
+
category_dir = out_root / _dir_for(e)
|
|
141
|
+
category_dir.mkdir(parents=True, exist_ok=True)
|
|
142
|
+
(category_dir / f"{_source_doc_id(chunk_id)}.md").write_text(
|
|
143
|
+
_concept_document(chunk_id, text, e), encoding="utf-8"
|
|
144
|
+
)
|
|
145
|
+
label = e.category if e.categorized else UNCATEGORIZED
|
|
146
|
+
categories[label] = categories.get(label, 0) + 1
|
|
147
|
+
touched_dirs.add(category_dir)
|
|
148
|
+
|
|
149
|
+
for category_dir in sorted(touched_dirs):
|
|
150
|
+
_write_category_index(category_dir)
|
|
151
|
+
_write_root_index(out_root, okf_version=okf_version, recipe_version=recipe_version)
|
|
152
|
+
|
|
153
|
+
manifest = OkfBundleManifest(
|
|
154
|
+
okf_version=okf_version,
|
|
155
|
+
recipe_version=recipe_version,
|
|
156
|
+
n_clauses=len(texts),
|
|
157
|
+
n_categorized=sum(1 for e in enrichment.values() if e.categorized),
|
|
158
|
+
categories=dict(sorted(categories.items())),
|
|
159
|
+
content_hash=content_hash,
|
|
160
|
+
)
|
|
161
|
+
manifest_path.write_text(manifest.model_dump_json(indent=2), encoding="utf-8")
|
|
162
|
+
return manifest
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
def recompile_category(
|
|
166
|
+
out_root: Path,
|
|
167
|
+
category_label: str,
|
|
168
|
+
texts: dict[str, str],
|
|
169
|
+
enrichment: dict[str, EnrichedClause],
|
|
170
|
+
*,
|
|
171
|
+
recipe_version: str = RECIPE_VERSION,
|
|
172
|
+
okf_version: str = OKF_VERSION,
|
|
173
|
+
) -> None:
|
|
174
|
+
"""Rewrite a single category subtree (its clause files + index) and refresh the root index only.
|
|
175
|
+
|
|
176
|
+
T51's fast-iteration path: a signpost-recipe change for one category costs one subtree, not a full
|
|
177
|
+
rebuild. `texts`/`enrichment` hold just that category's clauses.
|
|
178
|
+
"""
|
|
179
|
+
out_root = Path(out_root)
|
|
180
|
+
category_dir = out_root / _slug(category_label)
|
|
181
|
+
category_dir.mkdir(parents=True, exist_ok=True)
|
|
182
|
+
for chunk_id, text in sorted(texts.items()):
|
|
183
|
+
(category_dir / f"{_source_doc_id(chunk_id)}.md").write_text(
|
|
184
|
+
_concept_document(chunk_id, text, enrichment[chunk_id]), encoding="utf-8"
|
|
185
|
+
)
|
|
186
|
+
_write_category_index(category_dir)
|
|
187
|
+
_write_root_index(out_root, okf_version=okf_version, recipe_version=recipe_version)
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
def register_okf_compile(registry: CapabilityRegistry) -> None:
|
|
191
|
+
"""Register `okf_compile` (FR-K.1-K.4): an in-process `function`, category-3 (no ARD manifest)."""
|
|
192
|
+
registry.register(
|
|
193
|
+
"okf_compile",
|
|
194
|
+
contract=OkfBundleManifest,
|
|
195
|
+
kind="function",
|
|
196
|
+
display_name="OKF bundle compile (chunk-only, signpost-enriched)",
|
|
197
|
+
)
|
|
198
|
+
|
|
199
|
+
|
|
200
|
+
# --- ACORD compile driver (the RAC verify entrypoint: uv run python -m rag_wright.okf.compile) ----
|
|
201
|
+
|
|
202
|
+
_ACORD_SIDECAR = Path("data/acord/chunk_text")
|
|
203
|
+
_ACORD_OUT = Path("data/acord/okf/bundle")
|
|
204
|
+
_ACORD_ENRICH_CACHE = Path("data/acord/okf/enrichment.json")
|
|
205
|
+
|
|
206
|
+
|
|
207
|
+
def _load_sidecar_texts(sidecar_root: Path) -> dict[str, str]:
|
|
208
|
+
"""All ingested chunk texts as {chunk_id: text} by reading the T40 sidecar (one JSON file per doc)."""
|
|
209
|
+
texts: dict[str, str] = {}
|
|
210
|
+
for path in sorted(sidecar_root.glob("*.json")):
|
|
211
|
+
texts.update(json.loads(path.read_text(encoding="utf-8")))
|
|
212
|
+
return texts
|
|
213
|
+
|
|
214
|
+
|
|
215
|
+
def _fallback_enrichment(chunk_id: str, text: str) -> EnrichedClause:
|
|
216
|
+
"""Deterministic enrichment for a clause the model could not classify: uncategorized, first-sentence desc.
|
|
217
|
+
|
|
218
|
+
Guarantees every ingested clause gets a bundle file (RAC-46) even when the cheap model persistently fails
|
|
219
|
+
on it. These land in `_uncategorized/` and their count is reported, so the fallback is visible, not silent.
|
|
220
|
+
"""
|
|
221
|
+
first = " ".join(text.split())[:120]
|
|
222
|
+
return EnrichedClause(chunk_id=chunk_id, category="", description=first or "(no description)", categorized=False)
|
|
223
|
+
|
|
224
|
+
|
|
225
|
+
def _load_env() -> None:
|
|
226
|
+
"""Load `.env` into the environment for the CLI run (the seam reads OPENROUTER_* from os.environ).
|
|
227
|
+
|
|
228
|
+
Minimal parser mirroring conftest, so no new dependency and no reliance on the pytest env hook.
|
|
229
|
+
"""
|
|
230
|
+
import os
|
|
231
|
+
|
|
232
|
+
env = Path(".env")
|
|
233
|
+
if not env.exists():
|
|
234
|
+
return
|
|
235
|
+
for line in env.read_text().splitlines():
|
|
236
|
+
line = line.strip()
|
|
237
|
+
if not line or line.startswith("#") or "=" not in line:
|
|
238
|
+
continue
|
|
239
|
+
key, value = line.split("=", 1)
|
|
240
|
+
os.environ.setdefault(key.strip(), value.strip())
|
|
241
|
+
|
|
242
|
+
|
|
243
|
+
def main() -> None:
|
|
244
|
+
import asyncio
|
|
245
|
+
|
|
246
|
+
from rag_wright.okf.enrich import (
|
|
247
|
+
SeamClassifier,
|
|
248
|
+
enrich_all,
|
|
249
|
+
load_enrich_cache,
|
|
250
|
+
save_enrich_cache,
|
|
251
|
+
)
|
|
252
|
+
from rag_wright.okf.lint import lint_bundle
|
|
253
|
+
|
|
254
|
+
_load_env()
|
|
255
|
+
texts = _load_sidecar_texts(_ACORD_SIDECAR)
|
|
256
|
+
print(f"[okf_compile] {len(texts)} ingested clauses from {_ACORD_SIDECAR}")
|
|
257
|
+
|
|
258
|
+
cache = load_enrich_cache(_ACORD_ENRICH_CACHE, RECIPE_VERSION)
|
|
259
|
+
enrichment: dict[str, EnrichedClause] = cache
|
|
260
|
+
if len(enrichment) < len(texts):
|
|
261
|
+
SeamClassifier()(next(iter(texts.values()))) # preflight: fail loud on a systemic error (bad key/model)
|
|
262
|
+
for attempt in range(1, 8): # resumable: each pass gates on the cache, retries only what is still missing
|
|
263
|
+
before = len(enrichment)
|
|
264
|
+
enrichment = asyncio.run(enrich_all(texts, SeamClassifier(), concurrency=12, cache=enrichment))
|
|
265
|
+
save_enrich_cache(_ACORD_ENRICH_CACHE, RECIPE_VERSION, enrichment) # LLM results only (no fallbacks)
|
|
266
|
+
print(f"[okf_compile] enriched {len(enrichment)}/{len(texts)} (pass {attempt})")
|
|
267
|
+
if len(enrichment) == len(texts) or len(enrichment) == before:
|
|
268
|
+
break # done, or a pass made no progress -> the residue goes to the deterministic fallback
|
|
269
|
+
|
|
270
|
+
missing = [cid for cid in texts if cid not in enrichment]
|
|
271
|
+
if missing:
|
|
272
|
+
if len(enrichment) < len(texts) // 2: # a majority failing is systemic, not a stubborn tail
|
|
273
|
+
raise RuntimeError(
|
|
274
|
+
f"only {len(enrichment)}/{len(texts)} clauses enriched by the model; aborting (systemic issue)"
|
|
275
|
+
)
|
|
276
|
+
print(f"[okf_compile] {len(missing)} clauses fell back to deterministic (uncategorized) enrichment")
|
|
277
|
+
for cid in missing: # every ingested clause must get a bundle file (RAC-46); fallbacks are not cached
|
|
278
|
+
enrichment[cid] = _fallback_enrichment(cid, texts[cid])
|
|
279
|
+
|
|
280
|
+
manifest = compile_bundle(texts, enrichment, _ACORD_OUT)
|
|
281
|
+
print(f"[okf_compile] bundle: {manifest.n_clauses} clauses, {manifest.n_categorized} categorized")
|
|
282
|
+
for label, n in manifest.categories.items():
|
|
283
|
+
print(f" {n:4d} {label}")
|
|
284
|
+
report = lint_bundle(_ACORD_OUT)
|
|
285
|
+
print(f"[okf_compile] lint: passes={report.passes} | orphan_rate={report.orphan_rate:.3f} | "
|
|
286
|
+
f"description_coverage={report.description_coverage:.3f} | "
|
|
287
|
+
f"broken_link_ratio={report.broken_link_ratio:.3f}")
|
|
288
|
+
print(f"[okf_compile] wrote {_ACORD_OUT}")
|
|
289
|
+
|
|
290
|
+
|
|
291
|
+
if __name__ == "__main__":
|
|
292
|
+
main()
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
"""OKF concept-document (de)serialization: YAML frontmatter + markdown body.
|
|
2
|
+
|
|
3
|
+
Mirrors the OKF reference `OKFDocument` (knowledge-catalog/okf) on our stack: a document is a `---`
|
|
4
|
+
delimited YAML frontmatter block followed by a markdown body. `serialize_okf` preserves key order and
|
|
5
|
+
appends the body verbatim (so a chunk body stays byte-faithful to its sidecar text); `parse_okf`
|
|
6
|
+
round-trips it. YAML handles the escaping of descriptions that carry colons, quotes, or brackets, so
|
|
7
|
+
the messy-real-text failure mode does not reach the frontmatter.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
from typing import Any
|
|
13
|
+
|
|
14
|
+
import yaml
|
|
15
|
+
|
|
16
|
+
_DELIM = "---"
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class OkfParseError(ValueError):
|
|
20
|
+
"""A concept document whose frontmatter is unterminated or is not a YAML mapping."""
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def serialize_okf(frontmatter: dict[str, Any], body: str) -> str:
|
|
24
|
+
"""Serialize a concept document: `---` frontmatter (key order preserved) then the body verbatim."""
|
|
25
|
+
fm_text = yaml.safe_dump(frontmatter, sort_keys=False, allow_unicode=True).rstrip()
|
|
26
|
+
body = body if body.endswith("\n") else body + "\n"
|
|
27
|
+
return f"{_DELIM}\n{fm_text}\n{_DELIM}\n\n{body}"
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def parse_okf(text: str) -> tuple[dict[str, Any], str]:
|
|
31
|
+
"""Parse a concept document into (frontmatter, body). No frontmatter -> ({}, whole text)."""
|
|
32
|
+
lines = text.splitlines()
|
|
33
|
+
if not lines or lines[0].strip() != _DELIM:
|
|
34
|
+
return {}, text
|
|
35
|
+
end = next((i for i in range(1, len(lines)) if lines[i].strip() == _DELIM), None)
|
|
36
|
+
if end is None:
|
|
37
|
+
raise OkfParseError("unterminated YAML frontmatter block")
|
|
38
|
+
try:
|
|
39
|
+
fm = yaml.safe_load("\n".join(lines[1:end])) or {}
|
|
40
|
+
except yaml.YAMLError as e:
|
|
41
|
+
raise OkfParseError(f"invalid YAML in frontmatter: {e}") from e
|
|
42
|
+
if not isinstance(fm, dict):
|
|
43
|
+
raise OkfParseError("frontmatter must be a YAML mapping")
|
|
44
|
+
body = "\n".join(lines[end + 1 :])
|
|
45
|
+
if body.startswith("\n"):
|
|
46
|
+
body = body[1:]
|
|
47
|
+
return fm, body
|
rag_wright/okf/enrich.py
ADDED
|
@@ -0,0 +1,176 @@
|
|
|
1
|
+
"""OKF signpost enrichment (FR-K.2, T46): category + one-line description per clause.
|
|
2
|
+
|
|
3
|
+
The one model-bearing step of the compile. For each clause it produces the two signposts a traversal
|
|
4
|
+
filters on without reading a body: the category (drives the directory tree and `tags`) and a one-line
|
|
5
|
+
description (the `index.md` entry text). ACORD ships no clause categories and its stored "summary" is the
|
|
6
|
+
full clause text, so both are manufactured here by a corpus-appropriate classifier through the
|
|
7
|
+
model-profile seam, NOT by `graph_extraction`'s fixed CUAD ontology (ADR-0022: 41-CUAD covers only 67% of
|
|
8
|
+
ACORD gold; a direct ACORD-9 classifier agrees 91.6%).
|
|
9
|
+
|
|
10
|
+
Model: a cheap model for this simple task ONLY (the `OKF_ENRICHMENT` profile role, ADR-0023). Everything
|
|
11
|
+
else in the system stays on its DeepSeek/Gemma role. The call is content-hash gated (keyed by `chunk_id`,
|
|
12
|
+
which embeds the
|
|
13
|
+
content hash) and concurrent (async + semaphore), per the parallel-LLM rule.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import asyncio
|
|
19
|
+
import json
|
|
20
|
+
from enum import Enum
|
|
21
|
+
from pathlib import Path
|
|
22
|
+
from typing import Optional, Protocol, runtime_checkable
|
|
23
|
+
|
|
24
|
+
from pydantic import BaseModel
|
|
25
|
+
|
|
26
|
+
from rag_wright.models.profiles import ModelRole, model_for
|
|
27
|
+
from rag_wright.models.seam import build_structured
|
|
28
|
+
|
|
29
|
+
# ACORD's own 9 attorney categories (the corpus-appropriate label set; ADR-0022). A different corpus
|
|
30
|
+
# supplies its own list.
|
|
31
|
+
ACORD_CATEGORIES = [
|
|
32
|
+
"Limitation of Liability",
|
|
33
|
+
"Indemnification",
|
|
34
|
+
"Restrictive Covenants",
|
|
35
|
+
"Governing Law",
|
|
36
|
+
"Affirmative Covenants",
|
|
37
|
+
"IP Ownership/License",
|
|
38
|
+
"Term",
|
|
39
|
+
"Liquidated Damages",
|
|
40
|
+
"third party beneficiary clause",
|
|
41
|
+
]
|
|
42
|
+
_NONE = "None of these"
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
class AcordCat(str, Enum):
|
|
46
|
+
LOL = "Limitation of Liability"
|
|
47
|
+
INDEMN = "Indemnification"
|
|
48
|
+
RESTRICT = "Restrictive Covenants"
|
|
49
|
+
GOVLAW = "Governing Law"
|
|
50
|
+
AFFIRM = "Affirmative Covenants"
|
|
51
|
+
IP = "IP Ownership/License"
|
|
52
|
+
TERM = "Term"
|
|
53
|
+
LIQDAM = "Liquidated Damages"
|
|
54
|
+
TPB = "third party beneficiary clause"
|
|
55
|
+
NONE = _NONE
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
class ClauseEnrichment(BaseModel):
|
|
59
|
+
"""The structured classifier output: one category (or 'None of these') and a one-line description."""
|
|
60
|
+
|
|
61
|
+
category: AcordCat
|
|
62
|
+
description: str
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
class EnrichedClause(BaseModel):
|
|
66
|
+
"""One clause's signposts, keyed by chunk_id. `categorized` is False when the classifier abstained."""
|
|
67
|
+
|
|
68
|
+
chunk_id: str
|
|
69
|
+
category: str
|
|
70
|
+
description: str
|
|
71
|
+
categorized: bool
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
@runtime_checkable
|
|
75
|
+
class ClauseClassifier(Protocol):
|
|
76
|
+
"""Text -> {category, description}. The seam a test stubs so the compile is exercised without a model."""
|
|
77
|
+
|
|
78
|
+
def __call__(self, text: str) -> ClauseEnrichment: ...
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def enrichment_prompt(text: str, categories: list[str] = ACORD_CATEGORIES) -> str:
|
|
82
|
+
"""The classify-and-describe prompt (shared by the real classifier and the bench, so they match)."""
|
|
83
|
+
return (
|
|
84
|
+
"You are enriching a contract clause for a knowledge index.\n"
|
|
85
|
+
"1) Classify it into exactly ONE category (or 'None of these' if none fit).\n"
|
|
86
|
+
"2) Write a ONE-LINE description (<= 15 words) that a lawyer could use to tell this clause apart "
|
|
87
|
+
"from others of the same category. Describe what THIS clause specifically says, not the category.\n\n"
|
|
88
|
+
"Categories:\n- " + "\n- ".join(categories) + "\n\n"
|
|
89
|
+
"Clause:\n" + text[:2000]
|
|
90
|
+
)
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
class SeamClassifier:
|
|
94
|
+
"""The real classifier: structured output through the model-profile seam on the OKF_ENRICHMENT role.
|
|
95
|
+
|
|
96
|
+
Retries a few times because a cheap model occasionally emits no valid tool call, which
|
|
97
|
+
`with_structured_output` surfaces as a `None` return (not an exception); a bare `None` is a transient
|
|
98
|
+
miss, so retry, and only raise when it persists (the caller skips a persistent failure and re-gates it).
|
|
99
|
+
"""
|
|
100
|
+
|
|
101
|
+
def __init__(
|
|
102
|
+
self, model_id: Optional[str] = None, categories: list[str] = ACORD_CATEGORIES, *, retries: int = 3
|
|
103
|
+
) -> None:
|
|
104
|
+
self._runnable = build_structured(model_id or model_for(ModelRole.OKF_ENRICHMENT), ClauseEnrichment)
|
|
105
|
+
self._categories = categories
|
|
106
|
+
self._retries = retries
|
|
107
|
+
|
|
108
|
+
def __call__(self, text: str) -> ClauseEnrichment:
|
|
109
|
+
prompt = enrichment_prompt(text, self._categories)
|
|
110
|
+
last_error: Exception | None = None
|
|
111
|
+
for _ in range(self._retries):
|
|
112
|
+
try:
|
|
113
|
+
v = self._runnable.invoke(prompt)
|
|
114
|
+
except Exception as e: # noqa: BLE001 - transient provider/parse error; retry
|
|
115
|
+
last_error = e
|
|
116
|
+
continue
|
|
117
|
+
if v is not None:
|
|
118
|
+
return v
|
|
119
|
+
raise last_error or ValueError("no structured output after retries")
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
async def enrich_all(
|
|
123
|
+
texts: dict[str, str],
|
|
124
|
+
classify: ClauseClassifier,
|
|
125
|
+
*,
|
|
126
|
+
concurrency: int = 8,
|
|
127
|
+
cache: Optional[dict[str, EnrichedClause]] = None,
|
|
128
|
+
) -> dict[str, EnrichedClause]:
|
|
129
|
+
"""Enrich every chunk, reusing `cache` (content-hash gated by chunk_id) and classifying only the rest.
|
|
130
|
+
|
|
131
|
+
Concurrent (async + semaphore) over the un-cached clauses. Determinism holds: the returned map is keyed
|
|
132
|
+
by chunk_id regardless of completion order.
|
|
133
|
+
"""
|
|
134
|
+
cache = cache or {}
|
|
135
|
+
todo = {cid: text for cid, text in texts.items() if cid not in cache}
|
|
136
|
+
sem = asyncio.Semaphore(concurrency)
|
|
137
|
+
|
|
138
|
+
async def one(chunk_id: str, text: str) -> tuple[str, Optional[EnrichedClause]]:
|
|
139
|
+
async with sem:
|
|
140
|
+
try:
|
|
141
|
+
v = await asyncio.to_thread(classify, text)
|
|
142
|
+
except Exception: # noqa: BLE001 - a rate-limited/failed call is skipped, not fatal;
|
|
143
|
+
return chunk_id, None # it stays un-cached so a gated re-run retries it
|
|
144
|
+
if v is None: # a classifier that yields no structured output is a skip, not a crash
|
|
145
|
+
return chunk_id, None
|
|
146
|
+
categorized = v.category != AcordCat.NONE
|
|
147
|
+
return chunk_id, EnrichedClause(
|
|
148
|
+
chunk_id=chunk_id,
|
|
149
|
+
category=v.category.value if categorized else "",
|
|
150
|
+
description=v.description,
|
|
151
|
+
categorized=categorized,
|
|
152
|
+
)
|
|
153
|
+
|
|
154
|
+
fresh = {cid: ec for cid, ec in await asyncio.gather(*(one(c, t) for c, t in todo.items())) if ec}
|
|
155
|
+
merged = {**cache, **fresh}
|
|
156
|
+
return {cid: merged[cid] for cid in texts if cid in merged} # excludes clauses that failed
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
def load_enrich_cache(path: Path, recipe_version: str) -> dict[str, EnrichedClause]:
|
|
160
|
+
"""Load the enrichment cache if it matches `recipe_version`; a recipe change invalidates it."""
|
|
161
|
+
if not path.exists():
|
|
162
|
+
return {}
|
|
163
|
+
data = json.loads(path.read_text(encoding="utf-8"))
|
|
164
|
+
if data.get("recipe_version") != recipe_version:
|
|
165
|
+
return {}
|
|
166
|
+
return {cid: EnrichedClause.model_validate(e) for cid, e in data.get("entries", {}).items()}
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
def save_enrich_cache(path: Path, recipe_version: str, entries: dict[str, EnrichedClause]) -> None:
|
|
170
|
+
"""Persist the enrichment cache keyed by recipe_version (the content-hash gate's durable half)."""
|
|
171
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
172
|
+
payload = {
|
|
173
|
+
"recipe_version": recipe_version,
|
|
174
|
+
"entries": {cid: e.model_dump() for cid, e in entries.items()},
|
|
175
|
+
}
|
|
176
|
+
path.write_text(json.dumps(payload, ensure_ascii=False, indent=2), encoding="utf-8")
|