rag-wright 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- rag_wright/__init__.py +13 -0
- rag_wright/api/__init__.py +33 -0
- rag_wright/api/config.py +59 -0
- rag_wright/api/discover.py +70 -0
- rag_wright/api/documents.py +39 -0
- rag_wright/api/ids.py +31 -0
- rag_wright/api/invoke.py +99 -0
- rag_wright/api/kg.py +61 -0
- rag_wright/api/mcp.py +94 -0
- rag_wright/api/usage.py +30 -0
- rag_wright/api/workspace.py +85 -0
- rag_wright/capabilities/__init__.py +8 -0
- rag_wright/capabilities/answer_generator.py +427 -0
- rag_wright/capabilities/ard.py +286 -0
- rag_wright/capabilities/assertion_extraction.py +79 -0
- rag_wright/capabilities/chunk_read.py +58 -0
- rag_wright/capabilities/chunk_write.py +163 -0
- rag_wright/capabilities/claim_extraction.py +153 -0
- rag_wright/capabilities/clause_exception_linking.py +117 -0
- rag_wright/capabilities/compliance_judgment.py +322 -0
- rag_wright/capabilities/compliance_store.py +87 -0
- rag_wright/capabilities/contract_kg_serve.py +156 -0
- rag_wright/capabilities/contract_kg_store.py +251 -0
- rag_wright/capabilities/dg_extraction.py +585 -0
- rag_wright/capabilities/disambiguation.py +163 -0
- rag_wright/capabilities/document_parse.py +87 -0
- rag_wright/capabilities/document_scope.py +49 -0
- rag_wright/capabilities/embedding.py +164 -0
- rag_wright/capabilities/embedding_profiles.py +43 -0
- rag_wright/capabilities/entity_resolution.py +154 -0
- rag_wright/capabilities/fusion.py +64 -0
- rag_wright/capabilities/graph_extraction.py +243 -0
- rag_wright/capabilities/graph_query.py +73 -0
- rag_wright/capabilities/graph_storage.py +111 -0
- rag_wright/capabilities/highlight_serve.py +142 -0
- rag_wright/capabilities/hybrid_search.py +65 -0
- rag_wright/capabilities/invoke.py +31 -0
- rag_wright/capabilities/jev_decision.py +38 -0
- rag_wright/capabilities/manifests.py +872 -0
- rag_wright/capabilities/okf_navigate.py +456 -0
- rag_wright/capabilities/parsing.py +286 -0
- rag_wright/capabilities/property_boosted_retrieval.py +125 -0
- rag_wright/capabilities/query_function_classifier.py +94 -0
- rag_wright/capabilities/query_understanding.py +109 -0
- rag_wright/capabilities/registry.py +262 -0
- rag_wright/capabilities/remote_encoders.py +94 -0
- rag_wright/capabilities/requirement_extraction.py +247 -0
- rag_wright/capabilities/reranking.py +123 -0
- rag_wright/capabilities/retrieval_core.py +126 -0
- rag_wright/capabilities/rlm_chunking.py +808 -0
- rag_wright/capabilities/rlm_synthesis.py +316 -0
- rag_wright/capabilities/scan_quality.py +136 -0
- rag_wright/capabilities/span_relevance_judgment.py +191 -0
- rag_wright/capabilities/vision_to_text.py +85 -0
- rag_wright/capabilities/vlm_ocr.py +85 -0
- rag_wright/contracts/__init__.py +6 -0
- rag_wright/contracts/chunk.py +79 -0
- rag_wright/contracts/compliance.py +303 -0
- rag_wright/contracts/contract_meta.py +27 -0
- rag_wright/contracts/extraction.py +130 -0
- rag_wright/contracts/function.py +167 -0
- rag_wright/contracts/function_routing.py +91 -0
- rag_wright/contracts/highlight.py +74 -0
- rag_wright/contracts/identifiers.py +153 -0
- rag_wright/contracts/jurisdiction.py +96 -0
- rag_wright/contracts/ontology.py +142 -0
- rag_wright/contracts/property.py +201 -0
- rag_wright/contracts/provenance.py +78 -0
- rag_wright/contracts/query_intent.py +53 -0
- rag_wright/contracts/span.py +76 -0
- rag_wright/contracts/value_match.py +84 -0
- rag_wright/corpus/__init__.py +0 -0
- rag_wright/corpus/canonicalize.py +116 -0
- rag_wright/corpus/cuad.py +153 -0
- rag_wright/corpus/cuad_ingestion.py +72 -0
- rag_wright/corpus/document_parser.py +299 -0
- rag_wright/corpus/edgar.py +231 -0
- rag_wright/corpus/gcs_ingestion.py +120 -0
- rag_wright/corpus/http.py +110 -0
- rag_wright/corpus/selection.py +152 -0
- rag_wright/mcp/__init__.py +11 -0
- rag_wright/mcp/compliance_server.py +299 -0
- rag_wright/mcp/intra_document_qa_server.py +170 -0
- rag_wright/mcp/relational_qa_server.py +171 -0
- rag_wright/mcp/session_store.py +64 -0
- rag_wright/mcp/typed_property_retrieval_server.py +191 -0
- rag_wright/models/__init__.py +8 -0
- rag_wright/models/profiles.py +331 -0
- rag_wright/models/seam.py +497 -0
- rag_wright/models/tag_structured.py +285 -0
- rag_wright/models/tracing.py +179 -0
- rag_wright/models/usage.py +102 -0
- rag_wright/okf/__init__.py +11 -0
- rag_wright/okf/compile.py +292 -0
- rag_wright/okf/document.py +47 -0
- rag_wright/okf/enrich.py +176 -0
- rag_wright/okf/links.py +190 -0
- rag_wright/okf/lint.py +105 -0
- rag_wright/ontology/__init__.py +6 -0
- rag_wright/ontology/_generated_template_meta.py +60 -0
- rag_wright/ontology/_generated_vocab.py +52 -0
- rag_wright/ontology/clause_template.py +964 -0
- rag_wright/ontology/codegen.py +84 -0
- rag_wright/ontology/compliance_bridge.ttl +186 -0
- rag_wright/ontology/contract_bridge.ttl +2685 -0
- rag_wright/ontology/contract_taxonomy.py +24 -0
- rag_wright/ontology/derive.py +58 -0
- rag_wright/ontology/loader.py +435 -0
- rag_wright/ontology/packs/ftc_16cfr255.ttl +29 -0
- rag_wright/ontology/registry.py +87 -0
- rag_wright/ontology/template_introspect.py +100 -0
- rag_wright/py.typed +0 -0
- rag_wright/reference/__init__.py +2 -0
- rag_wright/reference/compliance.py +41 -0
- rag_wright/reference/contract_seam.py +123 -0
- rag_wright/skills/__init__.py +7 -0
- rag_wright/skills/claim_extraction/SKILL.md +47 -0
- rag_wright/skills/claim_extraction/__init__.py +1 -0
- rag_wright/skills/claim_extraction/template.py +50 -0
- rag_wright/skills/compliance_judgment/SKILL.md +59 -0
- rag_wright/skills/corpus_ingest/SKILL.md +106 -0
- rag_wright/skills/extraction_semantic_judge/SKILL.md +51 -0
- rag_wright/skills/extraction_semantic_judge/__init__.py +1 -0
- rag_wright/skills/generation/SKILL.md +64 -0
- rag_wright/skills/generation/__init__.py +1 -0
- rag_wright/skills/generic_compliance_judgment/SKILL.md +58 -0
- rag_wright/skills/okf_navigate/SKILL.md +137 -0
- rag_wright/skills/requirement_extraction/SKILL.md +47 -0
- rag_wright/skills/requirement_extraction/__init__.py +1 -0
- rag_wright/skills/requirement_extraction/template.py +50 -0
- rag_wright/skills/rlm/SKILL.md +186 -0
- rag_wright/skills/rlm/__init__.py +31 -0
- rag_wright/skills/rlm/agent.py +292 -0
- rag_wright/skills/span_relevance_judgment/SKILL.md +67 -0
- rag_wright/skills/vision_to_text/SKILL.md +36 -0
- rag_wright/skills/vision_to_text/__init__.py +1 -0
- rag_wright/spans/__init__.py +1 -0
- rag_wright/spans/boundary.py +78 -0
- rag_wright/spans/clause_function_classifier.py +490 -0
- rag_wright/spans/clause_kg_extractor.py +337 -0
- rag_wright/spans/cuad_labels.py +81 -0
- rag_wright/spans/dim_classifier.py +158 -0
- rag_wright/spans/dim_fleet.json +411 -0
- rag_wright/spans/function_classifier.py +77 -0
- rag_wright/spans/function_families.py +62 -0
- rag_wright/spans/hybrid_classifier.py +103 -0
- rag_wright/spans/legalbert_classifier.py +83 -0
- rag_wright/spans/model_capabilities.py +107 -0
- rag_wright/spans/new_function_labels.py +111 -0
- rag_wright/spans/page_map.py +68 -0
- rag_wright/spans/property_extractor.py +365 -0
- rag_wright/spans/property_grounding.py +182 -0
- rag_wright/spans/reclassify.py +77 -0
- rag_wright/spans/scarce_function_labels.py +105 -0
- rag_wright/spans/segment.py +341 -0
- rag_wright/spans/semantic_judge.py +197 -0
- rag_wright/spans/symbolic_validation.py +131 -0
- rag_wright/spans/tag_clause_extractor.py +182 -0
- rag_wright/store/__init__.py +6 -0
- rag_wright/store/arcadedb.py +1135 -0
- rag_wright/store/chunk_text.py +66 -0
- rag_wright/store/seam.py +213 -0
- rag_wright/subgraphs/__init__.py +0 -0
- rag_wright/subgraphs/async_ingestion.py +204 -0
- rag_wright/subgraphs/compliance_check.py +1042 -0
- rag_wright/subgraphs/compliance_ingestion.py +306 -0
- rag_wright/subgraphs/contract_ingestion_pipeline.py +999 -0
- rag_wright/subgraphs/graph_extraction.py +102 -0
- rag_wright/subgraphs/intra_document_qa.py +328 -0
- rag_wright/subgraphs/observability.py +140 -0
- rag_wright/subgraphs/query_constraint_extraction.py +73 -0
- rag_wright/subgraphs/relational_qa.py +165 -0
- rag_wright/subgraphs/requirement_extraction.py +137 -0
- rag_wright/subgraphs/scaffold.py +65 -0
- rag_wright/subgraphs/semantic_chunking.py +183 -0
- rag_wright/subgraphs/typed_clause_extraction.py +172 -0
- rag_wright/subgraphs/typed_property_retrieval.py +278 -0
- rag_wright/util/__init__.py +1 -0
- rag_wright/util/concurrent.py +153 -0
- rag_wright/util/spacy_model.py +45 -0
- rag_wright-0.1.0.dist-info/METADATA +168 -0
- rag_wright-0.1.0.dist-info/RECORD +184 -0
- rag_wright-0.1.0.dist-info/WHEEL +4 -0
- rag_wright-0.1.0.dist-info/licenses/LICENSE +21 -0
rag_wright/okf/links.py
ADDED
|
@@ -0,0 +1,190 @@
|
|
|
1
|
+
"""T49 (FR-K.3): embedding-free cross-linking of the OKF bundle from shared distinctive terms.
|
|
2
|
+
|
|
3
|
+
Writes standard markdown links (absolute from the bundle root) between related clauses, so a traversal can
|
|
4
|
+
move laterally, not only down the category tree. Edges are derived from MEASURED structure, not guessed, and
|
|
5
|
+
kept embedding-free (Option 1, 2026-07-22): two clauses are related when their bodies share distinctive (rare,
|
|
6
|
+
legal) terms; the embedding-kNN structure T41 measured is the held fallback if this does not lift the ceiling.
|
|
7
|
+
|
|
8
|
+
Density is bounded two ways so the graph does not degenerate into near-complete connectivity: an ultra-common
|
|
9
|
+
term (appearing in more than `MAX_TERM_DF` clauses) carries no signal and is skipped, and each clause keeps at
|
|
10
|
+
most `max_degree` neighbours, ranked by shared-term count. The pass is idempotent and runs INDEPENDENTLY of the
|
|
11
|
+
compile (it rewrites only each concept's `## Related clauses` section), so a link recipe can be re-run or
|
|
12
|
+
ablated without a full recompile. A concept's clause text stays the leading body verbatim; links are appended.
|
|
13
|
+
|
|
14
|
+
Run: uv run python -m rag_wright.okf.links
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
import re
|
|
20
|
+
from collections import defaultdict
|
|
21
|
+
from pathlib import Path
|
|
22
|
+
|
|
23
|
+
from pydantic import BaseModel
|
|
24
|
+
|
|
25
|
+
from rag_wright.okf.document import parse_okf, serialize_okf
|
|
26
|
+
|
|
27
|
+
LINKS_RECIPE_VERSION = "okf-links-lex-v1"
|
|
28
|
+
DEFAULT_MAX_DEGREE = 8
|
|
29
|
+
DEFAULT_MIN_SHARED = 2 # a link needs >=2 shared distinctive terms (one rare word alone is not relatedness)
|
|
30
|
+
MAX_TERM_DF = 100 # a distinctive term in more than this many clauses is boilerplate -> no signal, skipped
|
|
31
|
+
_DISTINCTIVE_LEN = 8 # only long/rare tokens count as relatedness signal
|
|
32
|
+
_STEM = 6
|
|
33
|
+
_RELATED_HEADING = "## Related clauses"
|
|
34
|
+
_INDEX = "index.md"
|
|
35
|
+
_MANIFEST = ".okf_links.json"
|
|
36
|
+
_LINK = re.compile(r"\[[^\]]*\]\(([^)]+)\)")
|
|
37
|
+
|
|
38
|
+
_STOPWORDS = {
|
|
39
|
+
"agreement", "including", "pursuant", "provided", "otherwise", "applicable", "hereunder", "represents",
|
|
40
|
+
"respective", "obligations", "termination", "notwithstanding", "thereunder", "thereafter", "hereinafter",
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
class LinksManifest(BaseModel):
|
|
45
|
+
"""What the cross-link pass wrote: recipe params + edge-graph shape, stamped for attributability."""
|
|
46
|
+
|
|
47
|
+
recipe_version: str
|
|
48
|
+
max_degree: int
|
|
49
|
+
min_shared: int
|
|
50
|
+
n_concepts: int
|
|
51
|
+
n_edges: int # directed edges written (sum of out-degrees)
|
|
52
|
+
mean_degree: float
|
|
53
|
+
isolated: int # concepts with no related link
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def _terms(text: str) -> set[str]:
|
|
57
|
+
"""Distinctive-term stems of a clause body: long, non-boilerplate tokens reduced to a prefix stem."""
|
|
58
|
+
toks = [t for t in re.split(r"[^a-z0-9]+", text.lower()) if len(t) >= _DISTINCTIVE_LEN and t not in _STOPWORDS]
|
|
59
|
+
return {t[:_STEM] for t in toks}
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
class _Concept(BaseModel):
|
|
63
|
+
chunk_id: str
|
|
64
|
+
rel_path: str # bundle-root-relative, e.g. "indemnification/abc.md"
|
|
65
|
+
title: str
|
|
66
|
+
description: str
|
|
67
|
+
terms: set[str]
|
|
68
|
+
|
|
69
|
+
model_config = {"arbitrary_types_allowed": True}
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def _clause_body(body: str) -> str:
|
|
73
|
+
"""The clause text (no trailing newlines) with any prior `## Related clauses` section stripped.
|
|
74
|
+
|
|
75
|
+
Returned without a trailing newline so re-linking is byte-idempotent regardless of whether a prior
|
|
76
|
+
section was present.
|
|
77
|
+
"""
|
|
78
|
+
idx = body.find(_RELATED_HEADING)
|
|
79
|
+
return (body[:idx] if idx != -1 else body).rstrip("\n")
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def _load_concepts(bundle_root: Path) -> list[_Concept]:
|
|
83
|
+
concepts: list[_Concept] = []
|
|
84
|
+
for md in sorted(bundle_root.rglob("*.md")):
|
|
85
|
+
if md.name in (_INDEX, "log.md"):
|
|
86
|
+
continue
|
|
87
|
+
fm, body = parse_okf(md.read_text(encoding="utf-8"))
|
|
88
|
+
chunk_id = fm.get("chunk_id")
|
|
89
|
+
if not chunk_id:
|
|
90
|
+
continue
|
|
91
|
+
concepts.append(
|
|
92
|
+
_Concept(
|
|
93
|
+
chunk_id=str(chunk_id),
|
|
94
|
+
rel_path=str(md.relative_to(bundle_root)),
|
|
95
|
+
title=str(fm.get("title") or md.stem),
|
|
96
|
+
description=str(fm.get("description") or ""),
|
|
97
|
+
terms=_terms(_clause_body(body)),
|
|
98
|
+
)
|
|
99
|
+
)
|
|
100
|
+
return concepts
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def build_edges(
|
|
104
|
+
concepts: list[_Concept], *, max_degree: int = DEFAULT_MAX_DEGREE, min_shared: int = DEFAULT_MIN_SHARED
|
|
105
|
+
) -> dict[str, list[str]]:
|
|
106
|
+
"""chunk_id -> up to `max_degree` related chunk_ids (>= min_shared shared distinctive terms), by count."""
|
|
107
|
+
term_index: dict[str, list[str]] = defaultdict(list)
|
|
108
|
+
for c in concepts:
|
|
109
|
+
for t in c.terms:
|
|
110
|
+
term_index[t].append(c.chunk_id)
|
|
111
|
+
|
|
112
|
+
shared: dict[str, dict[str, int]] = defaultdict(lambda: defaultdict(int))
|
|
113
|
+
for ids in term_index.values():
|
|
114
|
+
if len(ids) > MAX_TERM_DF: # boilerplate term -> no relatedness signal, and bounds the pair blowup
|
|
115
|
+
continue
|
|
116
|
+
for i in range(len(ids)):
|
|
117
|
+
for j in range(i + 1, len(ids)):
|
|
118
|
+
a, b = ids[i], ids[j]
|
|
119
|
+
shared[a][b] += 1
|
|
120
|
+
shared[b][a] += 1
|
|
121
|
+
|
|
122
|
+
edges: dict[str, list[str]] = {}
|
|
123
|
+
for c in concepts:
|
|
124
|
+
ranked = sorted(
|
|
125
|
+
((nbr, cnt) for nbr, cnt in shared.get(c.chunk_id, {}).items() if cnt >= min_shared),
|
|
126
|
+
key=lambda nc: (-nc[1], nc[0]),
|
|
127
|
+
)
|
|
128
|
+
edges[c.chunk_id] = [nbr for nbr, _ in ranked[:max_degree]]
|
|
129
|
+
return edges
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def _related_section(neighbours: list[str], by_id: dict[str, _Concept]) -> str:
|
|
133
|
+
lines = [_RELATED_HEADING, ""]
|
|
134
|
+
for nbr in neighbours:
|
|
135
|
+
c = by_id[nbr]
|
|
136
|
+
suffix = f" - {c.description}" if c.description else ""
|
|
137
|
+
lines.append(f"* [{c.title}](/{c.rel_path}){suffix}") # absolute from bundle root, per OKF §5
|
|
138
|
+
return "\n".join(lines) + "\n"
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
def apply_links(
|
|
142
|
+
bundle_root: Path,
|
|
143
|
+
*,
|
|
144
|
+
max_degree: int = DEFAULT_MAX_DEGREE,
|
|
145
|
+
min_shared: int = DEFAULT_MIN_SHARED,
|
|
146
|
+
recipe_version: str = LINKS_RECIPE_VERSION,
|
|
147
|
+
) -> LinksManifest:
|
|
148
|
+
"""Build edges and write each concept's `## Related clauses` section. Idempotent; independent of compile."""
|
|
149
|
+
bundle_root = Path(bundle_root)
|
|
150
|
+
concepts = _load_concepts(bundle_root)
|
|
151
|
+
by_id = {c.chunk_id: c for c in concepts}
|
|
152
|
+
edges = build_edges(concepts, max_degree=max_degree, min_shared=min_shared)
|
|
153
|
+
|
|
154
|
+
n_edges = 0
|
|
155
|
+
isolated = 0
|
|
156
|
+
for c in concepts:
|
|
157
|
+
neighbours = edges.get(c.chunk_id, [])
|
|
158
|
+
n_edges += len(neighbours)
|
|
159
|
+
isolated += 1 if not neighbours else 0
|
|
160
|
+
path = bundle_root / c.rel_path
|
|
161
|
+
fm, body = parse_okf(path.read_text(encoding="utf-8"))
|
|
162
|
+
clause = _clause_body(body) # no trailing newline
|
|
163
|
+
new_body = f"{clause}\n\n{_related_section(neighbours, by_id)}" if neighbours else f"{clause}\n"
|
|
164
|
+
path.write_text(serialize_okf(fm, new_body), encoding="utf-8")
|
|
165
|
+
|
|
166
|
+
manifest = LinksManifest(
|
|
167
|
+
recipe_version=recipe_version,
|
|
168
|
+
max_degree=max_degree,
|
|
169
|
+
min_shared=min_shared,
|
|
170
|
+
n_concepts=len(concepts),
|
|
171
|
+
n_edges=n_edges,
|
|
172
|
+
mean_degree=(n_edges / len(concepts)) if concepts else 0.0,
|
|
173
|
+
isolated=isolated,
|
|
174
|
+
)
|
|
175
|
+
(bundle_root / _MANIFEST).write_text(manifest.model_dump_json(indent=2), encoding="utf-8")
|
|
176
|
+
return manifest
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
def _default_bundle() -> Path:
|
|
180
|
+
return Path("data/acord/okf/bundle")
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
def main() -> None:
|
|
184
|
+
manifest = apply_links(_default_bundle())
|
|
185
|
+
print(f"[okf_links] recipe={manifest.recipe_version} concepts={manifest.n_concepts} "
|
|
186
|
+
f"edges={manifest.n_edges} mean_degree={manifest.mean_degree:.2f} isolated={manifest.isolated}")
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
if __name__ == "__main__":
|
|
190
|
+
main()
|
rag_wright/okf/lint.py
ADDED
|
@@ -0,0 +1,105 @@
|
|
|
1
|
+
"""OKF conformance linter (FR-K.4, T46): parseable frontmatter, non-empty type, resolvable links.
|
|
2
|
+
|
|
3
|
+
Index and description quality is the retrieval ceiling of the whole embedding-free path, so the linter is a
|
|
4
|
+
quality gate, not a formality: it reports orphan rate, description coverage, and broken-link ratio as numbers,
|
|
5
|
+
not pass-or-fail alone. `passes` covers only the hard OKF v0.1 conformance rules (every concept has parseable
|
|
6
|
+
frontmatter with a non-empty `type`); the rates are quality signals the compile recipe is tuned against.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import re
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
|
|
14
|
+
from pydantic import BaseModel
|
|
15
|
+
|
|
16
|
+
from rag_wright.okf.document import OkfParseError, parse_okf
|
|
17
|
+
|
|
18
|
+
_INDEX_NAME = "index.md"
|
|
19
|
+
_RESERVED = {"index.md", "log.md"}
|
|
20
|
+
_LINK = re.compile(r"\[[^\]]*\]\(([^)]+)\)")
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class LintReport(BaseModel):
|
|
24
|
+
"""Bundle conformance + quality numbers. `passes` is the hard OKF v0.1 conformance verdict."""
|
|
25
|
+
|
|
26
|
+
total_concepts: int
|
|
27
|
+
frontmatter_parseable: int
|
|
28
|
+
type_non_empty: int
|
|
29
|
+
total_index_entries: int
|
|
30
|
+
description_coverage: float # fraction of index entries carrying a description
|
|
31
|
+
orphan_rate: float # fraction of concept files not linked from any index.md
|
|
32
|
+
broken_link_ratio: float # fraction of intra-bundle links whose target file is absent
|
|
33
|
+
passes: bool
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def _concept_files(root: Path) -> list[Path]:
|
|
37
|
+
return [p for p in root.rglob("*.md") if p.name not in _RESERVED]
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def _resolve(link: str, source: Path, root: Path) -> Path | None:
|
|
41
|
+
"""Resolve an intra-bundle markdown link to a path, or None if it is external (a URL / anchor)."""
|
|
42
|
+
target = link.split("#", 1)[0].strip()
|
|
43
|
+
if not target or "://" in target:
|
|
44
|
+
return None
|
|
45
|
+
if target.startswith("/"):
|
|
46
|
+
return root / target.lstrip("/")
|
|
47
|
+
return (source.parent / target).resolve()
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def lint_bundle(root: Path) -> LintReport:
|
|
51
|
+
"""Lint an OKF bundle, returning conformance + quality numbers."""
|
|
52
|
+
root = Path(root)
|
|
53
|
+
concepts = _concept_files(root)
|
|
54
|
+
|
|
55
|
+
parseable = type_ok = 0
|
|
56
|
+
for path in concepts:
|
|
57
|
+
try:
|
|
58
|
+
fm, _ = parse_okf(path.read_text(encoding="utf-8"))
|
|
59
|
+
except OkfParseError:
|
|
60
|
+
continue
|
|
61
|
+
parseable += 1
|
|
62
|
+
if str(fm.get("type") or "").strip():
|
|
63
|
+
type_ok += 1
|
|
64
|
+
|
|
65
|
+
# Broken-link ratio is over ALL links (index entries AND concept-body cross-links); orphan_rate is over
|
|
66
|
+
# index links only, since an index entry is the progressive-disclosure entry point from the root.
|
|
67
|
+
entries_total = entries_with_desc = 0
|
|
68
|
+
links_total = links_broken = 0
|
|
69
|
+
index_linked_targets: set[Path] = set()
|
|
70
|
+
for md in root.rglob("*.md"):
|
|
71
|
+
content = md.read_text(encoding="utf-8")
|
|
72
|
+
is_index = md.name == _INDEX_NAME
|
|
73
|
+
if is_index:
|
|
74
|
+
for raw in content.splitlines():
|
|
75
|
+
line = raw.strip()
|
|
76
|
+
if not line.startswith("*"):
|
|
77
|
+
continue
|
|
78
|
+
if not _LINK.search(line):
|
|
79
|
+
continue
|
|
80
|
+
entries_total += 1
|
|
81
|
+
if " - " in line.split(")", 1)[-1]:
|
|
82
|
+
entries_with_desc += 1
|
|
83
|
+
for m in _LINK.finditer(content):
|
|
84
|
+
resolved = _resolve(m.group(1), md, root)
|
|
85
|
+
if resolved is None:
|
|
86
|
+
continue
|
|
87
|
+
links_total += 1
|
|
88
|
+
if resolved.exists():
|
|
89
|
+
if is_index:
|
|
90
|
+
index_linked_targets.add(resolved.resolve())
|
|
91
|
+
else:
|
|
92
|
+
links_broken += 1
|
|
93
|
+
|
|
94
|
+
orphans = sum(1 for c in concepts if c.resolve() not in index_linked_targets)
|
|
95
|
+
n = len(concepts) or 1
|
|
96
|
+
return LintReport(
|
|
97
|
+
total_concepts=len(concepts),
|
|
98
|
+
frontmatter_parseable=parseable,
|
|
99
|
+
type_non_empty=type_ok,
|
|
100
|
+
total_index_entries=entries_total,
|
|
101
|
+
description_coverage=(entries_with_desc / entries_total) if entries_total else 1.0,
|
|
102
|
+
orphan_rate=orphans / n,
|
|
103
|
+
broken_link_ratio=(links_broken / links_total) if links_total else 0.0,
|
|
104
|
+
passes=parseable == len(concepts) and type_ok == len(concepts),
|
|
105
|
+
)
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
"""Ontology and entity-registry derivation (FR-C.8).
|
|
2
|
+
|
|
3
|
+
Pulls the entity and relationship types (as Pydantic models) and the populated entity
|
|
4
|
+
registry from the Data Catalog (Nessie or equivalent). Where the catalog does not exist,
|
|
5
|
+
extraction degrades to the lightweight path plus an open-ended language model (assumption 3).
|
|
6
|
+
"""
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
"""GENERATED FROM contract_bridge.ttl by scripts/generate_contract_python.py -- DO NOT EDIT BY HAND.
|
|
2
|
+
|
|
3
|
+
ADR-0066 P1b-2 (Rule 3): the extraction template's KNOWLEDGE -- each field's LOOK-FOR description and examples,
|
|
4
|
+
generated from the ttl. `clause_template.py` CONSUMES this (via `_d()` / `_ex()`); the mechanism (validators,
|
|
5
|
+
normalizers, __str__, config) stays clean hand-code there. To change a description/example, edit the ttl and
|
|
6
|
+
re-run the generator; a hand-edit here is caught by tests/ontology/test_generated_template_meta_in_sync.py."""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
DESCRIPTIONS: dict[str, str] = {
|
|
12
|
+
'Clause.document_reference': "OPTIONAL. ONLY the clause's own section number or short heading if one is written in the text (e.g. 'Section 8', '8.1 Limitation of Liability'). If there is NO explicit section number or heading, leave this null/empty -- do NOT invent one and NEVER quote the clause body or any sentence of it here.",
|
|
13
|
+
'Clause.audit_frequency': "How often an audit-rights clause permits audits, a short value ONLY, e.g. 'annual', 'quarterly', 'once per year'.",
|
|
14
|
+
'Clause.clause_type': "The clause function/type LABEL ONLY (a few words), e.g. 'Cap on Liability', 'Governing Law', 'Non-Solicit of Employees'.",
|
|
15
|
+
'Clause.collateral_type': 'The collateral / assets a SECURITY-INTEREST clause attaches to (may be several), e.g. inventory, equipment, accounts receivable, all assets.',
|
|
16
|
+
'Clause.commitment_quantum': "The minimum-commitment / volume amount ONLY (a short value), e.g. '$1,000,000', '100 units/year'.",
|
|
17
|
+
'Clause.condition_type': 'The kind of condition a CONDITION-PRECEDENT clause requires: regulatory approval / financing / third-party consent / due diligence / board approval / no material adverse change / court approval / closing condition.',
|
|
18
|
+
'Clause.confidentiality_exception': 'Permitted disclosures / exceptions to a CONFIDENTIALITY obligation (may be several): required by law, publicly available, independently developed, prior possession, received from a third party.',
|
|
19
|
+
'Clause.force_majeure_event': 'Events a FORCE-MAJEURE clause lists as excusing performance (may be several): act of god, war, pandemic, government action, labor dispute, supply failure, natural disaster.',
|
|
20
|
+
'Clause.royalty_basis': 'How a ROYALTY is calculated: percentage of net sales / percentage of gross sales / per unit / fixed / tiered.',
|
|
21
|
+
'Clause.covers': 'Subjects the clause covers (ODRL target; may be several).',
|
|
22
|
+
'Clause.covers_party_scope': 'Which affiliated parties the clause extends to: affiliates (either side) / licensor_affiliates / licensee_affiliates.',
|
|
23
|
+
'Clause.dispute_method': 'How a DISPUTE-RESOLUTION clause resolves disputes: arbitration / litigation (courts) / mediation / expert determination / negotiation.',
|
|
24
|
+
'Clause.excepts': 'Carve-outs / exceptions the clause lists (may be several).',
|
|
25
|
+
'Clause.has_assignment_consent': 'How an anti-assignment clause treats consent (required / notice-only / free).',
|
|
26
|
+
'Clause.has_asymmetry': "Whether the clause's terms apply the same to both parties (symmetric) or differently per party (different_per_party).",
|
|
27
|
+
'Clause.has_claim_scope': "The scope of claims covered: first_party (the parties' own claims) / third_party (third-party claims) / broad_based (both / any).",
|
|
28
|
+
'Clause.has_coc_consent': 'How a change-of-control clause treats consent (required / notice-only / unrestricted).',
|
|
29
|
+
'Clause.has_escrow_release_trigger': 'What triggers a source-code escrow release (bankruptcy / breach / discontinuance).',
|
|
30
|
+
'Clause.has_exclusivity_type': 'The exclusivity a licensing/distribution clause grants (exclusive / sole / non-exclusive).',
|
|
31
|
+
'Clause.has_favorability': 'Which side the clause favors: buyer_favorable (the buyer / customer) or seller_favorable (the seller / vendor).',
|
|
32
|
+
'Clause.has_ip_ownership': 'How intellectual-property ownership is treated: assigned (transferred) / joint (shared) / retained (kept by the originating party).',
|
|
33
|
+
'Clause.has_mfn_scope': 'What a most-favored-nation clause covers (price / terms / both).',
|
|
34
|
+
'Clause.has_mutuality': "Whether the clause's obligation runs both ways (mutual) or one way (unilateral).",
|
|
35
|
+
'Clause.has_renewal': 'How the term renews: auto (automatic renewal) or requires_notice (renews only on notice / an affirmative election).',
|
|
36
|
+
'Clause.has_restriction_scope': 'What a non-compete restricts: geographic area, activity, or both.',
|
|
37
|
+
'Clause.has_right_of_first_type': 'The first-refusal/offer/negotiation right the clause grants (ROFR / ROFO / ROFN).',
|
|
38
|
+
'Clause.has_termination_right': 'Who may terminate for convenience (either party / one party).',
|
|
39
|
+
'Clause.has_warranty_scope': "Which warranties the clause disclaims or limits: 'implied' if it disclaims IMPLIED warranties (look for 'implied', 'merchantability', 'fitness for a particular purpose', 'in lieu of all other warranties/conditions') / 'express' if it excludes or limits EXPRESS warranties (look for 'no express warranty', 'except for the express warranties', 'express provisions ... in place of') / 'as_is' if goods or services are provided 'as is' or 'with all faults' / 'non_reliance' for a non-reliance disclaimer. A disclaimer that mentions merchantability or fitness for purpose is 'implied' (not 'as_is').",
|
|
40
|
+
'Clause.ld_trigger': "What triggers liquidated damages, e.g. 'late delivery', 'early termination'.",
|
|
41
|
+
'Clause.prohibits_damage': 'Damage types the clause waives/excludes (may be several).',
|
|
42
|
+
'Clause.prohibits_solicit': 'Whom the clause forbids soliciting (employees / customers).',
|
|
43
|
+
'Clause.requires_duty': 'A procedural duty the clause imposes (duty to defend / control of defense).',
|
|
44
|
+
'Clause.bounded_by': 'A temporal bound (term or notice period), if any.',
|
|
45
|
+
'Clause.caps': "The liability cap, if the clause limits the AMOUNT of liability (e.g. 'liability shall not exceed', 'limited to the fees paid', 'in no event shall ... exceed', 'total/aggregate liability ... capped at'). Fill its basis and quantum. Leave absent if the clause states no monetary limit (a pure exclusion or waiver of a damage TYPE is not a cap).",
|
|
46
|
+
'Clause.governed_by': 'The governing-law jurisdiction, if the clause states one.',
|
|
47
|
+
'CapConstraint.cap_basis': 'The basis of the liability cap: fixed_fee (a fixed amount) / multiple_of_fees (a multiple of fees paid) / other.',
|
|
48
|
+
'CapConstraint.cap_operator': "The comparison operator ONLY (a few words), e.g. 'lteq' (at most), 'eq' (fixed).",
|
|
49
|
+
'CapConstraint.cap_quantum': "The stated limit on liability. Look for 'shall not exceed', 'limited to', 'capped at', 'in no event ... exceed', 'total/aggregate liability ... shall not exceed'. Copy the limit VERBATIM even if it is a phrase, e.g. '$1,000,000', '12 months of fees', 'the fees paid in the prior 12 months', 'the greater of X or Y'.",
|
|
50
|
+
'TemporalConstraint.temporal_duration': "The duration ONLY (a short value), e.g. '12_months', '30_days', 'unbounded'.",
|
|
51
|
+
'TemporalConstraint.temporal_kind': "What is bounded: 'term' (temporal_bound) or 'notice_period'.",
|
|
52
|
+
'TemporalConstraint.temporal_operator': "The temporal comparison operator ONLY (a few words), e.g. 'lteq' (within / at most), 'gteq' (at least), 'eq' (exactly).",
|
|
53
|
+
'Jurisdiction.jurisdiction_name': "The jurisdiction NAME ONLY (a few words, not a sentence), e.g. 'New York', 'England and Wales', 'Delaware'.",
|
|
54
|
+
'Jurisdiction.law_multiplicity': 'Whether one law governs (single) or multiple laws govern (multiple).',
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
EXAMPLES: dict[str, tuple[str, ...]] = {
|
|
58
|
+
'Clause.document_reference': ('Section 8', '8.1 Limitation of Liability', 'Governing Law'),
|
|
59
|
+
'Jurisdiction.jurisdiction_name': ('New York', 'England and Wales', 'Delaware'),
|
|
60
|
+
}
|
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
"""GENERATED FROM contract_bridge.ttl by scripts/generate_contract_python.py -- DO NOT EDIT BY HAND.
|
|
2
|
+
|
|
3
|
+
ADR-0066: the ontology `.ttl` is the single source of truth. To change the closed vocabulary, edit the ttl and
|
|
4
|
+
re-run the generator; a hand-edit here (or a stale regeneration) is caught by
|
|
5
|
+
tests/ontology/test_generated_vocab_in_sync.py."""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
VOCAB: dict[str, frozenset[str]] = {
|
|
11
|
+
'assignment_consent': frozenset({'consent_required', 'free', 'notice_only'}),
|
|
12
|
+
'cap_basis': frozenset({'fixed_fee', 'multiple_of_fees', 'other'}),
|
|
13
|
+
'carve_out': frozenset({'applicable_law', 'bodily_injury', 'confidentiality', 'fraud', 'gross_negligence', 'indemnification', 'third_party_ip_infringement', 'willful_misconduct'}),
|
|
14
|
+
'claim_scope': frozenset({'broad_based', 'first_party', 'third_party'}),
|
|
15
|
+
'coc_consent': frozenset({'consent_required', 'notice_only', 'unrestricted'}),
|
|
16
|
+
'collateral_type': frozenset({'accounts_receivable', 'all_assets', 'deposit_accounts', 'equipment', 'fixtures', 'general_intangibles', 'inventory', 'ip', 'real_property'}),
|
|
17
|
+
'condition_type': frozenset({'board_approval', 'closing_condition', 'court_approval', 'due_diligence', 'financing', 'no_material_adverse_change', 'regulatory_approval', 'third_party_consent'}),
|
|
18
|
+
'confidentiality_exception': frozenset({'independently_developed', 'prior_possession', 'publicly_available', 'required_by_law', 'third_party_source'}),
|
|
19
|
+
'covered_parties': frozenset({'affiliates', 'licensee_affiliates', 'licensor_affiliates'}),
|
|
20
|
+
'covered_subject': frozenset({'copyright', 'fraud', 'gross_negligence', 'ip_infringement', 'trademark', 'violation_of_law'}),
|
|
21
|
+
'damage_type': frozenset({'consequential', 'incidental', 'indirect', 'punitive', 'special'}),
|
|
22
|
+
'dispute_method': frozenset({'arbitration', 'expert_determination', 'litigation', 'mediation', 'negotiation'}),
|
|
23
|
+
'escrow_release_trigger': frozenset({'bankruptcy', 'breach', 'discontinuance'}),
|
|
24
|
+
'exclusivity_type': frozenset({'exclusive', 'non_exclusive', 'sole'}),
|
|
25
|
+
'favorability': frozenset({'buyer_favorable', 'seller_favorable'}),
|
|
26
|
+
'force_majeure_event': frozenset({'act_of_god', 'government_action', 'labor_dispute', 'natural_disaster', 'pandemic', 'supply_failure', 'war'}),
|
|
27
|
+
'ip_ownership': frozenset({'assigned', 'joint', 'retained'}),
|
|
28
|
+
'law_multiplicity': frozenset({'multiple', 'single'}),
|
|
29
|
+
'mfn_scope': frozenset({'price', 'price_and_terms', 'terms'}),
|
|
30
|
+
'mutuality': frozenset({'mutual', 'unilateral'}),
|
|
31
|
+
'nonsolicit_target': frozenset({'customers', 'employees'}),
|
|
32
|
+
'party_asymmetry': frozenset({'different_per_party', 'symmetric'}),
|
|
33
|
+
'procedural': frozenset({'control_of_defense', 'duty_to_defend'}),
|
|
34
|
+
'renewal_mechanism': frozenset({'auto', 'requires_notice'}),
|
|
35
|
+
'restriction_scope': frozenset({'activity', 'geographic', 'geographic_and_activity'}),
|
|
36
|
+
'right_of_first_type': frozenset({'rofn', 'rofo', 'rofr'}),
|
|
37
|
+
'royalty_basis': frozenset({'fixed', 'pct_gross_sales', 'pct_net_sales', 'per_unit', 'tiered'}),
|
|
38
|
+
'termination_right': frozenset({'either_party', 'one_party'}),
|
|
39
|
+
'warranty_scope': frozenset({'as_is', 'express', 'implied', 'non_reliance'}),
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
VALUE_SYNONYMS: dict[str, dict[str, str]] = {
|
|
43
|
+
'damage_type': {'businessinterruption': 'consequential', 'costofcover': 'consequential', 'lossofdata': 'consequential', 'lossofprofit': 'consequential', 'lossofprofits': 'consequential', 'lossofrevenue': 'consequential', 'lossofsavings': 'consequential', 'lossofuse': 'consequential', 'lostbusinessrevenue': 'consequential', 'lostprofits': 'consequential', 'lostrevenue': 'consequential', 'lostsavings': 'consequential'},
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
AFFILIATE_OF = 'Affiliate Of'
|
|
47
|
+
CONTRACTS_WITH = 'Contracts With'
|
|
48
|
+
ORGANIZATION = 'Organization'
|
|
49
|
+
PERSON = 'Person'
|
|
50
|
+
|
|
51
|
+
ENTITY_TYPES: frozenset[str] = frozenset({'Organization', 'Person'})
|
|
52
|
+
RELATIONSHIP_TYPES: frozenset[str] = frozenset({'Affiliate Of', 'Contracts With'})
|