rag-wright 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (184) hide show
  1. rag_wright/__init__.py +13 -0
  2. rag_wright/api/__init__.py +33 -0
  3. rag_wright/api/config.py +59 -0
  4. rag_wright/api/discover.py +70 -0
  5. rag_wright/api/documents.py +39 -0
  6. rag_wright/api/ids.py +31 -0
  7. rag_wright/api/invoke.py +99 -0
  8. rag_wright/api/kg.py +61 -0
  9. rag_wright/api/mcp.py +94 -0
  10. rag_wright/api/usage.py +30 -0
  11. rag_wright/api/workspace.py +85 -0
  12. rag_wright/capabilities/__init__.py +8 -0
  13. rag_wright/capabilities/answer_generator.py +427 -0
  14. rag_wright/capabilities/ard.py +286 -0
  15. rag_wright/capabilities/assertion_extraction.py +79 -0
  16. rag_wright/capabilities/chunk_read.py +58 -0
  17. rag_wright/capabilities/chunk_write.py +163 -0
  18. rag_wright/capabilities/claim_extraction.py +153 -0
  19. rag_wright/capabilities/clause_exception_linking.py +117 -0
  20. rag_wright/capabilities/compliance_judgment.py +322 -0
  21. rag_wright/capabilities/compliance_store.py +87 -0
  22. rag_wright/capabilities/contract_kg_serve.py +156 -0
  23. rag_wright/capabilities/contract_kg_store.py +251 -0
  24. rag_wright/capabilities/dg_extraction.py +585 -0
  25. rag_wright/capabilities/disambiguation.py +163 -0
  26. rag_wright/capabilities/document_parse.py +87 -0
  27. rag_wright/capabilities/document_scope.py +49 -0
  28. rag_wright/capabilities/embedding.py +164 -0
  29. rag_wright/capabilities/embedding_profiles.py +43 -0
  30. rag_wright/capabilities/entity_resolution.py +154 -0
  31. rag_wright/capabilities/fusion.py +64 -0
  32. rag_wright/capabilities/graph_extraction.py +243 -0
  33. rag_wright/capabilities/graph_query.py +73 -0
  34. rag_wright/capabilities/graph_storage.py +111 -0
  35. rag_wright/capabilities/highlight_serve.py +142 -0
  36. rag_wright/capabilities/hybrid_search.py +65 -0
  37. rag_wright/capabilities/invoke.py +31 -0
  38. rag_wright/capabilities/jev_decision.py +38 -0
  39. rag_wright/capabilities/manifests.py +872 -0
  40. rag_wright/capabilities/okf_navigate.py +456 -0
  41. rag_wright/capabilities/parsing.py +286 -0
  42. rag_wright/capabilities/property_boosted_retrieval.py +125 -0
  43. rag_wright/capabilities/query_function_classifier.py +94 -0
  44. rag_wright/capabilities/query_understanding.py +109 -0
  45. rag_wright/capabilities/registry.py +262 -0
  46. rag_wright/capabilities/remote_encoders.py +94 -0
  47. rag_wright/capabilities/requirement_extraction.py +247 -0
  48. rag_wright/capabilities/reranking.py +123 -0
  49. rag_wright/capabilities/retrieval_core.py +126 -0
  50. rag_wright/capabilities/rlm_chunking.py +808 -0
  51. rag_wright/capabilities/rlm_synthesis.py +316 -0
  52. rag_wright/capabilities/scan_quality.py +136 -0
  53. rag_wright/capabilities/span_relevance_judgment.py +191 -0
  54. rag_wright/capabilities/vision_to_text.py +85 -0
  55. rag_wright/capabilities/vlm_ocr.py +85 -0
  56. rag_wright/contracts/__init__.py +6 -0
  57. rag_wright/contracts/chunk.py +79 -0
  58. rag_wright/contracts/compliance.py +303 -0
  59. rag_wright/contracts/contract_meta.py +27 -0
  60. rag_wright/contracts/extraction.py +130 -0
  61. rag_wright/contracts/function.py +167 -0
  62. rag_wright/contracts/function_routing.py +91 -0
  63. rag_wright/contracts/highlight.py +74 -0
  64. rag_wright/contracts/identifiers.py +153 -0
  65. rag_wright/contracts/jurisdiction.py +96 -0
  66. rag_wright/contracts/ontology.py +142 -0
  67. rag_wright/contracts/property.py +201 -0
  68. rag_wright/contracts/provenance.py +78 -0
  69. rag_wright/contracts/query_intent.py +53 -0
  70. rag_wright/contracts/span.py +76 -0
  71. rag_wright/contracts/value_match.py +84 -0
  72. rag_wright/corpus/__init__.py +0 -0
  73. rag_wright/corpus/canonicalize.py +116 -0
  74. rag_wright/corpus/cuad.py +153 -0
  75. rag_wright/corpus/cuad_ingestion.py +72 -0
  76. rag_wright/corpus/document_parser.py +299 -0
  77. rag_wright/corpus/edgar.py +231 -0
  78. rag_wright/corpus/gcs_ingestion.py +120 -0
  79. rag_wright/corpus/http.py +110 -0
  80. rag_wright/corpus/selection.py +152 -0
  81. rag_wright/mcp/__init__.py +11 -0
  82. rag_wright/mcp/compliance_server.py +299 -0
  83. rag_wright/mcp/intra_document_qa_server.py +170 -0
  84. rag_wright/mcp/relational_qa_server.py +171 -0
  85. rag_wright/mcp/session_store.py +64 -0
  86. rag_wright/mcp/typed_property_retrieval_server.py +191 -0
  87. rag_wright/models/__init__.py +8 -0
  88. rag_wright/models/profiles.py +331 -0
  89. rag_wright/models/seam.py +497 -0
  90. rag_wright/models/tag_structured.py +285 -0
  91. rag_wright/models/tracing.py +179 -0
  92. rag_wright/models/usage.py +102 -0
  93. rag_wright/okf/__init__.py +11 -0
  94. rag_wright/okf/compile.py +292 -0
  95. rag_wright/okf/document.py +47 -0
  96. rag_wright/okf/enrich.py +176 -0
  97. rag_wright/okf/links.py +190 -0
  98. rag_wright/okf/lint.py +105 -0
  99. rag_wright/ontology/__init__.py +6 -0
  100. rag_wright/ontology/_generated_template_meta.py +60 -0
  101. rag_wright/ontology/_generated_vocab.py +52 -0
  102. rag_wright/ontology/clause_template.py +964 -0
  103. rag_wright/ontology/codegen.py +84 -0
  104. rag_wright/ontology/compliance_bridge.ttl +186 -0
  105. rag_wright/ontology/contract_bridge.ttl +2685 -0
  106. rag_wright/ontology/contract_taxonomy.py +24 -0
  107. rag_wright/ontology/derive.py +58 -0
  108. rag_wright/ontology/loader.py +435 -0
  109. rag_wright/ontology/packs/ftc_16cfr255.ttl +29 -0
  110. rag_wright/ontology/registry.py +87 -0
  111. rag_wright/ontology/template_introspect.py +100 -0
  112. rag_wright/py.typed +0 -0
  113. rag_wright/reference/__init__.py +2 -0
  114. rag_wright/reference/compliance.py +41 -0
  115. rag_wright/reference/contract_seam.py +123 -0
  116. rag_wright/skills/__init__.py +7 -0
  117. rag_wright/skills/claim_extraction/SKILL.md +47 -0
  118. rag_wright/skills/claim_extraction/__init__.py +1 -0
  119. rag_wright/skills/claim_extraction/template.py +50 -0
  120. rag_wright/skills/compliance_judgment/SKILL.md +59 -0
  121. rag_wright/skills/corpus_ingest/SKILL.md +106 -0
  122. rag_wright/skills/extraction_semantic_judge/SKILL.md +51 -0
  123. rag_wright/skills/extraction_semantic_judge/__init__.py +1 -0
  124. rag_wright/skills/generation/SKILL.md +64 -0
  125. rag_wright/skills/generation/__init__.py +1 -0
  126. rag_wright/skills/generic_compliance_judgment/SKILL.md +58 -0
  127. rag_wright/skills/okf_navigate/SKILL.md +137 -0
  128. rag_wright/skills/requirement_extraction/SKILL.md +47 -0
  129. rag_wright/skills/requirement_extraction/__init__.py +1 -0
  130. rag_wright/skills/requirement_extraction/template.py +50 -0
  131. rag_wright/skills/rlm/SKILL.md +186 -0
  132. rag_wright/skills/rlm/__init__.py +31 -0
  133. rag_wright/skills/rlm/agent.py +292 -0
  134. rag_wright/skills/span_relevance_judgment/SKILL.md +67 -0
  135. rag_wright/skills/vision_to_text/SKILL.md +36 -0
  136. rag_wright/skills/vision_to_text/__init__.py +1 -0
  137. rag_wright/spans/__init__.py +1 -0
  138. rag_wright/spans/boundary.py +78 -0
  139. rag_wright/spans/clause_function_classifier.py +490 -0
  140. rag_wright/spans/clause_kg_extractor.py +337 -0
  141. rag_wright/spans/cuad_labels.py +81 -0
  142. rag_wright/spans/dim_classifier.py +158 -0
  143. rag_wright/spans/dim_fleet.json +411 -0
  144. rag_wright/spans/function_classifier.py +77 -0
  145. rag_wright/spans/function_families.py +62 -0
  146. rag_wright/spans/hybrid_classifier.py +103 -0
  147. rag_wright/spans/legalbert_classifier.py +83 -0
  148. rag_wright/spans/model_capabilities.py +107 -0
  149. rag_wright/spans/new_function_labels.py +111 -0
  150. rag_wright/spans/page_map.py +68 -0
  151. rag_wright/spans/property_extractor.py +365 -0
  152. rag_wright/spans/property_grounding.py +182 -0
  153. rag_wright/spans/reclassify.py +77 -0
  154. rag_wright/spans/scarce_function_labels.py +105 -0
  155. rag_wright/spans/segment.py +341 -0
  156. rag_wright/spans/semantic_judge.py +197 -0
  157. rag_wright/spans/symbolic_validation.py +131 -0
  158. rag_wright/spans/tag_clause_extractor.py +182 -0
  159. rag_wright/store/__init__.py +6 -0
  160. rag_wright/store/arcadedb.py +1135 -0
  161. rag_wright/store/chunk_text.py +66 -0
  162. rag_wright/store/seam.py +213 -0
  163. rag_wright/subgraphs/__init__.py +0 -0
  164. rag_wright/subgraphs/async_ingestion.py +204 -0
  165. rag_wright/subgraphs/compliance_check.py +1042 -0
  166. rag_wright/subgraphs/compliance_ingestion.py +306 -0
  167. rag_wright/subgraphs/contract_ingestion_pipeline.py +999 -0
  168. rag_wright/subgraphs/graph_extraction.py +102 -0
  169. rag_wright/subgraphs/intra_document_qa.py +328 -0
  170. rag_wright/subgraphs/observability.py +140 -0
  171. rag_wright/subgraphs/query_constraint_extraction.py +73 -0
  172. rag_wright/subgraphs/relational_qa.py +165 -0
  173. rag_wright/subgraphs/requirement_extraction.py +137 -0
  174. rag_wright/subgraphs/scaffold.py +65 -0
  175. rag_wright/subgraphs/semantic_chunking.py +183 -0
  176. rag_wright/subgraphs/typed_clause_extraction.py +172 -0
  177. rag_wright/subgraphs/typed_property_retrieval.py +278 -0
  178. rag_wright/util/__init__.py +1 -0
  179. rag_wright/util/concurrent.py +153 -0
  180. rag_wright/util/spacy_model.py +45 -0
  181. rag_wright-0.1.0.dist-info/METADATA +168 -0
  182. rag_wright-0.1.0.dist-info/RECORD +184 -0
  183. rag_wright-0.1.0.dist-info/WHEEL +4 -0
  184. rag_wright-0.1.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,153 @@
1
+ """CC-3 (compliance §13 C-3), SKILL-SPLIT: subject ad -> `Claim[]`, split into a SKILL + a FUNCTION.
2
+
3
+ Per the capability-architecture principle (`function` = deterministic/no-model; a single LLM act = an authored
4
+ `agent_skill`; a workflow = `subgraph`), this is two capabilities:
5
+
6
+ - **`claim_extraction` (agent_skill)** -- the claim-extraction METHOD, authored as `skills/claim_extraction/`
7
+ (SKILL.md + the `template.py` schema asset: `ExtractedAd`/`ExtractedClaim`). One docling-graph call (`direct`,
8
+ ads are short) fills the template. `extract_ad` is its runtime; `extract_fn` is injected for hermetic tests.
9
+ - **`claim_adaptation` (function)** -- `to_claims`: DETERMINISTIC, no model. Maps the raw `ExtractedAd` to the
10
+ closed CC-1 `Claim` vocab (off-vocab claim_type kept-but-AMBIGUOUS -- a checkable assertion is never dropped),
11
+ attaches the span provenance + content-hash `claim_id`.
12
+
13
+ `claim_extraction(...)` composes them (skill -> function) for callers. The schema lives with the skill (an
14
+ Agent-Skill asset), re-exported here for consumers/tests.
15
+ """
16
+
17
+ from __future__ import annotations
18
+
19
+ from typing import Any, Callable
20
+
21
+ from rag_wright.capabilities.dg_extraction import aextract_parties, extract_parties
22
+ from rag_wright.contracts.compliance import Claim, ClaimType
23
+ from rag_wright.contracts.provenance import ConfidenceTag
24
+ from rag_wright.skills.claim_extraction.template import ExtractedAd, ExtractedClaim # the skill's schema asset
25
+
26
+ __all__ = ["ExtractedAd", "ExtractedClaim", "extract_ad", "aextract_ad", "to_claims", "claim_extraction",
27
+ "aclaim_extraction", "register_claim_extraction", "register_claim_adaptation"]
28
+
29
+ _CLAIM_TYPES = {c.value for c in ClaimType}
30
+ # off-vocab fallback: keep the claim (never drop a checkable assertion) but mark it AMBIGUOUS. EFFICACY is the
31
+ # broadest "the product works" type; the AMBIGUOUS tag is what actually signals the uncertainty downstream.
32
+ _FALLBACK_CLAIM_TYPE = ClaimType.EFFICACY
33
+
34
+
35
+ def _coerce_claim_type(raw: str) -> tuple[ClaimType, bool]:
36
+ """(ClaimType, ambiguous?) -- map the model's string to the closed vocab; an off-vocab value keeps the
37
+ claim under the fallback type and flags AMBIGUOUS (never drop a real assertion)."""
38
+ value = (raw or "").strip().lower()
39
+ if value in _CLAIM_TYPES:
40
+ return ClaimType(value), False
41
+ return _FALLBACK_CLAIM_TYPE, True
42
+
43
+
44
+ def _clean(v: str) -> str | None:
45
+ v = (v or "").strip()
46
+ return v or None
47
+
48
+
49
+ def to_claims(extracted: ExtractedAd, *, source_doc: str) -> list[Claim]:
50
+ """`claim_adaptation` (FUNCTION -- deterministic, no model): adapt an extracted ad to validated `Claim`s.
51
+ Off-vocab claim_type -> fallback + AMBIGUOUS; blank assertion skipped. `claim_id` uses the content-hash
52
+ scheme; the (source_doc, assertion) is the span cite."""
53
+ from rag_wright.contracts.compliance import Constraint
54
+
55
+ out: list[Claim] = []
56
+ for index, item in enumerate(extracted.claims):
57
+ text = (item.assertion_text or "").strip()
58
+ if not text:
59
+ continue
60
+ claim_type, ambiguous = _coerce_claim_type(item.claim_type)
61
+ actor = _clean(item.actor)
62
+ # DEON-8: surface the ROLE actor as a dimension-agnostic Constraint on `.scope` (mirroring the generic
63
+ # `to_facts`), so the obligation actor gate (DEON-7) works on the ad path too. The typed `.actor` is kept.
64
+ scope = [Constraint(dimension="actor", value=actor.lower())] if actor else []
65
+ out.append(Claim(
66
+ fact_id=Claim.make_id(source_doc, index, text),
67
+ source_doc=source_doc,
68
+ claim_type=claim_type,
69
+ assertion_text=text,
70
+ scope=scope,
71
+ actor=actor,
72
+ subject_product=_clean(item.subject_product),
73
+ quantitative_value=_clean(item.quantitative_value),
74
+ disclosures_present=[d.strip() for d in item.disclosures_present if d and d.strip()],
75
+ evidence_referenced=bool(item.evidence_referenced),
76
+ medium=_clean(item.medium),
77
+ confidence=ConfidenceTag.AMBIGUOUS if ambiguous else ConfidenceTag.EXTRACTED,
78
+ ))
79
+ return out
80
+
81
+
82
+ ExtractFn = Callable[..., Any] # (text, model, *, template, **kw) -> ExtractedAd | None
83
+
84
+
85
+ def extract_ad(
86
+ text: str, *, model: Any, extract_fn: ExtractFn = extract_parties,
87
+ max_tokens: int = 1500, preamble_chars: int = 8000, extraction_contract: str = "direct",
88
+ ) -> ExtractedAd | None:
89
+ """The `claim_extraction` SKILL's runtime: run the docling-graph extraction (the skill's `template.py`
90
+ schema) through the model seam and return the raw `ExtractedAd` (or None). Ads are short -> `direct`."""
91
+ return extract_fn(text, model, template=ExtractedAd,
92
+ max_tokens=max_tokens, preamble_chars=preamble_chars,
93
+ extraction_contract=extraction_contract)
94
+
95
+
96
+ def claim_extraction(
97
+ text: str, *, model: Any, source_doc: str,
98
+ extract_fn: ExtractFn = extract_parties, max_tokens: int = 1500, preamble_chars: int = 8000,
99
+ extraction_contract: str = "direct",
100
+ ) -> list[Claim]:
101
+ """Compose the SKILL (extract_ad) and the FUNCTION (to_claims): extract the raw ad, then adapt to `Claim`s.
102
+ Returns [] if extraction yields nothing."""
103
+ extracted = extract_ad(text, model=model, extract_fn=extract_fn, max_tokens=max_tokens,
104
+ preamble_chars=preamble_chars, extraction_contract=extraction_contract)
105
+ if extracted is None:
106
+ return []
107
+ return to_claims(extracted, source_doc=source_doc)
108
+
109
+
110
+ async def aextract_ad(
111
+ text: str, *, model: Any, aextract_fn: Any = aextract_parties,
112
+ max_tokens: int = 1500, preamble_chars: int = 8000, extraction_contract: str = "direct",
113
+ ) -> ExtractedAd | None:
114
+ """ASYNC-C1 (ADR-0057): the async twin of `extract_ad` -- the claim-extraction docling-graph act on the async
115
+ seam (`aextract_parties`, true wall-clock deadline via the injected client). `aextract_fn` injected for tests."""
116
+ return await aextract_fn(text, model, template=ExtractedAd,
117
+ max_tokens=max_tokens, preamble_chars=preamble_chars,
118
+ extraction_contract=extraction_contract)
119
+
120
+
121
+ async def aclaim_extraction(
122
+ text: str, *, model: Any, source_doc: str, aextract_fn: Any = aextract_parties,
123
+ max_tokens: int = 1500, preamble_chars: int = 8000, extraction_contract: str = "direct",
124
+ ) -> list[Claim]:
125
+ """ASYNC-C1 (ADR-0057): the async twin of `claim_extraction` -- await the async extraction ACT, then the
126
+ deterministic `to_claims` adaptation. Same contract: [] if extraction yields nothing."""
127
+ extracted = await aextract_ad(text, model=model, aextract_fn=aextract_fn, max_tokens=max_tokens,
128
+ preamble_chars=preamble_chars, extraction_contract=extraction_contract)
129
+ if extracted is None:
130
+ return []
131
+ return to_claims(extracted, source_doc=source_doc)
132
+
133
+
134
+ def register_claim_extraction(registry) -> None:
135
+ """Register `claim_extraction` as an AGENT_SKILL (CC-3): a single docling-graph LLM extraction act, authored
136
+ as `skills/claim_extraction/` (SKILL.md + the template.py schema). Typed output = `ExtractedAd`."""
137
+ registry.register(
138
+ "claim_extraction",
139
+ contract=ExtractedAd,
140
+ kind="agent_skill",
141
+ display_name="Claim extraction (subject ad -> checkable claims; authored skill)",
142
+ )
143
+
144
+
145
+ def register_claim_adaptation(registry) -> None:
146
+ """Register `claim_adaptation` (FUNCTION -- deterministic): the skill's raw `ExtractedAd` -> validated
147
+ `Claim[]` (closed vocab, off-vocab kept-but-AMBIGUOUS, span provenance, content-hash id)."""
148
+ registry.register(
149
+ "claim_adaptation",
150
+ contract=Claim,
151
+ kind="function",
152
+ display_name="Claim adaptation (extracted ad -> validated Claims)",
153
+ )
@@ -0,0 +1,117 @@
1
+ """ADR-0044: the `IS_EXCEPTION_TO` derived carve-out relationship (exception clause -> the Cap clause it excepts).
2
+
3
+ A single contract's liability structure is usually stored as TWO disconnected clauses -- a `Cap On Liability`
4
+ clause (often without its carve-outs) and a separate, often property-less `Uncapped Liability` clause (the
5
+ carve-out, e.g. "uncapped for negligence") -- with NO relationship between them. So a query like "how is
6
+ liability capped, and under what conditions?" gets two contradictory-looking fragments and the generator
7
+ abstains, instead of "capped at X, EXCEPT uncapped for negligence."
8
+
9
+ This capability adds the missing link WITHOUT re-ingesting: a pure pass over the already-populated clauses (a derived-relationship
10
+ linking pass over the existing KG) that writes `IsExceptionTo` edges. The signal is SYMBOLIC co-occurrence +
11
+ POSITIONAL PROXIMITY: within a contract, an `Uncapped` clause whose operative span is within `window` characters
12
+ of a `Cap` clause's span (~ the same liability section) is that cap's carve-out. Distant co-occurrence is NOT
13
+ linked (probably an unrelated standalone uncapped clause). The link is **INFERRED** (a reasoned inference, not an
14
+ extracted fact, FR-S.4): surfaced at query time and human-validatable, never a silent hard claim.
15
+
16
+ Scope (ADR-0044): the edge type is general; only the cap<->uncapped pair is populated for now.
17
+ """
18
+
19
+ from __future__ import annotations
20
+
21
+ from collections import defaultdict
22
+ from typing import Any
23
+
24
+ from pydantic import BaseModel
25
+
26
+ from rag_wright.capabilities.registry import CapabilityRegistry
27
+ from rag_wright.contracts.provenance import ConfidenceTag
28
+
29
+ CAP_FUNCTION = "Cap On Liability"
30
+ EXCEPTION_FUNCTION = "Uncapped Liability"
31
+ DEFAULT_PROXIMITY_WINDOW = 3000 # chars between the two spans' ranges (~ one liability section); tunable
32
+
33
+
34
+ class ClauseExceptionLink(BaseModel):
35
+ """One `IsExceptionTo` edge: the `exception_clause_id` (an Uncapped clause) is a carve-out/exception to the
36
+ `cap_clause_id` (a Cap clause) in `contract_id`. Confidence INFERRED -- a derived, reasoned link (FR-S.4)."""
37
+
38
+ exception_clause_id: str
39
+ cap_clause_id: str
40
+ contract_id: str
41
+ confidence: ConfidenceTag = ConfidenceTag.INFERRED
42
+
43
+
44
+ class ClauseExceptionLinkResult(BaseModel):
45
+ """The capability's output: the derived `IsExceptionTo` links + the honest unlinked count."""
46
+
47
+ links: list[ClauseExceptionLink]
48
+ unlinked_exceptions: int # Uncapped clauses with no in-window Cap clause (skipped, not linked)
49
+ contracts_processed: int
50
+
51
+
52
+ def _gap(a: dict, b: dict) -> int:
53
+ """Character gap between two spans' [doc_start, doc_end] ranges (0 if they overlap)."""
54
+ a0, a1 = int(a.get("doc_start") or 0), int(a.get("doc_end") or 0)
55
+ b0, b1 = int(b.get("doc_start") or 0), int(b.get("doc_end") or 0)
56
+ if a1 < b0:
57
+ return b0 - a1
58
+ if b1 < a0:
59
+ return a0 - b1
60
+ return 0
61
+
62
+
63
+ def derive_exception_links(
64
+ positions: list[dict], *, window: int = DEFAULT_PROXIMITY_WINDOW,
65
+ cap_function: str = CAP_FUNCTION, exception_function: str = EXCEPTION_FUNCTION,
66
+ ) -> ClauseExceptionLinkResult:
67
+ """Pure: within each contract, link each `exception_function` clause to the NEAREST `cap_function` clause
68
+ whose operative span is within `window` chars (proximity ~ same section) -> an INFERRED `IsExceptionTo` link.
69
+ A distant or cap-less exception clause is counted `unlinked`, never linked to a far cap (that would be a
70
+ false carve-out). One link per (exception, cap) pair. `positions`: rows with {clause_id, function,
71
+ contract_id, doc_start, doc_end}."""
72
+ by_contract: dict[str, dict[str, list[dict]]] = defaultdict(lambda: {"cap": [], "exc": []})
73
+ for p in positions:
74
+ bucket = by_contract[p["contract_id"]]
75
+ if p["function"] == cap_function:
76
+ bucket["cap"].append(p)
77
+ elif p["function"] == exception_function:
78
+ bucket["exc"].append(p)
79
+
80
+ links: list[ClauseExceptionLink] = []
81
+ unlinked = 0
82
+ for contract_id, bucket in by_contract.items():
83
+ caps = bucket["cap"]
84
+ if not caps: # uncapped clauses but no cap clause in this contract -> nothing to except
85
+ unlinked += len(bucket["exc"])
86
+ continue
87
+ for exc in bucket["exc"]:
88
+ nearest = min(caps, key=lambda cap: _gap(exc, cap))
89
+ if _gap(exc, nearest) <= window:
90
+ links.append(ClauseExceptionLink(
91
+ exception_clause_id=exc["clause_id"], cap_clause_id=nearest["clause_id"],
92
+ contract_id=contract_id))
93
+ else:
94
+ unlinked += 1 # co-occurs but distant -> probably unrelated; do not link (no false carve-out)
95
+
96
+ return ClauseExceptionLinkResult(
97
+ links=links, unlinked_exceptions=unlinked, contracts_processed=len(by_contract))
98
+
99
+
100
+ def clause_exception_linking(store: Any, *, window: int = DEFAULT_PROXIMITY_WINDOW) -> ClauseExceptionLinkResult:
101
+ """The registered capability (ADR-0044): read the Cap + Uncapped clause positions, derive the proximity-based
102
+ `IsExceptionTo` links, write them (idempotent, clears the layer first), and return the result. No re-ingest --
103
+ a derived-relationship pass over the existing KG."""
104
+ positions = store.clause_positions([CAP_FUNCTION, EXCEPTION_FUNCTION])
105
+ result = derive_exception_links(positions, window=window)
106
+ store.write_clause_exception_links(result.links)
107
+ return result
108
+
109
+
110
+ def register_clause_exception_linking(registry: CapabilityRegistry) -> None:
111
+ """ADR-0044: register `clause_exception_linking` (function; the cap<->uncapped carve-out relationship)."""
112
+ registry.register(
113
+ "clause_exception_linking",
114
+ contract=ClauseExceptionLinkResult,
115
+ kind="function",
116
+ display_name="Clause exception linking (IsExceptionTo carve-out edges over the contract KG)",
117
+ )
@@ -0,0 +1,322 @@
1
+ """CC-4 (compliance §13.2), SKILL-SPLIT: the judgment node, split into a SKILL + a deterministic FUNCTION.
2
+
3
+ Per the capability-architecture principle (a `function` is deterministic and takes no model; a single LLM act is
4
+ an authored `agent_skill`; a workflow is a `subgraph`), the judgment is two capabilities:
5
+
6
+ - **`compliance_judgment` (agent_skill)** -- the LLM judgment METHOD, authored as `skills/compliance_judgment/
7
+ SKILL.md` and applied through the model seam (product = Granite, ADR-0039). Given one claim + one requirement
8
+ and ONLY the ad text, it returns a raw `JudgeVerdict` (verdict / rationale / confidence). `build_compliance_
9
+ judge_fn` is its runtime; `structured_factory` is injected for hermetic tests.
10
+ - **`compliance_finding_assembly` (function)** -- `assemble_finding`: DETERMINISTIC, no model. Maps the raw
11
+ verdict to the closed vocab (unreadable/missing -> needs_review, the conservative default), attaches the
12
+ BOTH-SIDED citation FROM THE INPUTS (the model never authors a citation), and returns the `ComplianceFinding`.
13
+
14
+ `compliance_judgment(...)` composes them (skill -> function) for callers; `judge_pairs` runs the composition
15
+ concurrently (async + semaphore + LLM-CALL-TIMEOUT). The applying capability owns the guarantees the skill does
16
+ not (verdict vocab, conservative default, citation) -- the SKILL.md teaches only the reading.
17
+ """
18
+
19
+ from __future__ import annotations
20
+
21
+ import asyncio
22
+ import os
23
+ from pathlib import Path
24
+ from typing import Awaitable, Callable, Optional
25
+
26
+ from pydantic import BaseModel
27
+
28
+ from rag_wright.contracts.compliance import (
29
+ CheckableFact,
30
+ Claim,
31
+ ComplianceFinding,
32
+ DeonticType,
33
+ Requirement,
34
+ Verdict,
35
+ )
36
+ from rag_wright.models.seam import build_structured
37
+ from rag_wright.util.concurrent import map_concurrent
38
+
39
+ _VERDICTS = {v.value for v in Verdict}
40
+ _SKILL_PATH = Path(__file__).parents[1] / "skills" / "compliance_judgment" / "SKILL.md" # advertising method
41
+ # COMP-VERDICT-GENERIC: the DOMAIN-AGNOSTIC judgment method -- the base the advertising SKILL specializes.
42
+ _GENERIC_SKILL_PATH = Path(__file__).parents[1] / "skills" / "generic_compliance_judgment" / "SKILL.md"
43
+
44
+
45
+ class JudgeVerdict(BaseModel):
46
+ """The raw structured output of one judge call (the `compliance_judgment` SKILL's typed output). `verdict`
47
+ is a loose string mapped to the closed `Verdict` by the FUNCTION (an unreadable value -> needs_review);
48
+ citations are added from the inputs by the function, never authored here."""
49
+
50
+ verdict: str
51
+ rationale: str = ""
52
+ confidence: float = 0.0
53
+
54
+
55
+ # judge_fn: (subject_fact, requirement) -> JudgeVerdict, or None if the judge could not rule (-> conservative
56
+ # default). Accepts any `CheckableFact` (the advertising `Claim` is one).
57
+ JudgeFn = Callable[[CheckableFact, Requirement], Optional[JudgeVerdict]]
58
+ # ASYNC-C1 (ADR-0057): the async judge seam -- same signature, awaitable result (the model call gets a true
59
+ # wall-clock deadline via build_structured's .ainvoke).
60
+ AJudgeFn = Callable[[CheckableFact, Requirement], Awaitable[Optional[JudgeVerdict]]]
61
+
62
+ # The per-call appendix bound onto the SKILL method (the static method teaches the reading; the specific
63
+ # requirement + subject are appended at call time, the okf_navigate `_with_question` pattern).
64
+ # COMP-VERDICT-GENERIC: split into a domain-agnostic BASE tail (requirement + subject assertion -- works for ANY
65
+ # CheckableFact / domain) + an ADVERTISING enrichment line (claim_type / disclosures / evidence). The generic
66
+ # judge uses only the base; the advertising judge appends the enrichment (behavior unchanged).
67
+ _BASE_PROMPT_TAIL = (
68
+ "\n\nREQUIREMENT ({deontic}, {citation}; applies to {actor}):\n{requirement_text}\n\n"
69
+ "SUBJECT:\n{assertion}"
70
+ )
71
+ _AD_ENRICHMENT = "\n\nCLAIM SIGNALS: type={claim_type}; disclosures_present={disclosures}; evidence_referenced={evidence}"
72
+
73
+
74
+ def _skill_body(path: Path) -> str:
75
+ """A SKILL.md body with its YAML frontmatter stripped -- the judge's system/method prompt."""
76
+ text = path.read_text(encoding="utf-8")
77
+ if text.startswith("---"):
78
+ marker = text.find("\n---", 3)
79
+ if marker != -1:
80
+ text = text[marker + 4 :]
81
+ return text.strip()
82
+
83
+
84
+ def judgment_method() -> str:
85
+ """The ADVERTISING compliance-judgment method (skills/compliance_judgment/SKILL.md, FTC doctrine)."""
86
+ return _skill_body(_SKILL_PATH)
87
+
88
+
89
+ def generic_judgment_method() -> str:
90
+ """COMP-VERDICT-GENERIC: the DOMAIN-AGNOSTIC judgment method (skills/generic_compliance_judgment/SKILL.md) --
91
+ no advertising doctrine, so the generic judge reasons about "the subject" in any domain."""
92
+ return _skill_body(_GENERIC_SKILL_PATH)
93
+
94
+
95
+ def _base_tail(fact: CheckableFact, requirement: Requirement) -> str:
96
+ """The domain-agnostic judge appendix: the requirement + the subject assertion. Works for ANY CheckableFact.
97
+ issue 0044: the DEON-8 document signals (disclosure/evidence union) are appended here from `document_signals`
98
+ -- so the JUDGE still sees them (prompt unchanged), while `assertion_text` (and thus the citation) stays pure
99
+ document text. `document_signals` is '' for a plain fact, so the generic path is unaffected."""
100
+ return _BASE_PROMPT_TAIL.format(
101
+ deontic=requirement.deontic_type.value, citation=requirement.citation, actor=requirement.actor,
102
+ requirement_text=requirement.requirement_text, assertion=fact.assertion_text) \
103
+ + (getattr(fact, "document_signals", "") or "")
104
+
105
+
106
+ # DEON-1 (issue 0012): the neuro-symbolic deontic framing -- the KG's deontic_type tells the judge HOW to reason.
107
+ # An OBLIGATION is breached by ABSENCE: the subject text above is the document's relevant content (DEON judges an
108
+ # obligation ONCE over the document, not per sentence), so the judge must DECIDE yes/no, not hedge to "unclear"
109
+ # just because the required element is missing -- absence IS the violation.
110
+ _OBLIGATION_FRAMING = (
111
+ "\n\nDEONTIC FRAMING -- this requirement is an OBLIGATION (it must be SATISFIED): the subject text above is the "
112
+ "document's relevant content, checked as a whole. Decide definitively: if the required element (e.g. the "
113
+ "disclosure/action) IS present, COMPLIANT; if it is ABSENT from this text, that is a VIOLATION -- a missing "
114
+ "required element is itself the breach. Do NOT answer needs_review merely because the element is absent.")
115
+
116
+
117
+ def _deontic_framing(requirement: Requirement) -> str:
118
+ return _OBLIGATION_FRAMING if requirement.deontic_type is DeonticType.OBLIGATION else ""
119
+
120
+
121
+ # DEON-9 (issue 0012): permission-as-defense -- the same-source PERMISSIONS linked to this O/F rule (carve-outs /
122
+ # safe-harbors). The judge reasons over the rule AND its exceptions: a subject that falls within a permitted
123
+ # carve-out is NOT a violation. Only apply an exception that genuinely fits the subject (never a blanket excuse).
124
+ _DEFENSE_FRAMING = (
125
+ "\n\nEXCEPTIONS / DEFENSES (permitted carve-outs that may EXCUSE this rule):\n{defenses}\n"
126
+ "If the subject falls within one of these permitted exceptions, the rule is NOT violated -- rule COMPLIANT, "
127
+ "and say which exception applies. Only apply an exception that genuinely fits the subject.")
128
+
129
+
130
+ def _defense_framing(requirement: Requirement) -> str:
131
+ defenses = getattr(requirement, "defenses", None) or []
132
+ if not defenses:
133
+ return ""
134
+ return _DEFENSE_FRAMING.format(defenses="\n".join(f"- {d}" for d in defenses))
135
+
136
+
137
+ def _ad_signals(fact: CheckableFact) -> str:
138
+ """DEON-8: the ADVERTISING claim-signals line -- rendered only for a typed `Claim`. An obligation judged ONCE
139
+ (DEON-1) hands the judge a plain `CheckableFact` EVIDENCE BUNDLE with no claim_type; return '' so the ad judge
140
+ TOLERATES it (the bundle's disclosure signals ride in its text via DEON-8 Option 1) instead of crashing on a
141
+ missing attribute."""
142
+ ct = getattr(fact, "claim_type", None)
143
+ if ct is None:
144
+ return ""
145
+ return _AD_ENRICHMENT.format(
146
+ claim_type=ct.value, disclosures=fact.disclosures_present or "none", evidence=fact.evidence_referenced)
147
+
148
+
149
+ def build_generic_judge_fn(model_id: str, *, structured_factory=build_structured) -> JudgeFn:
150
+ """COMP-VERDICT-GENERIC: the DOMAIN-AGNOSTIC judge -- rules a `(subject_fact, requirement)` pair on TEXT alone
151
+ using the GENERIC judgment method (no advertising doctrine; reasons about "the subject" in any domain), so it
152
+ gives a verdict in ANY compliance domain. Same conservative default as the advertising judge; the method +
153
+ the appendix (base tail only, no claim signals) differ."""
154
+ method = generic_judgment_method()
155
+
156
+ def judge(fact: CheckableFact, requirement: Requirement) -> Optional[JudgeVerdict]:
157
+ return structured_factory(model_id, JudgeVerdict).invoke(
158
+ method + _base_tail(fact, requirement) + _defense_framing(requirement) + _deontic_framing(requirement))
159
+
160
+ return judge
161
+
162
+
163
+ def build_ageneric_judge_fn(model_id: str, *, structured_factory=build_structured) -> AJudgeFn:
164
+ """ASYNC-C1 (ADR-0057): the async twin of `build_generic_judge_fn` -- the DOMAIN-AGNOSTIC judge on the async
165
+ structured seam (`.ainvoke`, a true wall-clock deadline on the model call). Same method + base tail."""
166
+ method = generic_judgment_method()
167
+
168
+ async def judge(fact: CheckableFact, requirement: Requirement) -> Optional[JudgeVerdict]:
169
+ return await structured_factory(model_id, JudgeVerdict).ainvoke(
170
+ method + _base_tail(fact, requirement) + _defense_framing(requirement) + _deontic_framing(requirement))
171
+
172
+ return judge
173
+
174
+
175
+ def build_compliance_judge_fn(model_id: str, *, structured_factory=build_structured) -> JudgeFn:
176
+ """The advertising `compliance_judgment` SKILL runtime: a Granite-backed judge `JudgeFn` through the model seam
177
+ (product = vLLM-Granite; ADR-0039). Base tail (requirement + subject) + the ADVERTISING claim signals
178
+ (claim_type / disclosures / evidence). Behavior unchanged from before the CheckableFact split.
179
+ `structured_factory` is injected for tests."""
180
+ method = judgment_method()
181
+
182
+ def judge(claim: Claim, requirement: Requirement) -> Optional[JudgeVerdict]:
183
+ prompt = (method + _base_tail(claim, requirement) + _ad_signals(claim)
184
+ + _defense_framing(requirement) + _deontic_framing(requirement))
185
+ return structured_factory(model_id, JudgeVerdict).invoke(prompt)
186
+
187
+ return judge
188
+
189
+
190
+ def build_acompliance_judge_fn(model_id: str, *, structured_factory=build_structured) -> AJudgeFn:
191
+ """ASYNC-C1 (ADR-0057): the async twin of `build_compliance_judge_fn` -- the advertising judge on the async
192
+ structured seam (`.ainvoke`, a true wall-clock deadline). Same method + base tail + claim signals."""
193
+ method = judgment_method()
194
+
195
+ async def judge(claim: Claim, requirement: Requirement) -> Optional[JudgeVerdict]:
196
+ prompt = (method + _base_tail(claim, requirement) + _ad_signals(claim)
197
+ + _defense_framing(requirement) + _deontic_framing(requirement))
198
+ return await structured_factory(model_id, JudgeVerdict).ainvoke(prompt)
199
+
200
+ return judge
201
+
202
+
203
+ def _to_verdict(raw: str) -> Verdict:
204
+ """Map the LLM's verdict string to the closed vocab; an unreadable value -> NEEDS_REVIEW (conservative)."""
205
+ value = (raw or "").strip().lower().replace("-", "_").replace(" ", "_")
206
+ return Verdict(value) if value in _VERDICTS else Verdict.NEEDS_REVIEW
207
+
208
+
209
+ def assemble_finding(
210
+ claim: Claim, requirement: Requirement, ruling: Optional[JudgeVerdict]
211
+ ) -> ComplianceFinding:
212
+ """`compliance_finding_assembly` (FUNCTION -- deterministic, no model): map the skill's raw `JudgeVerdict`
213
+ (or None) to a `ComplianceFinding`. None or an off-vocab verdict conservatively defaults to `needs_review`;
214
+ the BOTH-SIDED citation is taken from the INPUTS (the model never authors a citation)."""
215
+ if ruling is None:
216
+ verdict, rationale, confidence = Verdict.NEEDS_REVIEW, "judge did not return a ruling", 0.0
217
+ else:
218
+ verdict = _to_verdict(ruling.verdict)
219
+ rationale = ruling.rationale
220
+ confidence = min(1.0, max(0.0, ruling.confidence))
221
+ # UNIFY-A / SEG-1: cite "doc {locator}: {assertion}" where the locator is the fact's structural path
222
+ # ("§ 4.2", "§ 4.2 ¶3", "§ 4.2 · bullet 2") when present, else "" -> the old "doc: assertion" (back-compat).
223
+ # Still input-authored (the model never writes the citation).
224
+ _loc = claim.locator()
225
+ _sec = f" {_loc}" if _loc else ""
226
+ return ComplianceFinding(
227
+ claim_id=claim.fact_id,
228
+ requirement_id=requirement.requirement_id,
229
+ verdict=verdict,
230
+ rationale=rationale,
231
+ citation_claim=f"{claim.source_doc}{_sec}: {claim.assertion_text}", # issue 0044: doc text only, no signals
232
+ citation_requirement=f"{requirement.citation} ({requirement.requirement_id}): {requirement.requirement_text}",
233
+ citation_claim_kind=getattr(claim, "citation_kind", "verbatim"), # issue 0044: verbatim vs assembled
234
+ confidence=confidence,
235
+ )
236
+
237
+
238
+ def compliance_judgment(claim: Claim, requirement: Requirement, *, judge_fn: JudgeFn) -> ComplianceFinding:
239
+ """Compose the SKILL (LLM judgment) and the FUNCTION (deterministic assembly): run `judge_fn` then
240
+ `assemble_finding`. A None ruling conservatively defaults to `needs_review`. Citations come from the inputs."""
241
+ return assemble_finding(claim, requirement, judge_fn(claim, requirement))
242
+
243
+
244
+ _JUDGE_TIMEOUT_S = float(os.environ.get("RAG_JUDGE_TIMEOUT_S", "90")) # per-pair wall-clock bound (LLM-CALL-TIMEOUT)
245
+
246
+
247
+ def judge_pairs(
248
+ pairs: list[tuple[Claim, Requirement]], *, judge_fn: JudgeFn, max_concurrency: int = 8,
249
+ timeout_s: float | None = _JUDGE_TIMEOUT_S,
250
+ ) -> list[ComplianceFinding]:
251
+ """Run the skill->function composition over many `(claim, requirement)` pairs concurrently (async +
252
+ semaphore, per the parallel-LLM rule). Order is preserved. CC-6 drives this over a subject doc's claims x
253
+ their applicable requirements.
254
+
255
+ `timeout_s` (LLM-CALL-TIMEOUT) bounds each judgment with a hard wall-clock deadline so a stalled provider
256
+ response never hangs the batch; a timed-out pair (map_concurrent -> None) becomes a conservative
257
+ needs_review finding (the same conservative default as a judge that could not rule)."""
258
+ results = map_concurrent(
259
+ pairs, lambda pair: compliance_judgment(pair[0], pair[1], judge_fn=judge_fn),
260
+ max_concurrency=max_concurrency, timeout_s=timeout_s,
261
+ )
262
+ return [
263
+ result if result is not None
264
+ else assemble_finding(claim, requirement, None) # timed out -> conservative needs_review
265
+ for (claim, requirement), result in zip(pairs, results)
266
+ ]
267
+
268
+
269
+ async def ajudge_pairs(
270
+ pairs: list[tuple[Claim, Requirement]], *, ajudge_fn: AJudgeFn, max_concurrency: int = 8,
271
+ timeout_s: float | None = _JUDGE_TIMEOUT_S, timeout_retries: int = 1,
272
+ ) -> list[ComplianceFinding]:
273
+ """ASYNC-C1 (ADR-0057): the async twin of `judge_pairs` -- run the async judge over many `(claim,
274
+ requirement)` pairs concurrently (asyncio.gather + Semaphore, the parallel-LLM rule), order preserved.
275
+
276
+ Matches `judge_pairs`/`map_concurrent_async` semantics exactly: each judgment is bounded by `timeout_s` (a
277
+ hard wall-clock deadline via `asyncio.timeout`); on the deadline the call is retried up to `timeout_retries`
278
+ times, then the pair becomes `None` -> a conservative needs_review finding. A NON-timeout judge error
279
+ propagates (the same as the sync path), so a genuine bug is never masked as needs_review."""
280
+ semaphore = asyncio.Semaphore(max_concurrency)
281
+
282
+ async def _rule(claim: Claim, requirement: Requirement) -> Optional[JudgeVerdict]:
283
+ if timeout_s is None:
284
+ return await ajudge_fn(claim, requirement) # no wall-clock bound (the seam still bounds each call)
285
+ for attempt in range(timeout_retries + 1):
286
+ try:
287
+ async with asyncio.timeout(timeout_s):
288
+ return await ajudge_fn(claim, requirement)
289
+ except (asyncio.TimeoutError, TimeoutError):
290
+ if attempt >= timeout_retries:
291
+ return None # give up -> conservative needs_review (a stalled provider never hangs the batch)
292
+ return None
293
+
294
+ async def _one(pair: tuple[Claim, Requirement]) -> ComplianceFinding:
295
+ claim, requirement = pair
296
+ async with semaphore: # backpressure
297
+ ruling = await _rule(claim, requirement)
298
+ return assemble_finding(claim, requirement, ruling)
299
+
300
+ return list(await asyncio.gather(*(_one(pair) for pair in pairs)))
301
+
302
+
303
+ def register_compliance_judgment(registry) -> None:
304
+ """Register `compliance_judgment` as an AGENT_SKILL (CC-4): a single grounded LLM judgment act, authored as
305
+ `skills/compliance_judgment/SKILL.md` and applied via the seam. Typed output = `JudgeVerdict`."""
306
+ registry.register(
307
+ "compliance_judgment",
308
+ contract=JudgeVerdict,
309
+ kind="agent_skill",
310
+ display_name="Compliance judgment (claim x requirement -> verdict; authored skill)",
311
+ )
312
+
313
+
314
+ def register_compliance_finding_assembly(registry) -> None:
315
+ """Register `compliance_finding_assembly` (FUNCTION -- deterministic): the skill's raw verdict + the inputs
316
+ -> a cited `ComplianceFinding` (conservative default, both-sided citation from the inputs)."""
317
+ registry.register(
318
+ "compliance_finding_assembly",
319
+ contract=ComplianceFinding,
320
+ kind="function",
321
+ display_name="Compliance finding assembly (verdict + inputs -> cited finding)",
322
+ )