rag-wright 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (184) hide show
  1. rag_wright/__init__.py +13 -0
  2. rag_wright/api/__init__.py +33 -0
  3. rag_wright/api/config.py +59 -0
  4. rag_wright/api/discover.py +70 -0
  5. rag_wright/api/documents.py +39 -0
  6. rag_wright/api/ids.py +31 -0
  7. rag_wright/api/invoke.py +99 -0
  8. rag_wright/api/kg.py +61 -0
  9. rag_wright/api/mcp.py +94 -0
  10. rag_wright/api/usage.py +30 -0
  11. rag_wright/api/workspace.py +85 -0
  12. rag_wright/capabilities/__init__.py +8 -0
  13. rag_wright/capabilities/answer_generator.py +427 -0
  14. rag_wright/capabilities/ard.py +286 -0
  15. rag_wright/capabilities/assertion_extraction.py +79 -0
  16. rag_wright/capabilities/chunk_read.py +58 -0
  17. rag_wright/capabilities/chunk_write.py +163 -0
  18. rag_wright/capabilities/claim_extraction.py +153 -0
  19. rag_wright/capabilities/clause_exception_linking.py +117 -0
  20. rag_wright/capabilities/compliance_judgment.py +322 -0
  21. rag_wright/capabilities/compliance_store.py +87 -0
  22. rag_wright/capabilities/contract_kg_serve.py +156 -0
  23. rag_wright/capabilities/contract_kg_store.py +251 -0
  24. rag_wright/capabilities/dg_extraction.py +585 -0
  25. rag_wright/capabilities/disambiguation.py +163 -0
  26. rag_wright/capabilities/document_parse.py +87 -0
  27. rag_wright/capabilities/document_scope.py +49 -0
  28. rag_wright/capabilities/embedding.py +164 -0
  29. rag_wright/capabilities/embedding_profiles.py +43 -0
  30. rag_wright/capabilities/entity_resolution.py +154 -0
  31. rag_wright/capabilities/fusion.py +64 -0
  32. rag_wright/capabilities/graph_extraction.py +243 -0
  33. rag_wright/capabilities/graph_query.py +73 -0
  34. rag_wright/capabilities/graph_storage.py +111 -0
  35. rag_wright/capabilities/highlight_serve.py +142 -0
  36. rag_wright/capabilities/hybrid_search.py +65 -0
  37. rag_wright/capabilities/invoke.py +31 -0
  38. rag_wright/capabilities/jev_decision.py +38 -0
  39. rag_wright/capabilities/manifests.py +872 -0
  40. rag_wright/capabilities/okf_navigate.py +456 -0
  41. rag_wright/capabilities/parsing.py +286 -0
  42. rag_wright/capabilities/property_boosted_retrieval.py +125 -0
  43. rag_wright/capabilities/query_function_classifier.py +94 -0
  44. rag_wright/capabilities/query_understanding.py +109 -0
  45. rag_wright/capabilities/registry.py +262 -0
  46. rag_wright/capabilities/remote_encoders.py +94 -0
  47. rag_wright/capabilities/requirement_extraction.py +247 -0
  48. rag_wright/capabilities/reranking.py +123 -0
  49. rag_wright/capabilities/retrieval_core.py +126 -0
  50. rag_wright/capabilities/rlm_chunking.py +808 -0
  51. rag_wright/capabilities/rlm_synthesis.py +316 -0
  52. rag_wright/capabilities/scan_quality.py +136 -0
  53. rag_wright/capabilities/span_relevance_judgment.py +191 -0
  54. rag_wright/capabilities/vision_to_text.py +85 -0
  55. rag_wright/capabilities/vlm_ocr.py +85 -0
  56. rag_wright/contracts/__init__.py +6 -0
  57. rag_wright/contracts/chunk.py +79 -0
  58. rag_wright/contracts/compliance.py +303 -0
  59. rag_wright/contracts/contract_meta.py +27 -0
  60. rag_wright/contracts/extraction.py +130 -0
  61. rag_wright/contracts/function.py +167 -0
  62. rag_wright/contracts/function_routing.py +91 -0
  63. rag_wright/contracts/highlight.py +74 -0
  64. rag_wright/contracts/identifiers.py +153 -0
  65. rag_wright/contracts/jurisdiction.py +96 -0
  66. rag_wright/contracts/ontology.py +142 -0
  67. rag_wright/contracts/property.py +201 -0
  68. rag_wright/contracts/provenance.py +78 -0
  69. rag_wright/contracts/query_intent.py +53 -0
  70. rag_wright/contracts/span.py +76 -0
  71. rag_wright/contracts/value_match.py +84 -0
  72. rag_wright/corpus/__init__.py +0 -0
  73. rag_wright/corpus/canonicalize.py +116 -0
  74. rag_wright/corpus/cuad.py +153 -0
  75. rag_wright/corpus/cuad_ingestion.py +72 -0
  76. rag_wright/corpus/document_parser.py +299 -0
  77. rag_wright/corpus/edgar.py +231 -0
  78. rag_wright/corpus/gcs_ingestion.py +120 -0
  79. rag_wright/corpus/http.py +110 -0
  80. rag_wright/corpus/selection.py +152 -0
  81. rag_wright/mcp/__init__.py +11 -0
  82. rag_wright/mcp/compliance_server.py +299 -0
  83. rag_wright/mcp/intra_document_qa_server.py +170 -0
  84. rag_wright/mcp/relational_qa_server.py +171 -0
  85. rag_wright/mcp/session_store.py +64 -0
  86. rag_wright/mcp/typed_property_retrieval_server.py +191 -0
  87. rag_wright/models/__init__.py +8 -0
  88. rag_wright/models/profiles.py +331 -0
  89. rag_wright/models/seam.py +497 -0
  90. rag_wright/models/tag_structured.py +285 -0
  91. rag_wright/models/tracing.py +179 -0
  92. rag_wright/models/usage.py +102 -0
  93. rag_wright/okf/__init__.py +11 -0
  94. rag_wright/okf/compile.py +292 -0
  95. rag_wright/okf/document.py +47 -0
  96. rag_wright/okf/enrich.py +176 -0
  97. rag_wright/okf/links.py +190 -0
  98. rag_wright/okf/lint.py +105 -0
  99. rag_wright/ontology/__init__.py +6 -0
  100. rag_wright/ontology/_generated_template_meta.py +60 -0
  101. rag_wright/ontology/_generated_vocab.py +52 -0
  102. rag_wright/ontology/clause_template.py +964 -0
  103. rag_wright/ontology/codegen.py +84 -0
  104. rag_wright/ontology/compliance_bridge.ttl +186 -0
  105. rag_wright/ontology/contract_bridge.ttl +2685 -0
  106. rag_wright/ontology/contract_taxonomy.py +24 -0
  107. rag_wright/ontology/derive.py +58 -0
  108. rag_wright/ontology/loader.py +435 -0
  109. rag_wright/ontology/packs/ftc_16cfr255.ttl +29 -0
  110. rag_wright/ontology/registry.py +87 -0
  111. rag_wright/ontology/template_introspect.py +100 -0
  112. rag_wright/py.typed +0 -0
  113. rag_wright/reference/__init__.py +2 -0
  114. rag_wright/reference/compliance.py +41 -0
  115. rag_wright/reference/contract_seam.py +123 -0
  116. rag_wright/skills/__init__.py +7 -0
  117. rag_wright/skills/claim_extraction/SKILL.md +47 -0
  118. rag_wright/skills/claim_extraction/__init__.py +1 -0
  119. rag_wright/skills/claim_extraction/template.py +50 -0
  120. rag_wright/skills/compliance_judgment/SKILL.md +59 -0
  121. rag_wright/skills/corpus_ingest/SKILL.md +106 -0
  122. rag_wright/skills/extraction_semantic_judge/SKILL.md +51 -0
  123. rag_wright/skills/extraction_semantic_judge/__init__.py +1 -0
  124. rag_wright/skills/generation/SKILL.md +64 -0
  125. rag_wright/skills/generation/__init__.py +1 -0
  126. rag_wright/skills/generic_compliance_judgment/SKILL.md +58 -0
  127. rag_wright/skills/okf_navigate/SKILL.md +137 -0
  128. rag_wright/skills/requirement_extraction/SKILL.md +47 -0
  129. rag_wright/skills/requirement_extraction/__init__.py +1 -0
  130. rag_wright/skills/requirement_extraction/template.py +50 -0
  131. rag_wright/skills/rlm/SKILL.md +186 -0
  132. rag_wright/skills/rlm/__init__.py +31 -0
  133. rag_wright/skills/rlm/agent.py +292 -0
  134. rag_wright/skills/span_relevance_judgment/SKILL.md +67 -0
  135. rag_wright/skills/vision_to_text/SKILL.md +36 -0
  136. rag_wright/skills/vision_to_text/__init__.py +1 -0
  137. rag_wright/spans/__init__.py +1 -0
  138. rag_wright/spans/boundary.py +78 -0
  139. rag_wright/spans/clause_function_classifier.py +490 -0
  140. rag_wright/spans/clause_kg_extractor.py +337 -0
  141. rag_wright/spans/cuad_labels.py +81 -0
  142. rag_wright/spans/dim_classifier.py +158 -0
  143. rag_wright/spans/dim_fleet.json +411 -0
  144. rag_wright/spans/function_classifier.py +77 -0
  145. rag_wright/spans/function_families.py +62 -0
  146. rag_wright/spans/hybrid_classifier.py +103 -0
  147. rag_wright/spans/legalbert_classifier.py +83 -0
  148. rag_wright/spans/model_capabilities.py +107 -0
  149. rag_wright/spans/new_function_labels.py +111 -0
  150. rag_wright/spans/page_map.py +68 -0
  151. rag_wright/spans/property_extractor.py +365 -0
  152. rag_wright/spans/property_grounding.py +182 -0
  153. rag_wright/spans/reclassify.py +77 -0
  154. rag_wright/spans/scarce_function_labels.py +105 -0
  155. rag_wright/spans/segment.py +341 -0
  156. rag_wright/spans/semantic_judge.py +197 -0
  157. rag_wright/spans/symbolic_validation.py +131 -0
  158. rag_wright/spans/tag_clause_extractor.py +182 -0
  159. rag_wright/store/__init__.py +6 -0
  160. rag_wright/store/arcadedb.py +1135 -0
  161. rag_wright/store/chunk_text.py +66 -0
  162. rag_wright/store/seam.py +213 -0
  163. rag_wright/subgraphs/__init__.py +0 -0
  164. rag_wright/subgraphs/async_ingestion.py +204 -0
  165. rag_wright/subgraphs/compliance_check.py +1042 -0
  166. rag_wright/subgraphs/compliance_ingestion.py +306 -0
  167. rag_wright/subgraphs/contract_ingestion_pipeline.py +999 -0
  168. rag_wright/subgraphs/graph_extraction.py +102 -0
  169. rag_wright/subgraphs/intra_document_qa.py +328 -0
  170. rag_wright/subgraphs/observability.py +140 -0
  171. rag_wright/subgraphs/query_constraint_extraction.py +73 -0
  172. rag_wright/subgraphs/relational_qa.py +165 -0
  173. rag_wright/subgraphs/requirement_extraction.py +137 -0
  174. rag_wright/subgraphs/scaffold.py +65 -0
  175. rag_wright/subgraphs/semantic_chunking.py +183 -0
  176. rag_wright/subgraphs/typed_clause_extraction.py +172 -0
  177. rag_wright/subgraphs/typed_property_retrieval.py +278 -0
  178. rag_wright/util/__init__.py +1 -0
  179. rag_wright/util/concurrent.py +153 -0
  180. rag_wright/util/spacy_model.py +45 -0
  181. rag_wright-0.1.0.dist-info/METADATA +168 -0
  182. rag_wright-0.1.0.dist-info/RECORD +184 -0
  183. rag_wright-0.1.0.dist-info/WHEEL +4 -0
  184. rag_wright-0.1.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,1042 @@
1
+ """CC-6 (compliance §13.3): the `compliance_check` subgraph -- the headline composite of the compliance module.
2
+
3
+ A hardened, query-side LangGraph subgraph on `scaffold.py`: for a subject ad, extract its claims (CC-3),
4
+ retrieve the APPLICABLE requirements (claim scope <-> requirement applicability), judge each `(claim,
5
+ requirement)` pair (CC-4, concurrent), and assemble cited findings + a per-requirement gap matrix + a verdict
6
+ summary. Two design requirements from earlier findings are built in here:
7
+
8
+ - **Section->claim_type applicability map** (the ontology enrichment): a requirement whose extracted
9
+ `applicability_scope` is empty is applied by its FTC section (255.5 material-connections -> endorsement, 255.1
10
+ general -> all, 255.0 definitions -> none). Authored in code referencing `ClaimType` (typos are test failures,
11
+ no drifting .ttl) -- the JUDGE-ONTOLOGY-1 pattern. See [[ontology-lever-vs-extraction-lever]].
12
+ - **Ad-level disclosure aggregation** (the CC-4 residual fix): disclosures are ad-level but claims are per-span,
13
+ so before judging, each claim's disclosures are enriched with the union across the whole ad -- a per-span
14
+ fragment with disc=none is not over-flagged when the ad as a whole discloses.
15
+
16
+ Query-side posture: every node degrades to empty on failure (never crash); the judge's conservative default
17
+ (needs_review) plus per-finding human-gating carry the trust guarantees.
18
+ """
19
+
20
+ from __future__ import annotations
21
+
22
+ import asyncio
23
+ from typing import Any, Awaitable, Callable, Optional, TypedDict
24
+
25
+ from langgraph.graph import END, START, StateGraph
26
+
27
+ from rag_wright.capabilities.compliance_judgment import AJudgeFn, ajudge_pairs
28
+ from rag_wright.capabilities.retrieval_core import _cosine
29
+ from enum import Enum
30
+
31
+ from rag_wright.contracts.compliance import (
32
+ CheckableFact,
33
+ Claim,
34
+ ClaimType,
35
+ ComplianceFinding,
36
+ ComplianceReport,
37
+ Constraint,
38
+ DeonticType,
39
+ Requirement,
40
+ RuleScope,
41
+ Verdict,
42
+ )
43
+ from rag_wright.contracts.provenance import ConfidenceTag
44
+ from rag_wright.ontology.loader import ( # ADR-0066 P4: query-side knowledge from the ontology + FTC domain pack
45
+ load_actor_synonyms,
46
+ load_role_domains,
47
+ load_section_overrides,
48
+ )
49
+ from rag_wright.subgraphs.scaffold import DEFAULT_RETRY, business_span
50
+
51
+ _ALL_CLAIM_TYPES = {c.value for c in ClaimType}
52
+
53
+ # ADR-0066 P4b: the per-section overrides (DEON-8 applicable claim types + DEON-1 rule scope) are AUTHORITATIVE in
54
+ # a DOMAIN PACK ttl (packs/ftc_16cfr255.ttl), not Python literals. To retarget a domain, ship its own pack; nothing
55
+ # FTC-specific is hardcoded here. (FTC finding CC-6: the endorsement guides apply by CONTEXT, not claim_type, so
56
+ # every operative section applies to ALL claim types and only definitions (255.0) is excluded -- now in the pack.)
57
+ _SECTION_RULE_SCOPE_RAW, SECTION_CLAIM_TYPES = load_section_overrides()
58
+ SECTION_RULE_SCOPE: dict[str, RuleScope] = {sec: RuleScope(v) for sec, v in _SECTION_RULE_SCOPE_RAW.items()}
59
+
60
+
61
+ def _section_of(citation: str) -> str:
62
+ """'§ 255.5' -> '255.5' (the section key for the applicability map)."""
63
+ return (citation or "").replace("§", "").strip()
64
+
65
+
66
+ def applicable_claim_types(requirement: Requirement) -> set[str]:
67
+ """DEON-8 (issue 0012): the claim types a requirement applies to. A CURATED override (`SECTION_CLAIM_TYPES`,
68
+ the FTC reference pack) WINS when the requirement's section is pinned there -- FTC behavior is byte-identical
69
+ (context sections apply to ALL claim types; §255.0 to none), and a noisy extracted claim_type can neither
70
+ narrow a context section nor rescue definitions. For ANY OTHER (customer) section the extracted `claim_type`
71
+ scope is LOAD-BEARING: it NARROWS (a pricing-scoped rule does not apply to a health claim); an empty scope is
72
+ recall-first (applies to all). So the KG's claim_type field routes for ANY policy, not just FTC -- the ad-path
73
+ analog of the DEON-6/7 actor gate (a curated override on top of a load-bearing typed field). The real per-
74
+ claim narrowing (most-relevant rule) is still semantic retrieval (Leg-B), the CC-7 refinement."""
75
+ section = _section_of(requirement.citation)
76
+ if section in SECTION_CLAIM_TYPES: # curated FTC override wins (context = all types, definitions = none)
77
+ return SECTION_CLAIM_TYPES[section]
78
+ scope = {c.value for c in requirement.applicability_scope if c.dimension == "claim_type"}
79
+ return scope or _ALL_CLAIM_TYPES # customer policy: extracted scope narrows; empty -> recall-first
80
+
81
+
82
+ def _constraints_by_dimension(constraints: list) -> dict:
83
+ """`[Constraint(dimension, value), ...]` -> {dimension: {values}}."""
84
+ out: dict = {}
85
+ for c in constraints:
86
+ out.setdefault(c.dimension, set()).add(c.value)
87
+ return out
88
+
89
+
90
+ _ROLE_GENERIC = frozenset({"", "party", "anyone", "any", "all", "everyone", "subject", "person", "other"})
91
+
92
+ # DEON-6/7: generic, DOMAIN-agnostic role-synonym normalization -- collapse common variants to one canonical role
93
+ # so the rule side and the subject side align (BGE cosine on bare role words does NOT encode role equivalence:
94
+ # employer~manufacturer 0.68 > advertiser~manufacturer 0.63, so a similarity threshold cannot separate them).
95
+ # An unknown role is KEPT as-is (both sides normalize identically, so an exotic domain still matches on its own
96
+ # term); this is role knowledge, NOT an FTC/corpus hardcode.
97
+ # ADR-0066 P4a: AUTHORITATIVE in compliance_bridge.ttl (cmp:ActorRole skos:altLabel) -- loaded, not a Python
98
+ # literal. To add a role synonym, edit the ttl (a new customer domain extends the role pack, not this code).
99
+ _ACTOR_SYNONYMS: dict[str, str] = load_actor_synonyms()
100
+
101
+ # ADR-0068 (engine issue 0013): the DISJOINTNESS knowledge for the recall-first actor gate -- `{canonical role ->
102
+ # domain}` from the ontology (cmp:roleDomain). AUTHORITATIVE in compliance_bridge.ttl; a customer domain adds its
103
+ # roles' domains in its own pack. Two roles are disjoint iff BOTH are here with DIFFERENT domains.
104
+ _ROLE_DOMAINS: dict[str, str] = load_role_domains()
105
+
106
+
107
+ def canonical_actor(raw: str) -> str:
108
+ """DEON-6/7: normalize an actor ROLE to its canonical form -- collapse a known synonym (manufacturer ->
109
+ advertiser), else keep the role as-is (lower-cased). Applied identically to the rule and subject side, so the
110
+ KG actor gate matches on aligned roles."""
111
+ a = (raw or "").strip().lower()
112
+ return _ACTOR_SYNONYMS.get(a, a)
113
+
114
+
115
+ def _actor_set(scope: list) -> set:
116
+ """The canonical `actor` roles in a scope (a claim's, or the document's aggregated)."""
117
+ return {canonical_actor(c.value) for c in (scope or []) if c.dimension == "actor" and c.value}
118
+
119
+
120
+ def roles_disjoint(a: str, b: str) -> bool:
121
+ """ADR-0068: are two CANONICAL actor roles ontology-DISJOINT? True ONLY when both carry a `cmp:roleDomain` and
122
+ the domains DIFFER (e.g. an advertising role vs a labor role). An unmodelled role (no domain), or two roles in
123
+ the same domain, are NOT disjoint -- the recall-first default is compatible."""
124
+ da, db = _ROLE_DOMAINS.get(a), _ROLE_DOMAINS.get(b)
125
+ return da is not None and db is not None and da != db
126
+
127
+
128
+ def roles_compatible(a: str, b: str) -> bool:
129
+ """ADR-0068: two CANONICAL actor roles are COMPATIBLE (a pair worth judging) unless the ontology makes them
130
+ disjoint -- recall-first. A generic/absent role on either side is always compatible. Replaces exact role
131
+ equality: two different-but-overlapping roles (advertiser vs seller) now match, so a real violation is never
132
+ silently dropped because two independent extractions chose different words for the same party."""
133
+ return not a or not b or a in _ROLE_GENERIC or b in _ROLE_GENERIC or not roles_disjoint(a, b)
134
+
135
+
136
+ def actor_matches(rule_actor: str, subject_actors: set) -> bool:
137
+ """DEON-6/7 + ADR-0068: is the requirement's actor ROLE COMPATIBLE with the subject's (already-canonical)
138
+ actors? RECALL-FIRST on two axes: (1) a generic/absent rule actor or a subject with no actor info never gates
139
+ (True); (2) a specific rule actor matches unless it is ontology-DISJOINT from EVERY subject actor -- so an
140
+ unmodelled or merely-different-but-overlapping role is judged, not dropped (issue 0013)."""
141
+ ra = canonical_actor(rule_actor)
142
+ if not ra or ra in _ROLE_GENERIC or not subject_actors:
143
+ return True # recall-first: nothing to gate on
144
+ return any(roles_compatible(ra, sa) for sa in subject_actors)
145
+
146
+
147
+ def _actor_compatible(a: str, b: str) -> bool:
148
+ """DEON-9 + ADR-0068: the symbolic gate for which permissions can defend which O/F rule -- two CANONICAL actor
149
+ roles are compatible unless ontology-disjoint (recall-first), same predicate as the actor gate."""
150
+ return roles_compatible(a, b)
151
+
152
+
153
+ def constraint_applies(requirement_scope: list, subject_scope: list) -> bool:
154
+ """COMP-APPLIC-1 Increment 0: the DIMENSION-AGNOSTIC applicability matcher. A requirement applies to a subject
155
+ iff, for EVERY dimension the requirement constrains, the subject's value(s) on that dimension INTERSECT the
156
+ requirement's allowed values. A dimension the requirement does NOT constrain, or one the subject does NOT
157
+ carry, never excludes (recall-first). Both scopes are `(dimension, value)` Constraint lists, so ANY domain
158
+ routes with NO new compliance_check code -- the ontology's dimensions are DATA, not per-domain matcher logic.
159
+ (The advertising `applies_to` keeps its own path: its "definitions section applies to nothing" is authored
160
+ doctrine that pure constraint matching does not express -- a domain with such authored routing adds a thin
161
+ wrapper; a domain with pure scope matching adds none.)"""
162
+ req = _constraints_by_dimension(requirement_scope)
163
+ subj = _constraints_by_dimension(subject_scope)
164
+ for dim, allowed in req.items():
165
+ vals = subj.get(dim)
166
+ if vals is not None and not (vals & allowed):
167
+ return False # subject HAS this dimension but with a non-matching value -> excluded
168
+ return True
169
+
170
+
171
+ def applies_to(requirement: Requirement, claim: Claim) -> bool:
172
+ """Does `requirement` apply to `claim`? (the claim's type is in the requirement's applicable claim types)."""
173
+ return claim.claim_type.value in applicable_claim_types(requirement)
174
+
175
+
176
+ class DeonticRoute(str, Enum):
177
+ """DEON-1 (issue 0012): the JUDGE route a rule takes, derived from its DEONTIC TYPE (a KG-typed field), not
178
+ its FTC section number -- so it works for ANY customer policy."""
179
+
180
+ OBLIGATION = "obligation" # breach = ABSENCE -> document-scoped, judged ONCE (a per-sentence judge cannot
181
+ # answer "is it present anywhere?"); always-included (CONTEXT).
182
+ PROHIBITION = "prohibition" # breach = PRESENCE -> per-assertion, where the subject asserts something related.
183
+ PERMISSION = "permission" # cannot be violated standalone -> EXCLUDED from violation-judging (an exception /
184
+ # defense that modifies an O/F rule; linked in DEON-9).
185
+ AMBIGUOUS = "ambiguous" # deontic force unreadable (off-vocab, coerced) -> recall-first: per-assertion + flag.
186
+
187
+
188
+ def deontic_route(requirement: Requirement) -> DeonticRoute:
189
+ """DEON-1: the judge route for a rule, from its deontic type. Precedence: a curated `SECTION_RULE_SCOPE`
190
+ override (a hand-tuned domain pack may still pin scope by section) > an AMBIGUOUS deontic (off-vocab, coerced
191
+ to OBLIGATION but flagged -> recall-first, never trusted as an obligation) > the deontic type itself. FTC-
192
+ agnostic: a customer policy routes by what its rules ARE, not by matching FTC 16 CFR 255 section numbers."""
193
+ override = SECTION_RULE_SCOPE.get(_section_of(requirement.citation))
194
+ if override is RuleScope.CONTEXT:
195
+ return DeonticRoute.OBLIGATION
196
+ if override is RuleScope.CONTENT:
197
+ return DeonticRoute.PROHIBITION
198
+ if requirement.confidence is ConfidenceTag.AMBIGUOUS: # the extractor could not read the deontic force
199
+ return DeonticRoute.AMBIGUOUS
200
+ if requirement.deontic_type is DeonticType.OBLIGATION:
201
+ return DeonticRoute.OBLIGATION
202
+ if requirement.deontic_type is DeonticType.PERMISSION:
203
+ return DeonticRoute.PERMISSION
204
+ return DeonticRoute.PROHIBITION
205
+
206
+
207
+ def rule_scope_of(requirement: Requirement) -> RuleScope:
208
+ """DEON-1: CONTEXT (always-include) for an obligation, else CONTENT (narrow by similarity) -- derived from
209
+ `deontic_route`, so it is DEONTIC-driven, not FTC-section-driven."""
210
+ return RuleScope.CONTEXT if deontic_route(requirement) is DeonticRoute.OBLIGATION else RuleScope.CONTENT
211
+
212
+
213
+ # DEON-2: how much of the subject the obligation judge sees -- the top-N most-relevant passages up to a char
214
+ # budget, NOT the whole document (an obligation is judged once over BOUNDED retrieved evidence, not per-sentence
215
+ # and not by dumping a contract-length document into one prompt).
216
+ OBLIGATION_TOP_N = 5
217
+ OBLIGATION_CHAR_BUDGET = 4000
218
+
219
+
220
+ def _within_budget(claims: list, *, top_n: int, char_budget: int) -> list:
221
+ """Take up to `top_n` claims (already ranked) but stop once the cumulative assertion text exceeds
222
+ `char_budget` -- the bounded evidence window for one obligation judgment. Always keeps at least the first."""
223
+ out: list = []
224
+ used = 0
225
+ for c in claims[:top_n]:
226
+ text = c.assertion_text or ""
227
+ if out and used + len(text) > char_budget:
228
+ break
229
+ out.append(c)
230
+ used += len(text)
231
+ return out
232
+
233
+
234
+ def subject_scope(facts: list) -> list:
235
+ """DEON-5: the document-level SubjectScope -- the deduped union of every fact's per-assertion `scope`
236
+ `Constraint`s (actor + any inferred dimension). Feeds the obligation actor-gate (DEON-7: is the rule's actor
237
+ present in the document at all?) and, per-assertion, the prohibition constraint router (DEON-6). Returns a
238
+ `list[Constraint]`; the actor-set is the values on dimension 'actor'."""
239
+ seen: set = set()
240
+ out: list = []
241
+ for f in facts:
242
+ for c in getattr(f, "scope", None) or []:
243
+ key = (c.dimension, c.value)
244
+ if key not in seen:
245
+ seen.add(key)
246
+ out.append(c)
247
+ return out
248
+
249
+
250
+ def _document_signal_line(claims: list) -> str:
251
+ """DEON-8 (Option 1): the ad-level structured SIGNALS rendered as document content for the obligation judge --
252
+ the disclosure union + whether evidence is referenced, aggregated across ALL claims (a disclosure made
253
+ ANYWHERE in the ad satisfies a disclosure obligation, so the whole-document union matters, not just the top-N
254
+ evidence window). getattr-tolerant so a generic (non-ad) fact contributes nothing -> '' (domain-neutral: the
255
+ engine's obligation retriever stays free of ad concepts, the signals only appear when the facts carry them)."""
256
+ disclosures = sorted({d for c in claims for d in (getattr(c, "disclosures_present", None) or [])})
257
+ evidence = any(getattr(c, "evidence_referenced", False) for c in claims)
258
+ parts: list[str] = []
259
+ if disclosures:
260
+ parts.append("disclosures present in the document: " + "; ".join(disclosures))
261
+ if evidence:
262
+ parts.append("the document references supporting evidence")
263
+ return ("\n\n[DOCUMENT SIGNALS] " + "; ".join(parts) + ".") if parts else ""
264
+
265
+
266
+ def build_obligation_pairs_fn(embedder: Any, *, top_n: int = OBLIGATION_TOP_N,
267
+ char_budget: int = OBLIGATION_CHAR_BUDGET) -> Any:
268
+ """DEON-2: the obligation evidence retriever. For each obligation, embed it and RANK the subject assertions,
269
+ take the top-N most-relevant up to a char budget, and build one bounded evidence `CheckableFact` -> one
270
+ `(evidence_fact, obligation)` pair. Symbolic/vector narrowing (KG deontic route + embeddings) selects the
271
+ small evidence set; the LLM then judges once over it (breach = absence). Claims are embedded ONCE.
272
+
273
+ DEON-8 (Option 1): the ad-level structured SIGNALS (the disclosure union / evidence-referenced) are appended
274
+ to each bundle as document content, so an obligation judged ONCE on the ad path still sees a disclosure made
275
+ anywhere in the ad. getattr-tolerant, so the generic path is unaffected."""
276
+ def obligation_pairs(obligations: list, claims: list, source_doc: str) -> list:
277
+ if not (obligations and claims):
278
+ return []
279
+ # DEON-7: the ACTOR GATE (symbolic, zero LLM) -- an obligation whose bound actor is NOT present in the
280
+ # document's actor-set is out of scope, so it is SKIPPED entirely (no retrieval, no judge call). Robust to
281
+ # role synonyms via `actor_matches`; recall-first (a generic/absent actor never gates).
282
+ doc_actors = _actor_set(subject_scope(claims))
283
+ obligations = [ob for ob in obligations if actor_matches(ob.actor, doc_actors)]
284
+ if not obligations:
285
+ return []
286
+ signal_line = _document_signal_line(claims) # DEON-8: carry the ad-level disclosure/evidence signals
287
+ claim_vecs = [(c, embedder.encode_dense(c.assertion_text)) for c in claims] # embed the subject once
288
+ pairs: list = []
289
+ for ob in obligations:
290
+ ob_vec = embedder.encode_dense(ob.requirement_text)
291
+ ranked = [c for c, _ in sorted(claim_vecs, key=lambda cv: _cosine(ob_vec, cv[1]), reverse=True)]
292
+ evidence = _within_budget(ranked, top_n=top_n, char_budget=char_budget)
293
+ # issue 0044: `assertion_text` is document text ONLY (so the citation stays a quote from the user's
294
+ # doc); the DEON-8 signals ride in `document_signals`, seen by the judge but never cited. The bundle is
295
+ # the top-N spans joined -> ASSEMBLED evidence, flagged so a consumer never renders it as one verbatim.
296
+ text = "\n\n".join(c.assertion_text for c in evidence) or "(empty subject)"
297
+ fact = CheckableFact(fact_id=CheckableFact.make_id(source_doc, ob.requirement_id, text),
298
+ source_doc=source_doc, assertion_text=text,
299
+ document_signals=signal_line, citation_kind="assembled")
300
+ pairs.append((fact, ob))
301
+ return pairs
302
+
303
+ return obligation_pairs
304
+
305
+
306
+ DEFENSE_TOP_N = 3
307
+
308
+
309
+ def build_defense_linker(embedder: Any, requirements: list, *, top_n: int = DEFENSE_TOP_N) -> Any:
310
+ """DEON-9 (issue 0012): the PERMISSION-AS-DEFENSE linker (ADR-0044 pattern, requirement side). For an
311
+ obligation/prohibition rule, return the same-`source` PERMISSIONS that may EXCUSE it (a carve-out/safe-harbor),
312
+ so the judge can rule a legitimate exception COMPLIANT instead of a false violation. Symbolic candidacy: same
313
+ policy source + actor-compatible (canonical, recall-first); ranked by semantic proximity and capped at `top_n`
314
+ -- rank+cap, NOT a fragile similarity threshold. Zero extra LLM: the ONE judge call now reasons over the rule
315
+ plus its linked defenses. Vectors are precomputed ONCE (like `build_select_fn`)."""
316
+ vectors = {r.requirement_id: embedder.encode_dense(r.requirement_text) for r in requirements}
317
+ perms_by_source: dict[str, list] = {}
318
+ for r in requirements:
319
+ if deontic_route(r) is DeonticRoute.PERMISSION:
320
+ perms_by_source.setdefault(r.source, []).append(r)
321
+
322
+ def defenses_for(rule: Any) -> list:
323
+ if deontic_route(rule) is DeonticRoute.PERMISSION:
324
+ return [] # a permission is not judged for violation, so it carries no defenses of its own
325
+ perms = perms_by_source.get(rule.source, [])
326
+ if not perms:
327
+ return []
328
+ ra = canonical_actor(rule.actor)
329
+ cands = [p for p in perms if _actor_compatible(ra, canonical_actor(p.actor))]
330
+ rule_vec = vectors.get(rule.requirement_id, [])
331
+ ranked = sorted(cands, key=lambda p: _cosine(rule_vec, vectors.get(p.requirement_id, [])), reverse=True)
332
+ return ranked[:top_n]
333
+
334
+ return defenses_for
335
+
336
+
337
+ def _with_defenses(requirement: Any, defense_linker: Any) -> Any:
338
+ """DEON-9: attach the linked permissions (rendered `citation: text`) to a rule as query-time `defenses`, so
339
+ the judge renders them as structured exception context. A no-op (returns the rule unchanged) when nothing is
340
+ linked, so an unrelated rule is untouched."""
341
+ if defense_linker is None:
342
+ return requirement
343
+ linked = defense_linker(requirement)
344
+ if not linked:
345
+ return requirement
346
+ return requirement.model_copy(
347
+ update={"defenses": [f"{p.citation}: {p.requirement_text}" for p in linked]})
348
+
349
+
350
+ SelectFn = Callable[[Claim, list], list] # (claim, requirements) -> the narrowed requirements to judge
351
+
352
+
353
+ def _dedup(requirements: list, vectors: dict, threshold: float) -> list:
354
+ """Greedy near-duplicate collapse: keep a requirement unless it is >= `threshold` cosine-similar to one
355
+ already kept (the 3 near-identical §255.5 disclosure rules -> one). Order-preserving."""
356
+ kept: list = []
357
+ for req in requirements:
358
+ vec = vectors.get(req.requirement_id)
359
+ if vec is not None and any(_cosine(vec, vectors[k.requirement_id]) >= threshold for k in kept
360
+ if vectors.get(k.requirement_id) is not None):
361
+ continue
362
+ kept.append(req)
363
+ return kept
364
+
365
+
366
+ def build_select_fn(
367
+ embedder, requirements: list, *, k: int = 5, context_k: int = 3, dedup_threshold: float = 0.92,
368
+ filter_applicability: bool = True, constraint_scope_fn: Any = None
369
+ ) -> SelectFn:
370
+ """CC-8b: build the semantic-narrowing selector. Precomputes each requirement's BGE vector ONCE. Per claim it
371
+ returns the top-`context_k` CONTEXT rules (disclosure -- kept regardless of content so similarity can't miss
372
+ them, Example B) + the top-`k` CONTENT rules (substantiation etc., ranked by cosine to the claim), deduped.
373
+ Both classes are CAPPED so an over-extracted section (§255.5 -> 32 near-identical disclosure rules) collapses
374
+ to a few representatives rather than re-exploding the cross-product. A precision/cost win that keeps the
375
+ context rules (the recall guarantee) while cutting the redundant-rule noise.
376
+
377
+ Three applicability-routing modes (in precedence): `constraint_scope_fn` (COMP-APPLIC-1 Increment 0: generic
378
+ DIMENSION-AGNOSTIC structured routing -- `subject -> [Constraint]`, matched against each requirement's scope by
379
+ `constraint_applies`; ANY domain, no per-domain matcher code) > `filter_applicability=True` (advertising
380
+ claim_type routing via `applies_to`) > `filter_applicability=False` (semantic-only, COMP-VERDICT-GENERIC)."""
381
+ vectors = {r.requirement_id: embedder.encode_dense(r.requirement_text) for r in requirements}
382
+
383
+ def _ranked(reqs: list, claim_vec: list) -> list:
384
+ return sorted(reqs, key=lambda r: _cosine(claim_vec, vectors.get(r.requirement_id, [])), reverse=True)
385
+
386
+ def select(claim: Any, reqs: list) -> list:
387
+ if constraint_scope_fn is not None: # generic structured routing (any domain, ontology-driven, DATA)
388
+ subject_scope = constraint_scope_fn(claim)
389
+ actors = _actor_set(subject_scope) # DEON-6: prohibition gated by (non-actor constraints) AND actor role
390
+ applicable = [r for r in reqs if constraint_applies(r.applicability_scope, subject_scope)
391
+ and actor_matches(r.actor, actors)]
392
+ elif filter_applicability: # advertising claim_type routing
393
+ applicable = [r for r in reqs if applies_to(r, claim)]
394
+ else: # semantic-only (generic verdict)
395
+ applicable = list(reqs)
396
+ claim_vec = embedder.encode_dense(claim.assertion_text)
397
+ context = _dedup(_ranked([r for r in applicable if rule_scope_of(r) is RuleScope.CONTEXT], claim_vec),
398
+ vectors, dedup_threshold)[:context_k]
399
+ content = _dedup(_ranked([r for r in applicable if rule_scope_of(r) is RuleScope.CONTENT], claim_vec),
400
+ vectors, dedup_threshold)[:k]
401
+ return _dedup(context + content, vectors, dedup_threshold)
402
+
403
+ return select
404
+
405
+
406
+ def _actor_gated_pairs(claims: list, prohibitions: list, obligations: list, constraint_scope_fn: Any) -> list[dict]:
407
+ """ADR-0068 (issue 0013): the (assertion|document, rule) pairs the symbolic ACTOR gate SKIPPED before any judge
408
+ call -- recomputed from the SAME module gate (`actor_matches`) the router applies, so the report can state
409
+ honest coverage and a gated pair is never silent. OBLIGATIONS: a rule whose bound actor is ontology-disjoint
410
+ from every document actor (DEON-7, document scope). PROHIBITIONS: per assertion, a rule that PASSES the
411
+ constraint router but is actor-disjoint from the assertion's actors (DEON-6) -- computed only when constraint
412
+ routing is active (the ad path narrows prohibitions by claim_type, not the actor gate, so it reports none).
413
+ Empty unless a disjoint role actually blocked a pair (the recall-first norm)."""
414
+ gated: list[dict] = []
415
+ doc_actors = _actor_set(subject_scope(claims)) if claims else set()
416
+ for ob in obligations:
417
+ if not actor_matches(ob.actor, doc_actors):
418
+ gated.append({"requirement_id": ob.requirement_id, "citation": ob.citation,
419
+ "actor": canonical_actor(ob.actor), "subject_actors": sorted(doc_actors),
420
+ "scope": "document", "claim_id": None})
421
+ if constraint_scope_fn is not None:
422
+ for claim in claims:
423
+ subj = constraint_scope_fn(claim)
424
+ actors = _actor_set(subj)
425
+ for p in prohibitions:
426
+ if constraint_applies(p.applicability_scope, subj) and not actor_matches(p.actor, actors):
427
+ gated.append({"requirement_id": p.requirement_id, "citation": p.citation,
428
+ "actor": canonical_actor(p.actor), "subject_actors": sorted(actors),
429
+ "scope": "assertion", "claim_id": getattr(claim, "fact_id", None)})
430
+ return gated
431
+
432
+
433
+ # ASYNC-C1 (ADR-0057): claims_fn is async (its extraction model call gets a true wall-clock deadline).
434
+ ClaimsFn = Callable[[str, str], Awaitable[list]] # (subject_text, source_doc) -> list[Claim]
435
+ RequirementsFn = Callable[[], list] # () -> list[Requirement]
436
+
437
+
438
+ class CheckState(TypedDict, total=False):
439
+ subject_text: str
440
+ source_doc: str
441
+ claims: list
442
+ ad_disclosures: list
443
+ pairs: list
444
+ gated: list # ADR-0068 (issue 0013): (assertion|document, rule) pairs the actor gate skipped -> the report
445
+ findings: list
446
+ report: ComplianceReport
447
+
448
+
449
+ def _enrich(claim: Claim, ad_disclosures: set[str]) -> Claim:
450
+ """Enrich a claim's disclosures with the ad-level union (the CC-4 fix). claim_id is content-hashed on the
451
+ assertion, not the disclosures, so it is unchanged -- the finding still cites the original claim."""
452
+ if not ad_disclosures:
453
+ return claim
454
+ merged = sorted(set(claim.disclosures_present) | ad_disclosures)
455
+ return claim.model_copy(update={"disclosures_present": merged})
456
+
457
+
458
+ def build_compliance_check(
459
+ *, claims_fn: ClaimsFn, requirements_fn: RequirementsFn, judge_fn: AJudgeFn,
460
+ select_fn: SelectFn | None = None, obligation_pairs_fn: Any = None, defense_linker: Any = None,
461
+ constraint_scope_fn: Any = None, retry_policy: Any = DEFAULT_RETRY
462
+ ):
463
+ """Compile the compliance-check subgraph. All seams are injected for hermetic testing. `select_fn` (CC-8b) is
464
+ the per-claim requirement narrower; when None, every APPLICABLE requirement is judged (the broad default).
465
+
466
+ DEON-1/DEON-2 (issue 0012): when `obligation_pairs_fn` (`(obligations, claims, source) -> [(evidence_fact,
467
+ obligation)]`) is provided, the requirements are split by `deontic_route`: PROHIBITION/AMBIGUOUS rules are
468
+ judged PER-ASSERTION (narrowed by `select_fn`), OBLIGATION rules are judged ONCE each over a BOUNDED retrieved
469
+ evidence bundle (breach = absence, unanswerable per-sentence), and PERMISSION rules are excluded from
470
+ violation-judging. When None (the ad path, until DEON-8), the prior per-assertion pairing is kept.
471
+
472
+ DEON-9: when `defense_linker` (`rule -> [permission]`) is provided, each O/F rule is enriched with the same-
473
+ source PERMISSIONS that may EXCUSE it (attached as query-time `defenses`), passed to that rule's judge as
474
+ structured exception context so a legitimate carve-out is not a false violation. Query-side: each node degrades
475
+ to empty on failure (never crashes)."""
476
+
477
+ async def extract_claims(state: CheckState) -> CheckState:
478
+ with business_span("compliance_check.extract_claims"):
479
+ try:
480
+ claims = await claims_fn(state["subject_text"], state["source_doc"])
481
+ except Exception: # noqa: BLE001 - degrade-to-empty (query-side never crashes)
482
+ return {"claims": [], "ad_disclosures": []}
483
+ # ad-level disclosure union (CC-4 fix). getattr-tolerant: a generic CheckableFact has no disclosures ->
484
+ # empty union -> _enrich is a no-op, so the same pipeline serves both advertising Claims and bare facts.
485
+ ad = sorted({d for c in claims for d in getattr(c, "disclosures_present", [])})
486
+ return {"claims": claims, "ad_disclosures": ad}
487
+
488
+ def retrieve_applicable(state: CheckState) -> CheckState:
489
+ claims = state.get("claims", [])
490
+ ad = set(state.get("ad_disclosures", []))
491
+ with business_span("compliance_check.retrieve_applicable"):
492
+ try:
493
+ requirements = requirements_fn()
494
+ except Exception: # noqa: BLE001 - degrade-to-empty
495
+ return {"pairs": []}
496
+ if defense_linker is not None: # DEON-9: enrich each O/F rule with its same-source permission carve-outs
497
+ requirements = [_with_defenses(r, defense_linker) for r in requirements]
498
+ if obligation_pairs_fn is None: # ad path (until DEON-8): the prior per-assertion pairing
499
+ if select_fn is not None: # CC-8b: semantic narrowing (top-k content + always-include context + dedup)
500
+ pairs = [(_enrich(claim, ad), req) for claim in claims for req in select_fn(claim, requirements)]
501
+ else: # broad default: every applicable requirement
502
+ pairs = [(_enrich(claim, ad), req)
503
+ for claim in claims for req in requirements if applies_to(req, claim)]
504
+ return {"pairs": pairs}
505
+ # DEON-1/2: deontic split -- prohibitions per-assertion, obligations judged ONCE over bounded retrieved
506
+ # evidence, permissions excluded.
507
+ prohibitions = [r for r in requirements
508
+ if deontic_route(r) in (DeonticRoute.PROHIBITION, DeonticRoute.AMBIGUOUS)]
509
+ obligations = [r for r in requirements if deontic_route(r) is DeonticRoute.OBLIGATION]
510
+ pairs = []
511
+ for claim in claims: # prohibitions/ambiguous: per-assertion, where the subject asserts something related
512
+ selected = select_fn(claim, prohibitions) if select_fn is not None else prohibitions
513
+ pairs.extend((_enrich(claim, ad), req) for req in selected)
514
+ if obligations and claims: # obligations: one bounded (evidence, obligation) pair each
515
+ pairs.extend(obligation_pairs_fn(obligations, claims, state["source_doc"]))
516
+ # ADR-0068 (issue 0013): surface what the ACTOR gate skipped, so a symbolic drop is never a silent recall
517
+ # loss (empty is the recall-first norm; a disjoint role that blocked a pair shows up here).
518
+ gated = _actor_gated_pairs(claims, prohibitions, obligations, constraint_scope_fn)
519
+ return {"pairs": pairs, "gated": gated}
520
+
521
+ async def judge(state: CheckState) -> CheckState:
522
+ pairs = state.get("pairs", [])
523
+ if not pairs:
524
+ return {"findings": []}
525
+ with business_span("compliance_check.judge"):
526
+ # concurrent (gather + semaphore); conservative default inside (a timed-out pair -> needs_review)
527
+ return {"findings": await ajudge_pairs(pairs, ajudge_fn=judge_fn)}
528
+
529
+ def assemble(state: CheckState) -> CheckState:
530
+ findings: list[ComplianceFinding] = state.get("findings", [])
531
+ summary: dict[str, int] = {}
532
+ for f in findings:
533
+ summary[f.verdict.value] = summary.get(f.verdict.value, 0) + 1
534
+ # gap matrix: one row per requirement that was checked, rolled up to its worst verdict
535
+ rank = {Verdict.VIOLATION: 3, Verdict.NEEDS_REVIEW: 2, Verdict.COMPLIANT: 1}
536
+ by_req: dict[str, dict] = {}
537
+ for f in findings:
538
+ row = by_req.setdefault(f.requirement_id, {
539
+ "requirement_id": f.requirement_id, "citation": f.citation_requirement.split(" (")[0],
540
+ "verdict": f.verdict, "claims_checked": 0})
541
+ row["claims_checked"] += 1
542
+ if rank[f.verdict] > rank[row["verdict"]]:
543
+ row["verdict"] = f.verdict
544
+ gap_matrix = [{**r, "verdict": r["verdict"].value} for r in by_req.values()]
545
+ report = ComplianceReport(
546
+ source_doc=state["source_doc"], findings=findings, summary=summary, gap_matrix=gap_matrix,
547
+ gated_pairs=state.get("gated", [])) # ADR-0068: honest coverage -- the actor gate is never silent
548
+ return {"report": report}
549
+
550
+ g = StateGraph(CheckState)
551
+ g.add_node("extract_claims", extract_claims, retry_policy=retry_policy)
552
+ g.add_node("retrieve_applicable", retrieve_applicable, retry_policy=retry_policy)
553
+ g.add_node("judge", judge, retry_policy=retry_policy)
554
+ g.add_node("assemble", assemble)
555
+ g.add_edge(START, "extract_claims")
556
+ g.add_edge("extract_claims", "retrieve_applicable")
557
+ g.add_edge("retrieve_applicable", "judge")
558
+ g.add_edge("judge", "assemble")
559
+ g.add_edge("assemble", END)
560
+ return g.compile()
561
+
562
+
563
+ def _requirement_from_row(row: dict) -> Requirement:
564
+ """Reconstruct a `Requirement` from a stored row (all_requirements), parsing applicability_json back to
565
+ constraints. Lenient: a bad row would raise, but the store wrote validated contracts."""
566
+ import json
567
+
568
+ from rag_wright.contracts.compliance import DeonticType, Severity
569
+ from rag_wright.contracts.provenance import ConfidenceTag
570
+
571
+ scope = [Constraint(dimension=d, value=v) for d, v in json.loads(row.get("applicability_json") or "[]")]
572
+ sev = row.get("severity") or None
573
+ _bbox = row.get("bbox") # issue 0043: best-effort [l,t,r,b] JSON string -> tuple, else None
574
+ bbox = tuple(json.loads(_bbox)) if _bbox else None
575
+ return Requirement(
576
+ requirement_id=row["requirement_id"], source=row["source"], citation=row["citation"],
577
+ deontic_type=DeonticType(row["deontic_type"]), actor=row["actor"],
578
+ applicability_scope=scope, requirement_text=row["requirement_text"],
579
+ evidence_standard=row.get("evidence_standard") or None,
580
+ severity=Severity(sev) if sev else None,
581
+ pages=[int(p) for p in (row.get("pages") or [])], bbox=bbox, # issue 0043: policy page provenance
582
+ confidence=ConfidenceTag(row.get("confidence") or "EXTRACTED"))
583
+
584
+
585
+ class UnknownComplianceSourceError(ValueError):
586
+ """Issue 0007: a `sources` filter named a policy `source` that has no requirements in the store. Distinct from
587
+ a zero-requirement check (which a caller may treat as `not_checked`): naming a policy that does not exist is a
588
+ caller error, surfaced explicitly rather than silently matching nothing. Carries `.unknown` and `.present`."""
589
+
590
+ def __init__(self, unknown: list[str], present: list[str]) -> None:
591
+ self.unknown = unknown
592
+ self.present = present
593
+ super().__init__(f"unknown compliance source(s): {unknown}; present in store: {present}")
594
+
595
+
596
+ def _validate_sources(store: Any, sources: Optional[list[str]]) -> None:
597
+ """SEG-7a: validate named policy `sources` against the store EARLY -- so an unknown source raises
598
+ `UnknownComplianceSourceError` BEFORE the expensive parse + assertion extraction, never wasting that work.
599
+ `None` (whole store) is not validated (a store without `requirement_sources()` still works)."""
600
+ if sources is None:
601
+ return
602
+ present = store.requirement_sources()
603
+ unknown = sorted(set(sources) - present)
604
+ if unknown:
605
+ raise UnknownComplianceSourceError(unknown, sorted(present))
606
+
607
+
608
+ def _load_requirements(store: Any, sources: Optional[list[str]] = None) -> list[Requirement]:
609
+ """Issue 0007: load the Requirement rows the check runs against, optionally scoped to named policy `source`s.
610
+
611
+ `sources=None` -> the whole store (unchanged; `requirement_sources()` is NOT consulted, so a store without it
612
+ still works). A list -> validate the names against `store.requirement_sources()` (an unknown one raises
613
+ `UnknownComplianceSourceError`, not a silent empty match), then load ONLY those via the DB-side filter
614
+ (`store.all_requirements(sources=...)`). An empty list is a valid scope-to-nothing -> zero requirements."""
615
+ if sources is None:
616
+ rows = store.all_requirements()
617
+ else:
618
+ unknown = sorted(set(sources) - store.requirement_sources())
619
+ if unknown:
620
+ raise UnknownComplianceSourceError(unknown, sorted(store.requirement_sources()))
621
+ rows = store.all_requirements(sources=list(sources))
622
+ return [_requirement_from_row(r) for r in rows]
623
+
624
+
625
+ def production_compliance_check(
626
+ store: Any, *, extract_model: Any, judge_model_id: str, embedder: Any = None, k: int = 5,
627
+ sources: Optional[list[str]] = None, claims_fn: Any = None,
628
+ ):
629
+ """Wire the real capabilities: claims = claim_extraction (CC-3), requirements = the store's Requirement KG
630
+ (CC-5), judge = the Granite compliance judge (CC-4). Requirements are loaded ONCE here; when an `embedder` is
631
+ given, CC-8b semantic narrowing is enabled (top-k content + always-include context + dedup), else broad.
632
+
633
+ UNIFY-F: `claims_fn` (async `(text, source) -> [Claim]`) overrides the default whole-text extractor -- the ad
634
+ entrypoint injects PRECOMPUTED per-section claims through it (parsed once via the shared front-end)."""
635
+ from rag_wright.capabilities.claim_extraction import aclaim_extraction
636
+ from rag_wright.capabilities.compliance_judgment import build_acompliance_judge_fn
637
+
638
+ requirements = _load_requirements(store, sources)
639
+ select_fn = build_select_fn(embedder, requirements, k=k) if embedder is not None else None
640
+ # DEON-8 (issue 0012): the ad path gets the SAME deontic split as the generic path when an embedder is
641
+ # available -- OBLIGATIONS judged ONCE over bounded, actor-gated evidence (not per-sentence), PROHIBITIONS
642
+ # per-assertion (claim_type-routed via select_fn), PERMISSIONS excluded. Without an embedder (no semantic
643
+ # narrowing) the prior per-assertion pairing is kept (back-compat).
644
+ obligation_pairs_fn = build_obligation_pairs_fn(embedder) if embedder is not None else None
645
+ # DEON-9: permission carve-outs linked as defenses to the O/F rules they modify (needs the embedder for ranking).
646
+ defense_linker = build_defense_linker(embedder, requirements) if embedder is not None else None
647
+
648
+ async def _default_claims_fn(text: str, source: str) -> list:
649
+ return await aclaim_extraction(text, model=extract_model, source_doc=source)
650
+
651
+ return build_compliance_check(
652
+ claims_fn=claims_fn or _default_claims_fn,
653
+ requirements_fn=lambda: requirements,
654
+ judge_fn=build_acompliance_judge_fn(judge_model_id),
655
+ select_fn=select_fn,
656
+ obligation_pairs_fn=obligation_pairs_fn,
657
+ defense_linker=defense_linker,
658
+ )
659
+
660
+
661
+ _CLAIM_EXTRACT_ATTEMPTS = 3 # bounded retries for a per-chunk ad claim extraction (recover a transient docling blip)
662
+
663
+
664
+ async def _aextract_ad_claims(chunks: list[str], source_doc: str, extract_model: Any,
665
+ *, aclaim_fn: Any = None, max_concurrency: int = 4) -> list:
666
+ """SEG-7b: the advertising claim extractor over the SAME semantic CHUNKS as the generic path -- extract typed
667
+ `Claim`s PER CHUNK (concurrently, per the parallel-LLM rule), re-indexed globally for unique ids. The
668
+ typed-Claim tail (claim_type / disclosures / routing) is UNTOUCHED. The structural locator (§/¶/bullet) is
669
+ attached AFTER, by `attach_structural_locators` (SEG-4, verbatim match), same as the generic path. `aclaim_fn`
670
+ is injected for hermetic tests."""
671
+ from rag_wright.capabilities.claim_extraction import aclaim_extraction
672
+ from rag_wright.contracts.compliance import Claim
673
+
674
+ fn = aclaim_fn or aclaim_extraction
675
+ sem = asyncio.Semaphore(max_concurrency)
676
+
677
+ async def _one(chunk: str) -> list:
678
+ text = (chunk or "").strip()
679
+ if not text:
680
+ return []
681
+ async with sem:
682
+ # PARTIAL-CAUSE-1 (compliance parity): the ad path extracts claims OUTSIDE the retry graph, so wrap
683
+ # the per-chunk call in a bounded retry. docling-graph's `ExtractionFailed` is raised on ANY logged
684
+ # error incl. TRANSIENT blips (empty content / gleaning / rate-limit / timeout), so retrying is what
685
+ # recovers them (the same fix as the contract clause extractor). A PERSISTENT failure re-raises --
686
+ # surfaced loudly, a chunk's claims are never silently dropped.
687
+ last_exc: Optional[BaseException] = None
688
+ for _attempt in range(_CLAIM_EXTRACT_ATTEMPTS):
689
+ try:
690
+ return await fn(text, model=extract_model, source_doc=source_doc)
691
+ except Exception as exc: # noqa: BLE001 - transient docling/LLM error -> retry; persistent -> raise
692
+ last_exc = exc
693
+ raise last_exc # type: ignore[misc] # persistent failure after retries (never None here)
694
+
695
+ per_chunk = await asyncio.gather(*(_one(c) for c in chunks))
696
+ claims = [c for group in per_chunk for c in group]
697
+ for i, c in enumerate(claims): # global re-index -> unique claim ids across chunks
698
+ c.fact_id = Claim.make_id(source_doc, i, c.assertion_text)
699
+ return claims
700
+
701
+
702
+ async def run_ad_compliance_check(
703
+ subject_text: Optional[str] = None, source_doc: str = "", *, store: Any, extract_model: Any,
704
+ judge_model_id: str, embedder: Any = None, k: int = 5, sources: Optional[list[str]] = None,
705
+ name: Optional[str] = None, data: Optional[bytes] = None, discoverer: Any = None, aclaim_fn: Any = None,
706
+ doc: Any = None,
707
+ ) -> ComplianceReport:
708
+ """The ADVERTISING compliance path (the subgraph behind the `check_ad_compliance` MCP tool; the generic
709
+ counterpart is `run_generic_compliance_verdict`) -> a cited `ComplianceReport`. SEG-7b: accepts EITHER a pasted
710
+ `subject_text` OR an uploaded ad (`name` + raw `data` bytes), and runs the SAME semantic front-end as the
711
+ generic path -- parse -> semantic chunk -> per-chunk typed-`Claim` extraction -> attach structural locators
712
+ (§/¶/bullet) -> judge. The typed-Claim tail (claim_type / disclosure routing) is UNCHANGED; a scanned ad's
713
+ unreadable pages surface on `report.ocr_unreadable_pages` (SEG-6). `sources` (0007) scopes to named policies;
714
+ `discoverer`/`aclaim_fn`/`doc` inject for tests."""
715
+ _validate_sources(store, sources) # reject an unknown policy BEFORE the expensive parse + extraction
716
+ parsed_doc, unreadable = await _aparse_subject_any(text=subject_text, name=name, data=data, doc=doc)
717
+ chunks = await subject_chunks(parsed_doc, discoverer=discoverer)
718
+ claims = await _aextract_ad_claims(chunks, source_doc, extract_model, aclaim_fn=aclaim_fn)
719
+ attach_structural_locators(claims, parsed_doc) # SEG-4: verbatim-match each claim to its docling element
720
+
721
+ async def _precomputed_claims_fn(_text: str, _source: str) -> list:
722
+ return claims # extracted once, per-chunk, above
723
+
724
+ graph = production_compliance_check(
725
+ store, extract_model=extract_model, judge_model_id=judge_model_id, embedder=embedder, k=k, sources=sources,
726
+ claims_fn=_precomputed_claims_fn)
727
+ subject_text_joined = "\n\n".join(c.assertion_text for c in claims)
728
+ out = await graph.ainvoke({"subject_text": subject_text_joined, "source_doc": source_doc})
729
+ report = out["report"]
730
+ report.ocr_unreadable_pages = unreadable # SEG-6: the ad path surfaces the OCR PARTIAL too
731
+ return report
732
+
733
+
734
+ def production_generic_compliance_check(store: Any, *, judge_model_id: str, embedder: Any, k: int = 8,
735
+ sources: Optional[list[str]] = None, facts_fn: Any):
736
+ """COMP-VERDICT-GENERIC: wire the DOMAIN-AGNOSTIC verdict path -- generic subject facts (no claim_type),
737
+ SEMANTIC-ONLY requirement narrowing (`filter_applicability=False`, no domain applicability ontology needed),
738
+ and the GENERIC judge (text-only). Gives a cited LLM verdict in ANY compliance domain; enrichment
739
+ (COMP-APPLIC-1) only ADDS structured precision on top. `embedder` is required (semantic retrieval is the
740
+ narrowing here). `sources` (issue 0007) optionally scopes the check to named policy `source`s (None = the
741
+ whole store). `facts_fn` (required) is the async-adapted producer `(text, source) -> [CheckableFact]`; the
742
+ caller (`run_subject_compliance_verdict`) precomputes the SEMANTIC facts and injects them here (SEG-7a)."""
743
+ from rag_wright.capabilities.compliance_judgment import build_ageneric_judge_fn
744
+
745
+ requirements = _load_requirements(store, sources)
746
+ # DEON-6: prohibitions narrow by the dimension-agnostic constraint router -- a claim's inferred scope (its
747
+ # actor, DEON-5) matched against each requirement's effective scope. Recall-first: a claim with no scope, or a
748
+ # requirement with a generic actor, is not excluded.
749
+ constraint_scope_fn = lambda claim: getattr(claim, "scope", None) or [] # noqa: E731 (a claim's inferred scope)
750
+ select_fn = build_select_fn(embedder, requirements, k=k, filter_applicability=False,
751
+ constraint_scope_fn=constraint_scope_fn)
752
+
753
+ async def _claims_fn(text: str, source: str) -> list:
754
+ return facts_fn(text, source) # precomputed semantic facts, adapted to the async claims seam
755
+
756
+ return build_compliance_check(
757
+ claims_fn=_claims_fn,
758
+ requirements_fn=lambda: requirements,
759
+ judge_fn=build_ageneric_judge_fn(judge_model_id),
760
+ select_fn=select_fn,
761
+ obligation_pairs_fn=build_obligation_pairs_fn(embedder), # DEON-2: bounded per-obligation evidence
762
+ defense_linker=build_defense_linker(embedder, requirements), # DEON-9: permission carve-outs as defenses
763
+ constraint_scope_fn=constraint_scope_fn, # ADR-0068: report the actor-gated prohibition pairs (issue 0013)
764
+ )
765
+
766
+
767
+ async def run_generic_compliance_verdict(
768
+ subject_text: str, source_doc: str, *, store: Any, judge_model_id: str, embedder: Any, k: int = 8,
769
+ sources: Optional[list[str]] = None, extract_model: Any = None, discoverer: Any = None,
770
+ aextract_fn: Any = None,
771
+ ) -> ComplianceReport:
772
+ """COMP-VERDICT-GENERIC: a domain-agnostic compliance verdict for a free-text subject against the Requirement
773
+ KG -> a cited `ComplianceReport`. Works with NO domain applicability enrichment (the always-answer guarantee).
774
+
775
+ `sources` (issue 0007) optionally scopes the check to named policy `source`s -- None checks against the whole
776
+ store; a list checks against ONLY those policies; `[]` scopes to nothing; an unknown name raises
777
+ `UnknownComplianceSourceError`.
778
+
779
+ SEG-7a: a thin shim over `run_subject_compliance_verdict` (text mode). The paste is parsed through docling and
780
+ run through the SAME semantic pipeline as an upload (chunk -> verbatim assertion extraction -> locator ->
781
+ judge); a structureless paste yields per-assertion facts with NO `§` locator. `extract_model`/`discoverer`/
782
+ `aextract_fn` are passed through (the latter two inject for tests)."""
783
+ return await run_subject_compliance_verdict(
784
+ source_doc, store=store, judge_model_id=judge_model_id, embedder=embedder, k=k, sources=sources,
785
+ text=subject_text, extract_model=extract_model, discoverer=discoverer, aextract_fn=aextract_fn)
786
+
787
+
788
+ _HEADING_KINDS = frozenset({"section_header", "title", "field_heading"})
789
+
790
+
791
+ def _merge_wrapped_items(raw: list[tuple[str, str]]) -> list[tuple[str, str]]:
792
+ """SEG-4: coalesce docling's line-split of a wrapped paragraph back into ONE logical element. A line-based
793
+ backend (markdown, a hard-wrapped .txt) emits each physical line as a separate `text` item; a mid-sentence
794
+ line break is a soft-wrap, NOT a paragraph boundary. Signal: the previous same-kind body item does NOT end
795
+ with sentence-terminal punctuation (`.`/`!`/`?`) -> the current item continues it, so merge. Headings never
796
+ merge; a line ending in terminal punctuation starts a new element (a genuine paragraph break)."""
797
+ merged: list[list[str]] = []
798
+ for kind, text in raw:
799
+ if (kind not in _HEADING_KINDS and text and merged
800
+ and merged[-1][0] == kind and merged[-1][1] and merged[-1][1][-1] not in ".!?"):
801
+ merged[-1][1] = f"{merged[-1][1]} {text}" # soft-wrap continuation of the same logical element
802
+ else:
803
+ merged.append([kind, text])
804
+ return [(k, t) for k, t in merged]
805
+
806
+
807
+ def _item_provenance(parsed_doc: Any) -> list[dict]:
808
+ """SEG-4: docling items -> per-LOGICAL-ELEMENT structural provenance `{text, section, element_kind,
809
+ element_ordinal}`. Line-wrapped paragraphs are merged first (`_merge_wrapped_items`) so ordinals count real
810
+ paragraphs, not physical lines. `section` = the enclosing section number (shared `_section_number`; None
811
+ before the first heading -> a flat doc stays section-less). Ordinals count WITHIN a section, PER KIND (¶ for
812
+ body text, bullet for list items), reset at each heading."""
813
+ from rag_wright.corpus.document_parser import _section_number
814
+
815
+ raw: list[tuple[str, str]] = []
816
+ for item in getattr(parsed_doc, "texts", []) or []:
817
+ lab = getattr(item, "label", "")
818
+ kind = str(getattr(lab, "value", lab) or "") # DocItemLabel enum -> its value; a plain string stays as-is
819
+ text = (getattr(item, "text", "") or "").strip()
820
+ if not text and kind not in _HEADING_KINDS:
821
+ continue # empty body item: nothing to locate or count
822
+ raw.append((kind, text))
823
+
824
+ out: list[dict] = []
825
+ section: Optional[str] = None
826
+ n_sections = 0
827
+ para_ord = 0
828
+ bullet_ord = 0
829
+ for kind, text in _merge_wrapped_items(raw):
830
+ if kind in _HEADING_KINDS: # a heading opens a new section and resets the within-section ordinals
831
+ n_sections += 1
832
+ section = _section_number(text, n_sections)
833
+ para_ord = bullet_ord = 0
834
+ out.append({"text": text, "section": section, "element_kind": kind, "element_ordinal": None})
835
+ continue
836
+ if kind == "list_item":
837
+ bullet_ord += 1
838
+ ordinal: Optional[int] = bullet_ord
839
+ else:
840
+ para_ord += 1
841
+ ordinal = para_ord
842
+ out.append({"text": text, "section": section, "element_kind": kind, "element_ordinal": ordinal})
843
+ return out
844
+
845
+
846
+ def _norm_ws(s: str) -> str:
847
+ return " ".join((s or "").split())
848
+
849
+
850
+ def attach_structural_locators(facts: list, parsed_doc: Any) -> list:
851
+ """SEG-4: stamp each `CheckableFact` with the structural locator of the docling element its VERBATIM assertion
852
+ came from (`section`, `element_kind`, `element_ordinal`) -> `locator()` renders `§ N ¶M` / `§ N · bullet M`.
853
+
854
+ Robust to real-world parsing: docling can split a soft-wrapped paragraph (or a hard-wrapped .txt) into several
855
+ consecutive `text` items, so an assertion may SPAN items. We therefore match against the whitespace-normalized
856
+ CONCATENATION of the body items (each item's char range recorded), find the assertion, and attribute it to the
857
+ item where it STARTS -- so a cross-item assertion is located, never dropped. Unmatched (genuinely absent /
858
+ heavily paraphrased) stays unlocated, still citable by its text. Line-wrapped paragraphs are merged in
859
+ `_item_provenance` so ordinals count real paragraphs, not physical lines. Mutates + returns `facts`."""
860
+ body = [p for p in _item_provenance(parsed_doc) if _norm_ws(p["text"])]
861
+ concat = ""
862
+ ranges: list[tuple[int, int, dict]] = [] # (start, end, provenance) in the normalized concatenation
863
+ for p in body:
864
+ t = _norm_ws(p["text"])
865
+ start = len(concat)
866
+ concat += t + " "
867
+ ranges.append((start, start + len(t), p))
868
+ for f in facts:
869
+ needle = _norm_ws(f.assertion_text)
870
+ if not needle:
871
+ continue
872
+ pos = concat.find(needle)
873
+ if pos < 0:
874
+ continue
875
+ for start, end, p in ranges: # attribute to the item where the assertion STARTS
876
+ if start <= pos < end:
877
+ f.section = p["section"]
878
+ f.element_kind = p["element_kind"]
879
+ f.element_ordinal = p["element_ordinal"]
880
+ break
881
+ return facts
882
+
883
+
884
+ async def aextract_subject_facts(chunks: list[str], *, source_doc: str, model: Any, aextract_fn: Any = None,
885
+ max_concurrency: int = 4) -> list:
886
+ """SEG-3: extract the checkable assertions (verbatim) from each subject CHUNK CONCURRENTLY (semaphore, per the
887
+ parallel-LLM rule) -> `CheckableFact`s, re-indexed globally so `fact_id`s are unique across chunks.
888
+ Domain-neutral (`CheckableFact`, no `claim_type` -- that is the ad path). The structural locator
889
+ (section / ¶ / bullet) is attached later, in SEG-4. `aextract_fn` is injected for hermetic tests."""
890
+ from rag_wright.capabilities.assertion_extraction import aassertion_extraction
891
+ from rag_wright.capabilities.dg_extraction import aextract_parties
892
+
893
+ fn = aextract_fn or aextract_parties
894
+ sem = asyncio.Semaphore(max_concurrency)
895
+
896
+ async def _one(chunk: str) -> list:
897
+ async with sem:
898
+ return await aassertion_extraction(chunk, model=model, source_doc=source_doc, aextract_fn=fn)
899
+
900
+ per_chunk = await asyncio.gather(*(_one(c) for c in chunks))
901
+ facts = [f for group in per_chunk for f in group]
902
+ for i, f in enumerate(facts): # global re-index -> unique fact_ids across chunks
903
+ f.fact_id = CheckableFact.make_id(source_doc, i, f.assertion_text)
904
+ return facts
905
+
906
+
907
+ async def subject_chunks(parsed_doc: Any, *, discoverer: Any = None) -> list[str]:
908
+ """SEG-2/SEG-5: semantically chunk a parsed subject document into coherent chunk texts, via the SAME shared
909
+ chunker as ingestion (`achunk_texts`; no cache/summarize -- the subject is transient). The default discoverer
910
+ is `StructuralModelFallbackDiscoverer` -- the exact discoverer PRODUCTION INGESTION uses: structural boundaries
911
+ first (no model for a structured doc), a BOUNDED per-section model refinement only for an over-cap section, so
912
+ cost never scales with document length (no size bottleneck) and there is NO subject-specific large-doc code.
913
+ `parsed_doc` is the docling document (exposes `.texts`). RLM is a future escalation, as on ingestion. SEG-3
914
+ extracts the checkable assertions from each returned chunk."""
915
+ from rag_wright.capabilities.rlm_chunking import achunk_texts
916
+
917
+ return await achunk_texts(parsed_doc, discoverer=discoverer)
918
+
919
+
920
+ async def semantic_subject_facts(parsed_doc: Any, *, source_doc: str, model: Any = None, discoverer: Any = None,
921
+ aextract_fn: Any = None) -> list:
922
+ """SEG-7a: the SUBJECT fact producer -- the composed semantic pipeline that supersedes the old regex
923
+ sections producer. Chunk the parsed doc (SEG-2/SEG-5, the production ingestion discoverer) -> extract the
924
+ checkable assertions VERBATIM per chunk (SEG-3) -> attach each to its docling element for the structural
925
+ locator (SEG-4). Returns `CheckableFact`s cited "doc § {section} ¶{n}: {verbatim}" (or just the span for a
926
+ flat doc). `model` (assertion extractor) defaults to the production extraction model; `discoverer`/`aextract_fn`
927
+ are injected for hermetic tests (no model/parse)."""
928
+ m = model
929
+ if m is None and aextract_fn is None:
930
+ from rag_wright.capabilities.dg_extraction import default_extraction_model
931
+
932
+ m = default_extraction_model("subject-assert")
933
+ chunks = await subject_chunks(parsed_doc, discoverer=discoverer)
934
+ facts = await aextract_subject_facts(chunks, source_doc=source_doc, model=m, aextract_fn=aextract_fn)
935
+ attach_structural_locators(facts, parsed_doc)
936
+ return facts
937
+
938
+
939
+ async def _aparse_subject_any(*, text: Optional[str], name: Optional[str], data: Optional[bytes],
940
+ doc: Any = None) -> tuple[Any, list[int]]:
941
+ """SEG-7a: parse ANY subject input to a docling document + OCR unreadable pages, for the uniform semantic
942
+ pipeline. `doc` (a test injection) is returned as-is. Upload (`data`): `aparse_subject` (tiered OCR, captures
943
+ unreadable pages). Paste (`text`): parsed through docling as `.txt` bytes (decision A -- uniform semantic
944
+ handling, no OCR pages), NOT the old short-circuit."""
945
+ if doc is not None:
946
+ return doc, []
947
+ if data is not None:
948
+ return await aparse_subject(name, data)
949
+ if text is not None:
950
+ from rag_wright.corpus.document_parser import aparse_document_bytes
951
+
952
+ return await aparse_document_bytes("subject.txt", text.encode("utf-8")), []
953
+ raise ValueError("run_subject_compliance_verdict needs either text= or (name=, data=)")
954
+
955
+
956
+ async def aparse_subject(name: str, data: bytes, *, parser: Any = None) -> tuple[Any, list[int]]:
957
+ """SEG-6: parse an uploaded subject ONCE through the tiered OCR chokepoint (`TieredOCRParser`, the SAME OCR as
958
+ ingestion), returning `(docling_document, ocr_unreadable_pages)`. The unreadable pages (a degraded scan the
959
+ VLM still could not read) are captured from the tiered parser's report so they can surface on the
960
+ `ComplianceReport` -- a verdict is never silently based on half-read text. `parser` injected for tests."""
961
+ from rag_wright.capabilities.parsing import TieredOCRParser
962
+ from rag_wright.corpus.document_parser import aparse_document_bytes
963
+
964
+ tiered = parser if parser is not None else TieredOCRParser()
965
+ document = await aparse_document_bytes(name, data, parser=tiered)
966
+ unreadable = list(getattr(getattr(tiered, "report", None), "unreadable_pages", []) or [])
967
+ return document, unreadable
968
+
969
+
970
+ async def run_subject_compliance_verdict(
971
+ source_doc: str, *, store: Any, judge_model_id: str, embedder: Any, k: int = 8,
972
+ sources: Optional[list[str]] = None, text: Optional[str] = None, name: Optional[str] = None,
973
+ data: Optional[bytes] = None, extract_model: Any = None, discoverer: Any = None, aextract_fn: Any = None,
974
+ doc: Any = None,
975
+ ) -> ComplianceReport:
976
+ """SEG-7a: the ONE subject-compliance front-end. Accepts EITHER a pasted `text` OR an uploaded document
977
+ (`name` + raw `data` bytes: PDF/DOCX/HTML/TXT), and runs the SEMANTIC pipeline uniformly: parse -> semantic
978
+ chunk (the production ingestion discoverer) -> extract the checkable assertions VERBATIM per chunk -> attach
979
+ each to its docling element -> judge. Findings cite "doc § {section} ¶{n}: {verbatim}" (or just the span for a
980
+ structureless subject). The subject is TRANSIENT (parsed/checked, never written to the store); a scanned
981
+ subject's unreadable pages surface on `report.ocr_unreadable_pages` (SEG-6).
982
+
983
+ `extract_model` (assertion extractor) defaults to the production extraction model. `sources` (0007) scopes to
984
+ named policies (unknown -> `UnknownComplianceSourceError`). `discoverer`/`aextract_fn`/`doc` are injected for
985
+ hermetic tests. (SEG-7a replaced the old regex sections/granularity dial with the semantic producer.)"""
986
+ _validate_sources(store, sources) # SEG-7a: reject an unknown policy BEFORE the expensive parse + extraction
987
+ parsed_doc, unreadable = await _aparse_subject_any(text=text, name=name, data=data, doc=doc)
988
+ facts = await semantic_subject_facts(
989
+ parsed_doc, source_doc=source_doc, model=extract_model, discoverer=discoverer, aextract_fn=aextract_fn)
990
+ graph = production_generic_compliance_check(
991
+ store, judge_model_id=judge_model_id, embedder=embedder, k=k, sources=sources,
992
+ facts_fn=lambda _text, _source: facts) # precomputed semantic subject facts
993
+ subject_text = "\n\n".join(f.assertion_text for f in facts)
994
+ out = await graph.ainvoke({"subject_text": subject_text, "source_doc": source_doc})
995
+ report = out["report"]
996
+ report.ocr_unreadable_pages = unreadable # SEG-6: surface the OCR PARTIAL so a verdict is never silently partial
997
+ return report
998
+
999
+
1000
+ async def run_compliance_document_verdict(
1001
+ doc_name: str, data: bytes, *, store: Any, judge_model_id: str, embedder: Any, k: int = 8,
1002
+ sources: Optional[list[str]] = None, extract_model: Any = None, discoverer: Any = None,
1003
+ aextract_fn: Any = None, doc: Any = None,
1004
+ ) -> ComplianceReport:
1005
+ """Issue 0008 / SEG-7a: check an uploaded subject DOCUMENT (raw bytes: PDF/DOCX/HTML/TXT) for compliance --
1006
+ a thin shim over `run_subject_compliance_verdict` (the shared SEMANTIC front-end). Findings cite each verbatim
1007
+ assertion's "§ {section} ¶{n}" locator; a scanned subject's unreadable pages surface on the report (SEG-6).
1008
+ `sources` (0007) scopes to named policies; `discoverer`/`aextract_fn`/`doc` inject for tests."""
1009
+ return await run_subject_compliance_verdict(
1010
+ doc_name, store=store, judge_model_id=judge_model_id, embedder=embedder, k=k, sources=sources,
1011
+ name=doc_name, data=data, extract_model=extract_model, discoverer=discoverer, aextract_fn=aextract_fn,
1012
+ doc=doc)
1013
+
1014
+
1015
+ def register_compliance_check(registry) -> None:
1016
+ """Register `compliance_check` (subgraph; CC-6). Contract = `ComplianceReport`."""
1017
+ registry.register(
1018
+ "compliance_check",
1019
+ contract=ComplianceReport,
1020
+ kind="subgraph",
1021
+ display_name="Compliance check (subject doc x requirements -> cited findings + gap matrix)",
1022
+ )
1023
+
1024
+
1025
+ async def ainvoke(resources, inputs: dict):
1026
+ """EP-REF-1c (ADR-0118): the capability invoke factory (impl_ref target) for the GENERIC compliance check --
1027
+ NOT the FTC-tuned `run_ad_compliance_check` (the product owns that variant + its guardrails). The store, the
1028
+ judge model (STRUCTURED_REASONING), and the embedder come from the workspace handle; the subject + scope from
1029
+ `inputs`. Two subject shapes: `{subject_text, source_doc}` (text) or `{doc_name, data}` (raw document bytes).
1030
+ `inputs` may also carry `k` (retrieval depth) and `sources` (scope to named policies)."""
1031
+ from rag_wright.models.profiles import ModelRole
1032
+
1033
+ judge = resources.model_id(ModelRole.STRUCTURED_REASONING)
1034
+ k = inputs.get("k", 8)
1035
+ sources = inputs.get("sources")
1036
+ if inputs.get("data") is not None: # a subject DOCUMENT (bytes)
1037
+ return await run_compliance_document_verdict(
1038
+ inputs["doc_name"], inputs["data"], store=resources._store, judge_model_id=judge,
1039
+ embedder=resources._embedder, k=k, sources=sources)
1040
+ return await run_generic_compliance_verdict( # a subject TEXT
1041
+ inputs["subject_text"], inputs["source_doc"], store=resources._store, judge_model_id=judge,
1042
+ embedder=resources._embedder, k=k, sources=sources)