rag-wright 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- rag_wright/__init__.py +13 -0
- rag_wright/api/__init__.py +33 -0
- rag_wright/api/config.py +59 -0
- rag_wright/api/discover.py +70 -0
- rag_wright/api/documents.py +39 -0
- rag_wright/api/ids.py +31 -0
- rag_wright/api/invoke.py +99 -0
- rag_wright/api/kg.py +61 -0
- rag_wright/api/mcp.py +94 -0
- rag_wright/api/usage.py +30 -0
- rag_wright/api/workspace.py +85 -0
- rag_wright/capabilities/__init__.py +8 -0
- rag_wright/capabilities/answer_generator.py +427 -0
- rag_wright/capabilities/ard.py +286 -0
- rag_wright/capabilities/assertion_extraction.py +79 -0
- rag_wright/capabilities/chunk_read.py +58 -0
- rag_wright/capabilities/chunk_write.py +163 -0
- rag_wright/capabilities/claim_extraction.py +153 -0
- rag_wright/capabilities/clause_exception_linking.py +117 -0
- rag_wright/capabilities/compliance_judgment.py +322 -0
- rag_wright/capabilities/compliance_store.py +87 -0
- rag_wright/capabilities/contract_kg_serve.py +156 -0
- rag_wright/capabilities/contract_kg_store.py +251 -0
- rag_wright/capabilities/dg_extraction.py +585 -0
- rag_wright/capabilities/disambiguation.py +163 -0
- rag_wright/capabilities/document_parse.py +87 -0
- rag_wright/capabilities/document_scope.py +49 -0
- rag_wright/capabilities/embedding.py +164 -0
- rag_wright/capabilities/embedding_profiles.py +43 -0
- rag_wright/capabilities/entity_resolution.py +154 -0
- rag_wright/capabilities/fusion.py +64 -0
- rag_wright/capabilities/graph_extraction.py +243 -0
- rag_wright/capabilities/graph_query.py +73 -0
- rag_wright/capabilities/graph_storage.py +111 -0
- rag_wright/capabilities/highlight_serve.py +142 -0
- rag_wright/capabilities/hybrid_search.py +65 -0
- rag_wright/capabilities/invoke.py +31 -0
- rag_wright/capabilities/jev_decision.py +38 -0
- rag_wright/capabilities/manifests.py +872 -0
- rag_wright/capabilities/okf_navigate.py +456 -0
- rag_wright/capabilities/parsing.py +286 -0
- rag_wright/capabilities/property_boosted_retrieval.py +125 -0
- rag_wright/capabilities/query_function_classifier.py +94 -0
- rag_wright/capabilities/query_understanding.py +109 -0
- rag_wright/capabilities/registry.py +262 -0
- rag_wright/capabilities/remote_encoders.py +94 -0
- rag_wright/capabilities/requirement_extraction.py +247 -0
- rag_wright/capabilities/reranking.py +123 -0
- rag_wright/capabilities/retrieval_core.py +126 -0
- rag_wright/capabilities/rlm_chunking.py +808 -0
- rag_wright/capabilities/rlm_synthesis.py +316 -0
- rag_wright/capabilities/scan_quality.py +136 -0
- rag_wright/capabilities/span_relevance_judgment.py +191 -0
- rag_wright/capabilities/vision_to_text.py +85 -0
- rag_wright/capabilities/vlm_ocr.py +85 -0
- rag_wright/contracts/__init__.py +6 -0
- rag_wright/contracts/chunk.py +79 -0
- rag_wright/contracts/compliance.py +303 -0
- rag_wright/contracts/contract_meta.py +27 -0
- rag_wright/contracts/extraction.py +130 -0
- rag_wright/contracts/function.py +167 -0
- rag_wright/contracts/function_routing.py +91 -0
- rag_wright/contracts/highlight.py +74 -0
- rag_wright/contracts/identifiers.py +153 -0
- rag_wright/contracts/jurisdiction.py +96 -0
- rag_wright/contracts/ontology.py +142 -0
- rag_wright/contracts/property.py +201 -0
- rag_wright/contracts/provenance.py +78 -0
- rag_wright/contracts/query_intent.py +53 -0
- rag_wright/contracts/span.py +76 -0
- rag_wright/contracts/value_match.py +84 -0
- rag_wright/corpus/__init__.py +0 -0
- rag_wright/corpus/canonicalize.py +116 -0
- rag_wright/corpus/cuad.py +153 -0
- rag_wright/corpus/cuad_ingestion.py +72 -0
- rag_wright/corpus/document_parser.py +299 -0
- rag_wright/corpus/edgar.py +231 -0
- rag_wright/corpus/gcs_ingestion.py +120 -0
- rag_wright/corpus/http.py +110 -0
- rag_wright/corpus/selection.py +152 -0
- rag_wright/mcp/__init__.py +11 -0
- rag_wright/mcp/compliance_server.py +299 -0
- rag_wright/mcp/intra_document_qa_server.py +170 -0
- rag_wright/mcp/relational_qa_server.py +171 -0
- rag_wright/mcp/session_store.py +64 -0
- rag_wright/mcp/typed_property_retrieval_server.py +191 -0
- rag_wright/models/__init__.py +8 -0
- rag_wright/models/profiles.py +331 -0
- rag_wright/models/seam.py +497 -0
- rag_wright/models/tag_structured.py +285 -0
- rag_wright/models/tracing.py +179 -0
- rag_wright/models/usage.py +102 -0
- rag_wright/okf/__init__.py +11 -0
- rag_wright/okf/compile.py +292 -0
- rag_wright/okf/document.py +47 -0
- rag_wright/okf/enrich.py +176 -0
- rag_wright/okf/links.py +190 -0
- rag_wright/okf/lint.py +105 -0
- rag_wright/ontology/__init__.py +6 -0
- rag_wright/ontology/_generated_template_meta.py +60 -0
- rag_wright/ontology/_generated_vocab.py +52 -0
- rag_wright/ontology/clause_template.py +964 -0
- rag_wright/ontology/codegen.py +84 -0
- rag_wright/ontology/compliance_bridge.ttl +186 -0
- rag_wright/ontology/contract_bridge.ttl +2685 -0
- rag_wright/ontology/contract_taxonomy.py +24 -0
- rag_wright/ontology/derive.py +58 -0
- rag_wright/ontology/loader.py +435 -0
- rag_wright/ontology/packs/ftc_16cfr255.ttl +29 -0
- rag_wright/ontology/registry.py +87 -0
- rag_wright/ontology/template_introspect.py +100 -0
- rag_wright/py.typed +0 -0
- rag_wright/reference/__init__.py +2 -0
- rag_wright/reference/compliance.py +41 -0
- rag_wright/reference/contract_seam.py +123 -0
- rag_wright/skills/__init__.py +7 -0
- rag_wright/skills/claim_extraction/SKILL.md +47 -0
- rag_wright/skills/claim_extraction/__init__.py +1 -0
- rag_wright/skills/claim_extraction/template.py +50 -0
- rag_wright/skills/compliance_judgment/SKILL.md +59 -0
- rag_wright/skills/corpus_ingest/SKILL.md +106 -0
- rag_wright/skills/extraction_semantic_judge/SKILL.md +51 -0
- rag_wright/skills/extraction_semantic_judge/__init__.py +1 -0
- rag_wright/skills/generation/SKILL.md +64 -0
- rag_wright/skills/generation/__init__.py +1 -0
- rag_wright/skills/generic_compliance_judgment/SKILL.md +58 -0
- rag_wright/skills/okf_navigate/SKILL.md +137 -0
- rag_wright/skills/requirement_extraction/SKILL.md +47 -0
- rag_wright/skills/requirement_extraction/__init__.py +1 -0
- rag_wright/skills/requirement_extraction/template.py +50 -0
- rag_wright/skills/rlm/SKILL.md +186 -0
- rag_wright/skills/rlm/__init__.py +31 -0
- rag_wright/skills/rlm/agent.py +292 -0
- rag_wright/skills/span_relevance_judgment/SKILL.md +67 -0
- rag_wright/skills/vision_to_text/SKILL.md +36 -0
- rag_wright/skills/vision_to_text/__init__.py +1 -0
- rag_wright/spans/__init__.py +1 -0
- rag_wright/spans/boundary.py +78 -0
- rag_wright/spans/clause_function_classifier.py +490 -0
- rag_wright/spans/clause_kg_extractor.py +337 -0
- rag_wright/spans/cuad_labels.py +81 -0
- rag_wright/spans/dim_classifier.py +158 -0
- rag_wright/spans/dim_fleet.json +411 -0
- rag_wright/spans/function_classifier.py +77 -0
- rag_wright/spans/function_families.py +62 -0
- rag_wright/spans/hybrid_classifier.py +103 -0
- rag_wright/spans/legalbert_classifier.py +83 -0
- rag_wright/spans/model_capabilities.py +107 -0
- rag_wright/spans/new_function_labels.py +111 -0
- rag_wright/spans/page_map.py +68 -0
- rag_wright/spans/property_extractor.py +365 -0
- rag_wright/spans/property_grounding.py +182 -0
- rag_wright/spans/reclassify.py +77 -0
- rag_wright/spans/scarce_function_labels.py +105 -0
- rag_wright/spans/segment.py +341 -0
- rag_wright/spans/semantic_judge.py +197 -0
- rag_wright/spans/symbolic_validation.py +131 -0
- rag_wright/spans/tag_clause_extractor.py +182 -0
- rag_wright/store/__init__.py +6 -0
- rag_wright/store/arcadedb.py +1135 -0
- rag_wright/store/chunk_text.py +66 -0
- rag_wright/store/seam.py +213 -0
- rag_wright/subgraphs/__init__.py +0 -0
- rag_wright/subgraphs/async_ingestion.py +204 -0
- rag_wright/subgraphs/compliance_check.py +1042 -0
- rag_wright/subgraphs/compliance_ingestion.py +306 -0
- rag_wright/subgraphs/contract_ingestion_pipeline.py +999 -0
- rag_wright/subgraphs/graph_extraction.py +102 -0
- rag_wright/subgraphs/intra_document_qa.py +328 -0
- rag_wright/subgraphs/observability.py +140 -0
- rag_wright/subgraphs/query_constraint_extraction.py +73 -0
- rag_wright/subgraphs/relational_qa.py +165 -0
- rag_wright/subgraphs/requirement_extraction.py +137 -0
- rag_wright/subgraphs/scaffold.py +65 -0
- rag_wright/subgraphs/semantic_chunking.py +183 -0
- rag_wright/subgraphs/typed_clause_extraction.py +172 -0
- rag_wright/subgraphs/typed_property_retrieval.py +278 -0
- rag_wright/util/__init__.py +1 -0
- rag_wright/util/concurrent.py +153 -0
- rag_wright/util/spacy_model.py +45 -0
- rag_wright-0.1.0.dist-info/METADATA +168 -0
- rag_wright-0.1.0.dist-info/RECORD +184 -0
- rag_wright-0.1.0.dist-info/WHEEL +4 -0
- rag_wright-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,1042 @@
|
|
|
1
|
+
"""CC-6 (compliance §13.3): the `compliance_check` subgraph -- the headline composite of the compliance module.
|
|
2
|
+
|
|
3
|
+
A hardened, query-side LangGraph subgraph on `scaffold.py`: for a subject ad, extract its claims (CC-3),
|
|
4
|
+
retrieve the APPLICABLE requirements (claim scope <-> requirement applicability), judge each `(claim,
|
|
5
|
+
requirement)` pair (CC-4, concurrent), and assemble cited findings + a per-requirement gap matrix + a verdict
|
|
6
|
+
summary. Two design requirements from earlier findings are built in here:
|
|
7
|
+
|
|
8
|
+
- **Section->claim_type applicability map** (the ontology enrichment): a requirement whose extracted
|
|
9
|
+
`applicability_scope` is empty is applied by its FTC section (255.5 material-connections -> endorsement, 255.1
|
|
10
|
+
general -> all, 255.0 definitions -> none). Authored in code referencing `ClaimType` (typos are test failures,
|
|
11
|
+
no drifting .ttl) -- the JUDGE-ONTOLOGY-1 pattern. See [[ontology-lever-vs-extraction-lever]].
|
|
12
|
+
- **Ad-level disclosure aggregation** (the CC-4 residual fix): disclosures are ad-level but claims are per-span,
|
|
13
|
+
so before judging, each claim's disclosures are enriched with the union across the whole ad -- a per-span
|
|
14
|
+
fragment with disc=none is not over-flagged when the ad as a whole discloses.
|
|
15
|
+
|
|
16
|
+
Query-side posture: every node degrades to empty on failure (never crash); the judge's conservative default
|
|
17
|
+
(needs_review) plus per-finding human-gating carry the trust guarantees.
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
from __future__ import annotations
|
|
21
|
+
|
|
22
|
+
import asyncio
|
|
23
|
+
from typing import Any, Awaitable, Callable, Optional, TypedDict
|
|
24
|
+
|
|
25
|
+
from langgraph.graph import END, START, StateGraph
|
|
26
|
+
|
|
27
|
+
from rag_wright.capabilities.compliance_judgment import AJudgeFn, ajudge_pairs
|
|
28
|
+
from rag_wright.capabilities.retrieval_core import _cosine
|
|
29
|
+
from enum import Enum
|
|
30
|
+
|
|
31
|
+
from rag_wright.contracts.compliance import (
|
|
32
|
+
CheckableFact,
|
|
33
|
+
Claim,
|
|
34
|
+
ClaimType,
|
|
35
|
+
ComplianceFinding,
|
|
36
|
+
ComplianceReport,
|
|
37
|
+
Constraint,
|
|
38
|
+
DeonticType,
|
|
39
|
+
Requirement,
|
|
40
|
+
RuleScope,
|
|
41
|
+
Verdict,
|
|
42
|
+
)
|
|
43
|
+
from rag_wright.contracts.provenance import ConfidenceTag
|
|
44
|
+
from rag_wright.ontology.loader import ( # ADR-0066 P4: query-side knowledge from the ontology + FTC domain pack
|
|
45
|
+
load_actor_synonyms,
|
|
46
|
+
load_role_domains,
|
|
47
|
+
load_section_overrides,
|
|
48
|
+
)
|
|
49
|
+
from rag_wright.subgraphs.scaffold import DEFAULT_RETRY, business_span
|
|
50
|
+
|
|
51
|
+
_ALL_CLAIM_TYPES = {c.value for c in ClaimType}
|
|
52
|
+
|
|
53
|
+
# ADR-0066 P4b: the per-section overrides (DEON-8 applicable claim types + DEON-1 rule scope) are AUTHORITATIVE in
|
|
54
|
+
# a DOMAIN PACK ttl (packs/ftc_16cfr255.ttl), not Python literals. To retarget a domain, ship its own pack; nothing
|
|
55
|
+
# FTC-specific is hardcoded here. (FTC finding CC-6: the endorsement guides apply by CONTEXT, not claim_type, so
|
|
56
|
+
# every operative section applies to ALL claim types and only definitions (255.0) is excluded -- now in the pack.)
|
|
57
|
+
_SECTION_RULE_SCOPE_RAW, SECTION_CLAIM_TYPES = load_section_overrides()
|
|
58
|
+
SECTION_RULE_SCOPE: dict[str, RuleScope] = {sec: RuleScope(v) for sec, v in _SECTION_RULE_SCOPE_RAW.items()}
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def _section_of(citation: str) -> str:
|
|
62
|
+
"""'§ 255.5' -> '255.5' (the section key for the applicability map)."""
|
|
63
|
+
return (citation or "").replace("§", "").strip()
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def applicable_claim_types(requirement: Requirement) -> set[str]:
|
|
67
|
+
"""DEON-8 (issue 0012): the claim types a requirement applies to. A CURATED override (`SECTION_CLAIM_TYPES`,
|
|
68
|
+
the FTC reference pack) WINS when the requirement's section is pinned there -- FTC behavior is byte-identical
|
|
69
|
+
(context sections apply to ALL claim types; §255.0 to none), and a noisy extracted claim_type can neither
|
|
70
|
+
narrow a context section nor rescue definitions. For ANY OTHER (customer) section the extracted `claim_type`
|
|
71
|
+
scope is LOAD-BEARING: it NARROWS (a pricing-scoped rule does not apply to a health claim); an empty scope is
|
|
72
|
+
recall-first (applies to all). So the KG's claim_type field routes for ANY policy, not just FTC -- the ad-path
|
|
73
|
+
analog of the DEON-6/7 actor gate (a curated override on top of a load-bearing typed field). The real per-
|
|
74
|
+
claim narrowing (most-relevant rule) is still semantic retrieval (Leg-B), the CC-7 refinement."""
|
|
75
|
+
section = _section_of(requirement.citation)
|
|
76
|
+
if section in SECTION_CLAIM_TYPES: # curated FTC override wins (context = all types, definitions = none)
|
|
77
|
+
return SECTION_CLAIM_TYPES[section]
|
|
78
|
+
scope = {c.value for c in requirement.applicability_scope if c.dimension == "claim_type"}
|
|
79
|
+
return scope or _ALL_CLAIM_TYPES # customer policy: extracted scope narrows; empty -> recall-first
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def _constraints_by_dimension(constraints: list) -> dict:
|
|
83
|
+
"""`[Constraint(dimension, value), ...]` -> {dimension: {values}}."""
|
|
84
|
+
out: dict = {}
|
|
85
|
+
for c in constraints:
|
|
86
|
+
out.setdefault(c.dimension, set()).add(c.value)
|
|
87
|
+
return out
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
_ROLE_GENERIC = frozenset({"", "party", "anyone", "any", "all", "everyone", "subject", "person", "other"})
|
|
91
|
+
|
|
92
|
+
# DEON-6/7: generic, DOMAIN-agnostic role-synonym normalization -- collapse common variants to one canonical role
|
|
93
|
+
# so the rule side and the subject side align (BGE cosine on bare role words does NOT encode role equivalence:
|
|
94
|
+
# employer~manufacturer 0.68 > advertiser~manufacturer 0.63, so a similarity threshold cannot separate them).
|
|
95
|
+
# An unknown role is KEPT as-is (both sides normalize identically, so an exotic domain still matches on its own
|
|
96
|
+
# term); this is role knowledge, NOT an FTC/corpus hardcode.
|
|
97
|
+
# ADR-0066 P4a: AUTHORITATIVE in compliance_bridge.ttl (cmp:ActorRole skos:altLabel) -- loaded, not a Python
|
|
98
|
+
# literal. To add a role synonym, edit the ttl (a new customer domain extends the role pack, not this code).
|
|
99
|
+
_ACTOR_SYNONYMS: dict[str, str] = load_actor_synonyms()
|
|
100
|
+
|
|
101
|
+
# ADR-0068 (engine issue 0013): the DISJOINTNESS knowledge for the recall-first actor gate -- `{canonical role ->
|
|
102
|
+
# domain}` from the ontology (cmp:roleDomain). AUTHORITATIVE in compliance_bridge.ttl; a customer domain adds its
|
|
103
|
+
# roles' domains in its own pack. Two roles are disjoint iff BOTH are here with DIFFERENT domains.
|
|
104
|
+
_ROLE_DOMAINS: dict[str, str] = load_role_domains()
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def canonical_actor(raw: str) -> str:
|
|
108
|
+
"""DEON-6/7: normalize an actor ROLE to its canonical form -- collapse a known synonym (manufacturer ->
|
|
109
|
+
advertiser), else keep the role as-is (lower-cased). Applied identically to the rule and subject side, so the
|
|
110
|
+
KG actor gate matches on aligned roles."""
|
|
111
|
+
a = (raw or "").strip().lower()
|
|
112
|
+
return _ACTOR_SYNONYMS.get(a, a)
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def _actor_set(scope: list) -> set:
|
|
116
|
+
"""The canonical `actor` roles in a scope (a claim's, or the document's aggregated)."""
|
|
117
|
+
return {canonical_actor(c.value) for c in (scope or []) if c.dimension == "actor" and c.value}
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def roles_disjoint(a: str, b: str) -> bool:
|
|
121
|
+
"""ADR-0068: are two CANONICAL actor roles ontology-DISJOINT? True ONLY when both carry a `cmp:roleDomain` and
|
|
122
|
+
the domains DIFFER (e.g. an advertising role vs a labor role). An unmodelled role (no domain), or two roles in
|
|
123
|
+
the same domain, are NOT disjoint -- the recall-first default is compatible."""
|
|
124
|
+
da, db = _ROLE_DOMAINS.get(a), _ROLE_DOMAINS.get(b)
|
|
125
|
+
return da is not None and db is not None and da != db
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def roles_compatible(a: str, b: str) -> bool:
|
|
129
|
+
"""ADR-0068: two CANONICAL actor roles are COMPATIBLE (a pair worth judging) unless the ontology makes them
|
|
130
|
+
disjoint -- recall-first. A generic/absent role on either side is always compatible. Replaces exact role
|
|
131
|
+
equality: two different-but-overlapping roles (advertiser vs seller) now match, so a real violation is never
|
|
132
|
+
silently dropped because two independent extractions chose different words for the same party."""
|
|
133
|
+
return not a or not b or a in _ROLE_GENERIC or b in _ROLE_GENERIC or not roles_disjoint(a, b)
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def actor_matches(rule_actor: str, subject_actors: set) -> bool:
|
|
137
|
+
"""DEON-6/7 + ADR-0068: is the requirement's actor ROLE COMPATIBLE with the subject's (already-canonical)
|
|
138
|
+
actors? RECALL-FIRST on two axes: (1) a generic/absent rule actor or a subject with no actor info never gates
|
|
139
|
+
(True); (2) a specific rule actor matches unless it is ontology-DISJOINT from EVERY subject actor -- so an
|
|
140
|
+
unmodelled or merely-different-but-overlapping role is judged, not dropped (issue 0013)."""
|
|
141
|
+
ra = canonical_actor(rule_actor)
|
|
142
|
+
if not ra or ra in _ROLE_GENERIC or not subject_actors:
|
|
143
|
+
return True # recall-first: nothing to gate on
|
|
144
|
+
return any(roles_compatible(ra, sa) for sa in subject_actors)
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def _actor_compatible(a: str, b: str) -> bool:
|
|
148
|
+
"""DEON-9 + ADR-0068: the symbolic gate for which permissions can defend which O/F rule -- two CANONICAL actor
|
|
149
|
+
roles are compatible unless ontology-disjoint (recall-first), same predicate as the actor gate."""
|
|
150
|
+
return roles_compatible(a, b)
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def constraint_applies(requirement_scope: list, subject_scope: list) -> bool:
|
|
154
|
+
"""COMP-APPLIC-1 Increment 0: the DIMENSION-AGNOSTIC applicability matcher. A requirement applies to a subject
|
|
155
|
+
iff, for EVERY dimension the requirement constrains, the subject's value(s) on that dimension INTERSECT the
|
|
156
|
+
requirement's allowed values. A dimension the requirement does NOT constrain, or one the subject does NOT
|
|
157
|
+
carry, never excludes (recall-first). Both scopes are `(dimension, value)` Constraint lists, so ANY domain
|
|
158
|
+
routes with NO new compliance_check code -- the ontology's dimensions are DATA, not per-domain matcher logic.
|
|
159
|
+
(The advertising `applies_to` keeps its own path: its "definitions section applies to nothing" is authored
|
|
160
|
+
doctrine that pure constraint matching does not express -- a domain with such authored routing adds a thin
|
|
161
|
+
wrapper; a domain with pure scope matching adds none.)"""
|
|
162
|
+
req = _constraints_by_dimension(requirement_scope)
|
|
163
|
+
subj = _constraints_by_dimension(subject_scope)
|
|
164
|
+
for dim, allowed in req.items():
|
|
165
|
+
vals = subj.get(dim)
|
|
166
|
+
if vals is not None and not (vals & allowed):
|
|
167
|
+
return False # subject HAS this dimension but with a non-matching value -> excluded
|
|
168
|
+
return True
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
def applies_to(requirement: Requirement, claim: Claim) -> bool:
|
|
172
|
+
"""Does `requirement` apply to `claim`? (the claim's type is in the requirement's applicable claim types)."""
|
|
173
|
+
return claim.claim_type.value in applicable_claim_types(requirement)
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
class DeonticRoute(str, Enum):
|
|
177
|
+
"""DEON-1 (issue 0012): the JUDGE route a rule takes, derived from its DEONTIC TYPE (a KG-typed field), not
|
|
178
|
+
its FTC section number -- so it works for ANY customer policy."""
|
|
179
|
+
|
|
180
|
+
OBLIGATION = "obligation" # breach = ABSENCE -> document-scoped, judged ONCE (a per-sentence judge cannot
|
|
181
|
+
# answer "is it present anywhere?"); always-included (CONTEXT).
|
|
182
|
+
PROHIBITION = "prohibition" # breach = PRESENCE -> per-assertion, where the subject asserts something related.
|
|
183
|
+
PERMISSION = "permission" # cannot be violated standalone -> EXCLUDED from violation-judging (an exception /
|
|
184
|
+
# defense that modifies an O/F rule; linked in DEON-9).
|
|
185
|
+
AMBIGUOUS = "ambiguous" # deontic force unreadable (off-vocab, coerced) -> recall-first: per-assertion + flag.
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
def deontic_route(requirement: Requirement) -> DeonticRoute:
|
|
189
|
+
"""DEON-1: the judge route for a rule, from its deontic type. Precedence: a curated `SECTION_RULE_SCOPE`
|
|
190
|
+
override (a hand-tuned domain pack may still pin scope by section) > an AMBIGUOUS deontic (off-vocab, coerced
|
|
191
|
+
to OBLIGATION but flagged -> recall-first, never trusted as an obligation) > the deontic type itself. FTC-
|
|
192
|
+
agnostic: a customer policy routes by what its rules ARE, not by matching FTC 16 CFR 255 section numbers."""
|
|
193
|
+
override = SECTION_RULE_SCOPE.get(_section_of(requirement.citation))
|
|
194
|
+
if override is RuleScope.CONTEXT:
|
|
195
|
+
return DeonticRoute.OBLIGATION
|
|
196
|
+
if override is RuleScope.CONTENT:
|
|
197
|
+
return DeonticRoute.PROHIBITION
|
|
198
|
+
if requirement.confidence is ConfidenceTag.AMBIGUOUS: # the extractor could not read the deontic force
|
|
199
|
+
return DeonticRoute.AMBIGUOUS
|
|
200
|
+
if requirement.deontic_type is DeonticType.OBLIGATION:
|
|
201
|
+
return DeonticRoute.OBLIGATION
|
|
202
|
+
if requirement.deontic_type is DeonticType.PERMISSION:
|
|
203
|
+
return DeonticRoute.PERMISSION
|
|
204
|
+
return DeonticRoute.PROHIBITION
|
|
205
|
+
|
|
206
|
+
|
|
207
|
+
def rule_scope_of(requirement: Requirement) -> RuleScope:
|
|
208
|
+
"""DEON-1: CONTEXT (always-include) for an obligation, else CONTENT (narrow by similarity) -- derived from
|
|
209
|
+
`deontic_route`, so it is DEONTIC-driven, not FTC-section-driven."""
|
|
210
|
+
return RuleScope.CONTEXT if deontic_route(requirement) is DeonticRoute.OBLIGATION else RuleScope.CONTENT
|
|
211
|
+
|
|
212
|
+
|
|
213
|
+
# DEON-2: how much of the subject the obligation judge sees -- the top-N most-relevant passages up to a char
|
|
214
|
+
# budget, NOT the whole document (an obligation is judged once over BOUNDED retrieved evidence, not per-sentence
|
|
215
|
+
# and not by dumping a contract-length document into one prompt).
|
|
216
|
+
OBLIGATION_TOP_N = 5
|
|
217
|
+
OBLIGATION_CHAR_BUDGET = 4000
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
def _within_budget(claims: list, *, top_n: int, char_budget: int) -> list:
|
|
221
|
+
"""Take up to `top_n` claims (already ranked) but stop once the cumulative assertion text exceeds
|
|
222
|
+
`char_budget` -- the bounded evidence window for one obligation judgment. Always keeps at least the first."""
|
|
223
|
+
out: list = []
|
|
224
|
+
used = 0
|
|
225
|
+
for c in claims[:top_n]:
|
|
226
|
+
text = c.assertion_text or ""
|
|
227
|
+
if out and used + len(text) > char_budget:
|
|
228
|
+
break
|
|
229
|
+
out.append(c)
|
|
230
|
+
used += len(text)
|
|
231
|
+
return out
|
|
232
|
+
|
|
233
|
+
|
|
234
|
+
def subject_scope(facts: list) -> list:
|
|
235
|
+
"""DEON-5: the document-level SubjectScope -- the deduped union of every fact's per-assertion `scope`
|
|
236
|
+
`Constraint`s (actor + any inferred dimension). Feeds the obligation actor-gate (DEON-7: is the rule's actor
|
|
237
|
+
present in the document at all?) and, per-assertion, the prohibition constraint router (DEON-6). Returns a
|
|
238
|
+
`list[Constraint]`; the actor-set is the values on dimension 'actor'."""
|
|
239
|
+
seen: set = set()
|
|
240
|
+
out: list = []
|
|
241
|
+
for f in facts:
|
|
242
|
+
for c in getattr(f, "scope", None) or []:
|
|
243
|
+
key = (c.dimension, c.value)
|
|
244
|
+
if key not in seen:
|
|
245
|
+
seen.add(key)
|
|
246
|
+
out.append(c)
|
|
247
|
+
return out
|
|
248
|
+
|
|
249
|
+
|
|
250
|
+
def _document_signal_line(claims: list) -> str:
|
|
251
|
+
"""DEON-8 (Option 1): the ad-level structured SIGNALS rendered as document content for the obligation judge --
|
|
252
|
+
the disclosure union + whether evidence is referenced, aggregated across ALL claims (a disclosure made
|
|
253
|
+
ANYWHERE in the ad satisfies a disclosure obligation, so the whole-document union matters, not just the top-N
|
|
254
|
+
evidence window). getattr-tolerant so a generic (non-ad) fact contributes nothing -> '' (domain-neutral: the
|
|
255
|
+
engine's obligation retriever stays free of ad concepts, the signals only appear when the facts carry them)."""
|
|
256
|
+
disclosures = sorted({d for c in claims for d in (getattr(c, "disclosures_present", None) or [])})
|
|
257
|
+
evidence = any(getattr(c, "evidence_referenced", False) for c in claims)
|
|
258
|
+
parts: list[str] = []
|
|
259
|
+
if disclosures:
|
|
260
|
+
parts.append("disclosures present in the document: " + "; ".join(disclosures))
|
|
261
|
+
if evidence:
|
|
262
|
+
parts.append("the document references supporting evidence")
|
|
263
|
+
return ("\n\n[DOCUMENT SIGNALS] " + "; ".join(parts) + ".") if parts else ""
|
|
264
|
+
|
|
265
|
+
|
|
266
|
+
def build_obligation_pairs_fn(embedder: Any, *, top_n: int = OBLIGATION_TOP_N,
|
|
267
|
+
char_budget: int = OBLIGATION_CHAR_BUDGET) -> Any:
|
|
268
|
+
"""DEON-2: the obligation evidence retriever. For each obligation, embed it and RANK the subject assertions,
|
|
269
|
+
take the top-N most-relevant up to a char budget, and build one bounded evidence `CheckableFact` -> one
|
|
270
|
+
`(evidence_fact, obligation)` pair. Symbolic/vector narrowing (KG deontic route + embeddings) selects the
|
|
271
|
+
small evidence set; the LLM then judges once over it (breach = absence). Claims are embedded ONCE.
|
|
272
|
+
|
|
273
|
+
DEON-8 (Option 1): the ad-level structured SIGNALS (the disclosure union / evidence-referenced) are appended
|
|
274
|
+
to each bundle as document content, so an obligation judged ONCE on the ad path still sees a disclosure made
|
|
275
|
+
anywhere in the ad. getattr-tolerant, so the generic path is unaffected."""
|
|
276
|
+
def obligation_pairs(obligations: list, claims: list, source_doc: str) -> list:
|
|
277
|
+
if not (obligations and claims):
|
|
278
|
+
return []
|
|
279
|
+
# DEON-7: the ACTOR GATE (symbolic, zero LLM) -- an obligation whose bound actor is NOT present in the
|
|
280
|
+
# document's actor-set is out of scope, so it is SKIPPED entirely (no retrieval, no judge call). Robust to
|
|
281
|
+
# role synonyms via `actor_matches`; recall-first (a generic/absent actor never gates).
|
|
282
|
+
doc_actors = _actor_set(subject_scope(claims))
|
|
283
|
+
obligations = [ob for ob in obligations if actor_matches(ob.actor, doc_actors)]
|
|
284
|
+
if not obligations:
|
|
285
|
+
return []
|
|
286
|
+
signal_line = _document_signal_line(claims) # DEON-8: carry the ad-level disclosure/evidence signals
|
|
287
|
+
claim_vecs = [(c, embedder.encode_dense(c.assertion_text)) for c in claims] # embed the subject once
|
|
288
|
+
pairs: list = []
|
|
289
|
+
for ob in obligations:
|
|
290
|
+
ob_vec = embedder.encode_dense(ob.requirement_text)
|
|
291
|
+
ranked = [c for c, _ in sorted(claim_vecs, key=lambda cv: _cosine(ob_vec, cv[1]), reverse=True)]
|
|
292
|
+
evidence = _within_budget(ranked, top_n=top_n, char_budget=char_budget)
|
|
293
|
+
# issue 0044: `assertion_text` is document text ONLY (so the citation stays a quote from the user's
|
|
294
|
+
# doc); the DEON-8 signals ride in `document_signals`, seen by the judge but never cited. The bundle is
|
|
295
|
+
# the top-N spans joined -> ASSEMBLED evidence, flagged so a consumer never renders it as one verbatim.
|
|
296
|
+
text = "\n\n".join(c.assertion_text for c in evidence) or "(empty subject)"
|
|
297
|
+
fact = CheckableFact(fact_id=CheckableFact.make_id(source_doc, ob.requirement_id, text),
|
|
298
|
+
source_doc=source_doc, assertion_text=text,
|
|
299
|
+
document_signals=signal_line, citation_kind="assembled")
|
|
300
|
+
pairs.append((fact, ob))
|
|
301
|
+
return pairs
|
|
302
|
+
|
|
303
|
+
return obligation_pairs
|
|
304
|
+
|
|
305
|
+
|
|
306
|
+
DEFENSE_TOP_N = 3
|
|
307
|
+
|
|
308
|
+
|
|
309
|
+
def build_defense_linker(embedder: Any, requirements: list, *, top_n: int = DEFENSE_TOP_N) -> Any:
|
|
310
|
+
"""DEON-9 (issue 0012): the PERMISSION-AS-DEFENSE linker (ADR-0044 pattern, requirement side). For an
|
|
311
|
+
obligation/prohibition rule, return the same-`source` PERMISSIONS that may EXCUSE it (a carve-out/safe-harbor),
|
|
312
|
+
so the judge can rule a legitimate exception COMPLIANT instead of a false violation. Symbolic candidacy: same
|
|
313
|
+
policy source + actor-compatible (canonical, recall-first); ranked by semantic proximity and capped at `top_n`
|
|
314
|
+
-- rank+cap, NOT a fragile similarity threshold. Zero extra LLM: the ONE judge call now reasons over the rule
|
|
315
|
+
plus its linked defenses. Vectors are precomputed ONCE (like `build_select_fn`)."""
|
|
316
|
+
vectors = {r.requirement_id: embedder.encode_dense(r.requirement_text) for r in requirements}
|
|
317
|
+
perms_by_source: dict[str, list] = {}
|
|
318
|
+
for r in requirements:
|
|
319
|
+
if deontic_route(r) is DeonticRoute.PERMISSION:
|
|
320
|
+
perms_by_source.setdefault(r.source, []).append(r)
|
|
321
|
+
|
|
322
|
+
def defenses_for(rule: Any) -> list:
|
|
323
|
+
if deontic_route(rule) is DeonticRoute.PERMISSION:
|
|
324
|
+
return [] # a permission is not judged for violation, so it carries no defenses of its own
|
|
325
|
+
perms = perms_by_source.get(rule.source, [])
|
|
326
|
+
if not perms:
|
|
327
|
+
return []
|
|
328
|
+
ra = canonical_actor(rule.actor)
|
|
329
|
+
cands = [p for p in perms if _actor_compatible(ra, canonical_actor(p.actor))]
|
|
330
|
+
rule_vec = vectors.get(rule.requirement_id, [])
|
|
331
|
+
ranked = sorted(cands, key=lambda p: _cosine(rule_vec, vectors.get(p.requirement_id, [])), reverse=True)
|
|
332
|
+
return ranked[:top_n]
|
|
333
|
+
|
|
334
|
+
return defenses_for
|
|
335
|
+
|
|
336
|
+
|
|
337
|
+
def _with_defenses(requirement: Any, defense_linker: Any) -> Any:
|
|
338
|
+
"""DEON-9: attach the linked permissions (rendered `citation: text`) to a rule as query-time `defenses`, so
|
|
339
|
+
the judge renders them as structured exception context. A no-op (returns the rule unchanged) when nothing is
|
|
340
|
+
linked, so an unrelated rule is untouched."""
|
|
341
|
+
if defense_linker is None:
|
|
342
|
+
return requirement
|
|
343
|
+
linked = defense_linker(requirement)
|
|
344
|
+
if not linked:
|
|
345
|
+
return requirement
|
|
346
|
+
return requirement.model_copy(
|
|
347
|
+
update={"defenses": [f"{p.citation}: {p.requirement_text}" for p in linked]})
|
|
348
|
+
|
|
349
|
+
|
|
350
|
+
SelectFn = Callable[[Claim, list], list] # (claim, requirements) -> the narrowed requirements to judge
|
|
351
|
+
|
|
352
|
+
|
|
353
|
+
def _dedup(requirements: list, vectors: dict, threshold: float) -> list:
|
|
354
|
+
"""Greedy near-duplicate collapse: keep a requirement unless it is >= `threshold` cosine-similar to one
|
|
355
|
+
already kept (the 3 near-identical §255.5 disclosure rules -> one). Order-preserving."""
|
|
356
|
+
kept: list = []
|
|
357
|
+
for req in requirements:
|
|
358
|
+
vec = vectors.get(req.requirement_id)
|
|
359
|
+
if vec is not None and any(_cosine(vec, vectors[k.requirement_id]) >= threshold for k in kept
|
|
360
|
+
if vectors.get(k.requirement_id) is not None):
|
|
361
|
+
continue
|
|
362
|
+
kept.append(req)
|
|
363
|
+
return kept
|
|
364
|
+
|
|
365
|
+
|
|
366
|
+
def build_select_fn(
|
|
367
|
+
embedder, requirements: list, *, k: int = 5, context_k: int = 3, dedup_threshold: float = 0.92,
|
|
368
|
+
filter_applicability: bool = True, constraint_scope_fn: Any = None
|
|
369
|
+
) -> SelectFn:
|
|
370
|
+
"""CC-8b: build the semantic-narrowing selector. Precomputes each requirement's BGE vector ONCE. Per claim it
|
|
371
|
+
returns the top-`context_k` CONTEXT rules (disclosure -- kept regardless of content so similarity can't miss
|
|
372
|
+
them, Example B) + the top-`k` CONTENT rules (substantiation etc., ranked by cosine to the claim), deduped.
|
|
373
|
+
Both classes are CAPPED so an over-extracted section (§255.5 -> 32 near-identical disclosure rules) collapses
|
|
374
|
+
to a few representatives rather than re-exploding the cross-product. A precision/cost win that keeps the
|
|
375
|
+
context rules (the recall guarantee) while cutting the redundant-rule noise.
|
|
376
|
+
|
|
377
|
+
Three applicability-routing modes (in precedence): `constraint_scope_fn` (COMP-APPLIC-1 Increment 0: generic
|
|
378
|
+
DIMENSION-AGNOSTIC structured routing -- `subject -> [Constraint]`, matched against each requirement's scope by
|
|
379
|
+
`constraint_applies`; ANY domain, no per-domain matcher code) > `filter_applicability=True` (advertising
|
|
380
|
+
claim_type routing via `applies_to`) > `filter_applicability=False` (semantic-only, COMP-VERDICT-GENERIC)."""
|
|
381
|
+
vectors = {r.requirement_id: embedder.encode_dense(r.requirement_text) for r in requirements}
|
|
382
|
+
|
|
383
|
+
def _ranked(reqs: list, claim_vec: list) -> list:
|
|
384
|
+
return sorted(reqs, key=lambda r: _cosine(claim_vec, vectors.get(r.requirement_id, [])), reverse=True)
|
|
385
|
+
|
|
386
|
+
def select(claim: Any, reqs: list) -> list:
|
|
387
|
+
if constraint_scope_fn is not None: # generic structured routing (any domain, ontology-driven, DATA)
|
|
388
|
+
subject_scope = constraint_scope_fn(claim)
|
|
389
|
+
actors = _actor_set(subject_scope) # DEON-6: prohibition gated by (non-actor constraints) AND actor role
|
|
390
|
+
applicable = [r for r in reqs if constraint_applies(r.applicability_scope, subject_scope)
|
|
391
|
+
and actor_matches(r.actor, actors)]
|
|
392
|
+
elif filter_applicability: # advertising claim_type routing
|
|
393
|
+
applicable = [r for r in reqs if applies_to(r, claim)]
|
|
394
|
+
else: # semantic-only (generic verdict)
|
|
395
|
+
applicable = list(reqs)
|
|
396
|
+
claim_vec = embedder.encode_dense(claim.assertion_text)
|
|
397
|
+
context = _dedup(_ranked([r for r in applicable if rule_scope_of(r) is RuleScope.CONTEXT], claim_vec),
|
|
398
|
+
vectors, dedup_threshold)[:context_k]
|
|
399
|
+
content = _dedup(_ranked([r for r in applicable if rule_scope_of(r) is RuleScope.CONTENT], claim_vec),
|
|
400
|
+
vectors, dedup_threshold)[:k]
|
|
401
|
+
return _dedup(context + content, vectors, dedup_threshold)
|
|
402
|
+
|
|
403
|
+
return select
|
|
404
|
+
|
|
405
|
+
|
|
406
|
+
def _actor_gated_pairs(claims: list, prohibitions: list, obligations: list, constraint_scope_fn: Any) -> list[dict]:
|
|
407
|
+
"""ADR-0068 (issue 0013): the (assertion|document, rule) pairs the symbolic ACTOR gate SKIPPED before any judge
|
|
408
|
+
call -- recomputed from the SAME module gate (`actor_matches`) the router applies, so the report can state
|
|
409
|
+
honest coverage and a gated pair is never silent. OBLIGATIONS: a rule whose bound actor is ontology-disjoint
|
|
410
|
+
from every document actor (DEON-7, document scope). PROHIBITIONS: per assertion, a rule that PASSES the
|
|
411
|
+
constraint router but is actor-disjoint from the assertion's actors (DEON-6) -- computed only when constraint
|
|
412
|
+
routing is active (the ad path narrows prohibitions by claim_type, not the actor gate, so it reports none).
|
|
413
|
+
Empty unless a disjoint role actually blocked a pair (the recall-first norm)."""
|
|
414
|
+
gated: list[dict] = []
|
|
415
|
+
doc_actors = _actor_set(subject_scope(claims)) if claims else set()
|
|
416
|
+
for ob in obligations:
|
|
417
|
+
if not actor_matches(ob.actor, doc_actors):
|
|
418
|
+
gated.append({"requirement_id": ob.requirement_id, "citation": ob.citation,
|
|
419
|
+
"actor": canonical_actor(ob.actor), "subject_actors": sorted(doc_actors),
|
|
420
|
+
"scope": "document", "claim_id": None})
|
|
421
|
+
if constraint_scope_fn is not None:
|
|
422
|
+
for claim in claims:
|
|
423
|
+
subj = constraint_scope_fn(claim)
|
|
424
|
+
actors = _actor_set(subj)
|
|
425
|
+
for p in prohibitions:
|
|
426
|
+
if constraint_applies(p.applicability_scope, subj) and not actor_matches(p.actor, actors):
|
|
427
|
+
gated.append({"requirement_id": p.requirement_id, "citation": p.citation,
|
|
428
|
+
"actor": canonical_actor(p.actor), "subject_actors": sorted(actors),
|
|
429
|
+
"scope": "assertion", "claim_id": getattr(claim, "fact_id", None)})
|
|
430
|
+
return gated
|
|
431
|
+
|
|
432
|
+
|
|
433
|
+
# ASYNC-C1 (ADR-0057): claims_fn is async (its extraction model call gets a true wall-clock deadline).
|
|
434
|
+
ClaimsFn = Callable[[str, str], Awaitable[list]] # (subject_text, source_doc) -> list[Claim]
|
|
435
|
+
RequirementsFn = Callable[[], list] # () -> list[Requirement]
|
|
436
|
+
|
|
437
|
+
|
|
438
|
+
class CheckState(TypedDict, total=False):
|
|
439
|
+
subject_text: str
|
|
440
|
+
source_doc: str
|
|
441
|
+
claims: list
|
|
442
|
+
ad_disclosures: list
|
|
443
|
+
pairs: list
|
|
444
|
+
gated: list # ADR-0068 (issue 0013): (assertion|document, rule) pairs the actor gate skipped -> the report
|
|
445
|
+
findings: list
|
|
446
|
+
report: ComplianceReport
|
|
447
|
+
|
|
448
|
+
|
|
449
|
+
def _enrich(claim: Claim, ad_disclosures: set[str]) -> Claim:
|
|
450
|
+
"""Enrich a claim's disclosures with the ad-level union (the CC-4 fix). claim_id is content-hashed on the
|
|
451
|
+
assertion, not the disclosures, so it is unchanged -- the finding still cites the original claim."""
|
|
452
|
+
if not ad_disclosures:
|
|
453
|
+
return claim
|
|
454
|
+
merged = sorted(set(claim.disclosures_present) | ad_disclosures)
|
|
455
|
+
return claim.model_copy(update={"disclosures_present": merged})
|
|
456
|
+
|
|
457
|
+
|
|
458
|
+
def build_compliance_check(
|
|
459
|
+
*, claims_fn: ClaimsFn, requirements_fn: RequirementsFn, judge_fn: AJudgeFn,
|
|
460
|
+
select_fn: SelectFn | None = None, obligation_pairs_fn: Any = None, defense_linker: Any = None,
|
|
461
|
+
constraint_scope_fn: Any = None, retry_policy: Any = DEFAULT_RETRY
|
|
462
|
+
):
|
|
463
|
+
"""Compile the compliance-check subgraph. All seams are injected for hermetic testing. `select_fn` (CC-8b) is
|
|
464
|
+
the per-claim requirement narrower; when None, every APPLICABLE requirement is judged (the broad default).
|
|
465
|
+
|
|
466
|
+
DEON-1/DEON-2 (issue 0012): when `obligation_pairs_fn` (`(obligations, claims, source) -> [(evidence_fact,
|
|
467
|
+
obligation)]`) is provided, the requirements are split by `deontic_route`: PROHIBITION/AMBIGUOUS rules are
|
|
468
|
+
judged PER-ASSERTION (narrowed by `select_fn`), OBLIGATION rules are judged ONCE each over a BOUNDED retrieved
|
|
469
|
+
evidence bundle (breach = absence, unanswerable per-sentence), and PERMISSION rules are excluded from
|
|
470
|
+
violation-judging. When None (the ad path, until DEON-8), the prior per-assertion pairing is kept.
|
|
471
|
+
|
|
472
|
+
DEON-9: when `defense_linker` (`rule -> [permission]`) is provided, each O/F rule is enriched with the same-
|
|
473
|
+
source PERMISSIONS that may EXCUSE it (attached as query-time `defenses`), passed to that rule's judge as
|
|
474
|
+
structured exception context so a legitimate carve-out is not a false violation. Query-side: each node degrades
|
|
475
|
+
to empty on failure (never crashes)."""
|
|
476
|
+
|
|
477
|
+
async def extract_claims(state: CheckState) -> CheckState:
|
|
478
|
+
with business_span("compliance_check.extract_claims"):
|
|
479
|
+
try:
|
|
480
|
+
claims = await claims_fn(state["subject_text"], state["source_doc"])
|
|
481
|
+
except Exception: # noqa: BLE001 - degrade-to-empty (query-side never crashes)
|
|
482
|
+
return {"claims": [], "ad_disclosures": []}
|
|
483
|
+
# ad-level disclosure union (CC-4 fix). getattr-tolerant: a generic CheckableFact has no disclosures ->
|
|
484
|
+
# empty union -> _enrich is a no-op, so the same pipeline serves both advertising Claims and bare facts.
|
|
485
|
+
ad = sorted({d for c in claims for d in getattr(c, "disclosures_present", [])})
|
|
486
|
+
return {"claims": claims, "ad_disclosures": ad}
|
|
487
|
+
|
|
488
|
+
def retrieve_applicable(state: CheckState) -> CheckState:
|
|
489
|
+
claims = state.get("claims", [])
|
|
490
|
+
ad = set(state.get("ad_disclosures", []))
|
|
491
|
+
with business_span("compliance_check.retrieve_applicable"):
|
|
492
|
+
try:
|
|
493
|
+
requirements = requirements_fn()
|
|
494
|
+
except Exception: # noqa: BLE001 - degrade-to-empty
|
|
495
|
+
return {"pairs": []}
|
|
496
|
+
if defense_linker is not None: # DEON-9: enrich each O/F rule with its same-source permission carve-outs
|
|
497
|
+
requirements = [_with_defenses(r, defense_linker) for r in requirements]
|
|
498
|
+
if obligation_pairs_fn is None: # ad path (until DEON-8): the prior per-assertion pairing
|
|
499
|
+
if select_fn is not None: # CC-8b: semantic narrowing (top-k content + always-include context + dedup)
|
|
500
|
+
pairs = [(_enrich(claim, ad), req) for claim in claims for req in select_fn(claim, requirements)]
|
|
501
|
+
else: # broad default: every applicable requirement
|
|
502
|
+
pairs = [(_enrich(claim, ad), req)
|
|
503
|
+
for claim in claims for req in requirements if applies_to(req, claim)]
|
|
504
|
+
return {"pairs": pairs}
|
|
505
|
+
# DEON-1/2: deontic split -- prohibitions per-assertion, obligations judged ONCE over bounded retrieved
|
|
506
|
+
# evidence, permissions excluded.
|
|
507
|
+
prohibitions = [r for r in requirements
|
|
508
|
+
if deontic_route(r) in (DeonticRoute.PROHIBITION, DeonticRoute.AMBIGUOUS)]
|
|
509
|
+
obligations = [r for r in requirements if deontic_route(r) is DeonticRoute.OBLIGATION]
|
|
510
|
+
pairs = []
|
|
511
|
+
for claim in claims: # prohibitions/ambiguous: per-assertion, where the subject asserts something related
|
|
512
|
+
selected = select_fn(claim, prohibitions) if select_fn is not None else prohibitions
|
|
513
|
+
pairs.extend((_enrich(claim, ad), req) for req in selected)
|
|
514
|
+
if obligations and claims: # obligations: one bounded (evidence, obligation) pair each
|
|
515
|
+
pairs.extend(obligation_pairs_fn(obligations, claims, state["source_doc"]))
|
|
516
|
+
# ADR-0068 (issue 0013): surface what the ACTOR gate skipped, so a symbolic drop is never a silent recall
|
|
517
|
+
# loss (empty is the recall-first norm; a disjoint role that blocked a pair shows up here).
|
|
518
|
+
gated = _actor_gated_pairs(claims, prohibitions, obligations, constraint_scope_fn)
|
|
519
|
+
return {"pairs": pairs, "gated": gated}
|
|
520
|
+
|
|
521
|
+
async def judge(state: CheckState) -> CheckState:
|
|
522
|
+
pairs = state.get("pairs", [])
|
|
523
|
+
if not pairs:
|
|
524
|
+
return {"findings": []}
|
|
525
|
+
with business_span("compliance_check.judge"):
|
|
526
|
+
# concurrent (gather + semaphore); conservative default inside (a timed-out pair -> needs_review)
|
|
527
|
+
return {"findings": await ajudge_pairs(pairs, ajudge_fn=judge_fn)}
|
|
528
|
+
|
|
529
|
+
def assemble(state: CheckState) -> CheckState:
|
|
530
|
+
findings: list[ComplianceFinding] = state.get("findings", [])
|
|
531
|
+
summary: dict[str, int] = {}
|
|
532
|
+
for f in findings:
|
|
533
|
+
summary[f.verdict.value] = summary.get(f.verdict.value, 0) + 1
|
|
534
|
+
# gap matrix: one row per requirement that was checked, rolled up to its worst verdict
|
|
535
|
+
rank = {Verdict.VIOLATION: 3, Verdict.NEEDS_REVIEW: 2, Verdict.COMPLIANT: 1}
|
|
536
|
+
by_req: dict[str, dict] = {}
|
|
537
|
+
for f in findings:
|
|
538
|
+
row = by_req.setdefault(f.requirement_id, {
|
|
539
|
+
"requirement_id": f.requirement_id, "citation": f.citation_requirement.split(" (")[0],
|
|
540
|
+
"verdict": f.verdict, "claims_checked": 0})
|
|
541
|
+
row["claims_checked"] += 1
|
|
542
|
+
if rank[f.verdict] > rank[row["verdict"]]:
|
|
543
|
+
row["verdict"] = f.verdict
|
|
544
|
+
gap_matrix = [{**r, "verdict": r["verdict"].value} for r in by_req.values()]
|
|
545
|
+
report = ComplianceReport(
|
|
546
|
+
source_doc=state["source_doc"], findings=findings, summary=summary, gap_matrix=gap_matrix,
|
|
547
|
+
gated_pairs=state.get("gated", [])) # ADR-0068: honest coverage -- the actor gate is never silent
|
|
548
|
+
return {"report": report}
|
|
549
|
+
|
|
550
|
+
g = StateGraph(CheckState)
|
|
551
|
+
g.add_node("extract_claims", extract_claims, retry_policy=retry_policy)
|
|
552
|
+
g.add_node("retrieve_applicable", retrieve_applicable, retry_policy=retry_policy)
|
|
553
|
+
g.add_node("judge", judge, retry_policy=retry_policy)
|
|
554
|
+
g.add_node("assemble", assemble)
|
|
555
|
+
g.add_edge(START, "extract_claims")
|
|
556
|
+
g.add_edge("extract_claims", "retrieve_applicable")
|
|
557
|
+
g.add_edge("retrieve_applicable", "judge")
|
|
558
|
+
g.add_edge("judge", "assemble")
|
|
559
|
+
g.add_edge("assemble", END)
|
|
560
|
+
return g.compile()
|
|
561
|
+
|
|
562
|
+
|
|
563
|
+
def _requirement_from_row(row: dict) -> Requirement:
|
|
564
|
+
"""Reconstruct a `Requirement` from a stored row (all_requirements), parsing applicability_json back to
|
|
565
|
+
constraints. Lenient: a bad row would raise, but the store wrote validated contracts."""
|
|
566
|
+
import json
|
|
567
|
+
|
|
568
|
+
from rag_wright.contracts.compliance import DeonticType, Severity
|
|
569
|
+
from rag_wright.contracts.provenance import ConfidenceTag
|
|
570
|
+
|
|
571
|
+
scope = [Constraint(dimension=d, value=v) for d, v in json.loads(row.get("applicability_json") or "[]")]
|
|
572
|
+
sev = row.get("severity") or None
|
|
573
|
+
_bbox = row.get("bbox") # issue 0043: best-effort [l,t,r,b] JSON string -> tuple, else None
|
|
574
|
+
bbox = tuple(json.loads(_bbox)) if _bbox else None
|
|
575
|
+
return Requirement(
|
|
576
|
+
requirement_id=row["requirement_id"], source=row["source"], citation=row["citation"],
|
|
577
|
+
deontic_type=DeonticType(row["deontic_type"]), actor=row["actor"],
|
|
578
|
+
applicability_scope=scope, requirement_text=row["requirement_text"],
|
|
579
|
+
evidence_standard=row.get("evidence_standard") or None,
|
|
580
|
+
severity=Severity(sev) if sev else None,
|
|
581
|
+
pages=[int(p) for p in (row.get("pages") or [])], bbox=bbox, # issue 0043: policy page provenance
|
|
582
|
+
confidence=ConfidenceTag(row.get("confidence") or "EXTRACTED"))
|
|
583
|
+
|
|
584
|
+
|
|
585
|
+
class UnknownComplianceSourceError(ValueError):
|
|
586
|
+
"""Issue 0007: a `sources` filter named a policy `source` that has no requirements in the store. Distinct from
|
|
587
|
+
a zero-requirement check (which a caller may treat as `not_checked`): naming a policy that does not exist is a
|
|
588
|
+
caller error, surfaced explicitly rather than silently matching nothing. Carries `.unknown` and `.present`."""
|
|
589
|
+
|
|
590
|
+
def __init__(self, unknown: list[str], present: list[str]) -> None:
|
|
591
|
+
self.unknown = unknown
|
|
592
|
+
self.present = present
|
|
593
|
+
super().__init__(f"unknown compliance source(s): {unknown}; present in store: {present}")
|
|
594
|
+
|
|
595
|
+
|
|
596
|
+
def _validate_sources(store: Any, sources: Optional[list[str]]) -> None:
|
|
597
|
+
"""SEG-7a: validate named policy `sources` against the store EARLY -- so an unknown source raises
|
|
598
|
+
`UnknownComplianceSourceError` BEFORE the expensive parse + assertion extraction, never wasting that work.
|
|
599
|
+
`None` (whole store) is not validated (a store without `requirement_sources()` still works)."""
|
|
600
|
+
if sources is None:
|
|
601
|
+
return
|
|
602
|
+
present = store.requirement_sources()
|
|
603
|
+
unknown = sorted(set(sources) - present)
|
|
604
|
+
if unknown:
|
|
605
|
+
raise UnknownComplianceSourceError(unknown, sorted(present))
|
|
606
|
+
|
|
607
|
+
|
|
608
|
+
def _load_requirements(store: Any, sources: Optional[list[str]] = None) -> list[Requirement]:
|
|
609
|
+
"""Issue 0007: load the Requirement rows the check runs against, optionally scoped to named policy `source`s.
|
|
610
|
+
|
|
611
|
+
`sources=None` -> the whole store (unchanged; `requirement_sources()` is NOT consulted, so a store without it
|
|
612
|
+
still works). A list -> validate the names against `store.requirement_sources()` (an unknown one raises
|
|
613
|
+
`UnknownComplianceSourceError`, not a silent empty match), then load ONLY those via the DB-side filter
|
|
614
|
+
(`store.all_requirements(sources=...)`). An empty list is a valid scope-to-nothing -> zero requirements."""
|
|
615
|
+
if sources is None:
|
|
616
|
+
rows = store.all_requirements()
|
|
617
|
+
else:
|
|
618
|
+
unknown = sorted(set(sources) - store.requirement_sources())
|
|
619
|
+
if unknown:
|
|
620
|
+
raise UnknownComplianceSourceError(unknown, sorted(store.requirement_sources()))
|
|
621
|
+
rows = store.all_requirements(sources=list(sources))
|
|
622
|
+
return [_requirement_from_row(r) for r in rows]
|
|
623
|
+
|
|
624
|
+
|
|
625
|
+
def production_compliance_check(
|
|
626
|
+
store: Any, *, extract_model: Any, judge_model_id: str, embedder: Any = None, k: int = 5,
|
|
627
|
+
sources: Optional[list[str]] = None, claims_fn: Any = None,
|
|
628
|
+
):
|
|
629
|
+
"""Wire the real capabilities: claims = claim_extraction (CC-3), requirements = the store's Requirement KG
|
|
630
|
+
(CC-5), judge = the Granite compliance judge (CC-4). Requirements are loaded ONCE here; when an `embedder` is
|
|
631
|
+
given, CC-8b semantic narrowing is enabled (top-k content + always-include context + dedup), else broad.
|
|
632
|
+
|
|
633
|
+
UNIFY-F: `claims_fn` (async `(text, source) -> [Claim]`) overrides the default whole-text extractor -- the ad
|
|
634
|
+
entrypoint injects PRECOMPUTED per-section claims through it (parsed once via the shared front-end)."""
|
|
635
|
+
from rag_wright.capabilities.claim_extraction import aclaim_extraction
|
|
636
|
+
from rag_wright.capabilities.compliance_judgment import build_acompliance_judge_fn
|
|
637
|
+
|
|
638
|
+
requirements = _load_requirements(store, sources)
|
|
639
|
+
select_fn = build_select_fn(embedder, requirements, k=k) if embedder is not None else None
|
|
640
|
+
# DEON-8 (issue 0012): the ad path gets the SAME deontic split as the generic path when an embedder is
|
|
641
|
+
# available -- OBLIGATIONS judged ONCE over bounded, actor-gated evidence (not per-sentence), PROHIBITIONS
|
|
642
|
+
# per-assertion (claim_type-routed via select_fn), PERMISSIONS excluded. Without an embedder (no semantic
|
|
643
|
+
# narrowing) the prior per-assertion pairing is kept (back-compat).
|
|
644
|
+
obligation_pairs_fn = build_obligation_pairs_fn(embedder) if embedder is not None else None
|
|
645
|
+
# DEON-9: permission carve-outs linked as defenses to the O/F rules they modify (needs the embedder for ranking).
|
|
646
|
+
defense_linker = build_defense_linker(embedder, requirements) if embedder is not None else None
|
|
647
|
+
|
|
648
|
+
async def _default_claims_fn(text: str, source: str) -> list:
|
|
649
|
+
return await aclaim_extraction(text, model=extract_model, source_doc=source)
|
|
650
|
+
|
|
651
|
+
return build_compliance_check(
|
|
652
|
+
claims_fn=claims_fn or _default_claims_fn,
|
|
653
|
+
requirements_fn=lambda: requirements,
|
|
654
|
+
judge_fn=build_acompliance_judge_fn(judge_model_id),
|
|
655
|
+
select_fn=select_fn,
|
|
656
|
+
obligation_pairs_fn=obligation_pairs_fn,
|
|
657
|
+
defense_linker=defense_linker,
|
|
658
|
+
)
|
|
659
|
+
|
|
660
|
+
|
|
661
|
+
_CLAIM_EXTRACT_ATTEMPTS = 3 # bounded retries for a per-chunk ad claim extraction (recover a transient docling blip)
|
|
662
|
+
|
|
663
|
+
|
|
664
|
+
async def _aextract_ad_claims(chunks: list[str], source_doc: str, extract_model: Any,
|
|
665
|
+
*, aclaim_fn: Any = None, max_concurrency: int = 4) -> list:
|
|
666
|
+
"""SEG-7b: the advertising claim extractor over the SAME semantic CHUNKS as the generic path -- extract typed
|
|
667
|
+
`Claim`s PER CHUNK (concurrently, per the parallel-LLM rule), re-indexed globally for unique ids. The
|
|
668
|
+
typed-Claim tail (claim_type / disclosures / routing) is UNTOUCHED. The structural locator (§/¶/bullet) is
|
|
669
|
+
attached AFTER, by `attach_structural_locators` (SEG-4, verbatim match), same as the generic path. `aclaim_fn`
|
|
670
|
+
is injected for hermetic tests."""
|
|
671
|
+
from rag_wright.capabilities.claim_extraction import aclaim_extraction
|
|
672
|
+
from rag_wright.contracts.compliance import Claim
|
|
673
|
+
|
|
674
|
+
fn = aclaim_fn or aclaim_extraction
|
|
675
|
+
sem = asyncio.Semaphore(max_concurrency)
|
|
676
|
+
|
|
677
|
+
async def _one(chunk: str) -> list:
|
|
678
|
+
text = (chunk or "").strip()
|
|
679
|
+
if not text:
|
|
680
|
+
return []
|
|
681
|
+
async with sem:
|
|
682
|
+
# PARTIAL-CAUSE-1 (compliance parity): the ad path extracts claims OUTSIDE the retry graph, so wrap
|
|
683
|
+
# the per-chunk call in a bounded retry. docling-graph's `ExtractionFailed` is raised on ANY logged
|
|
684
|
+
# error incl. TRANSIENT blips (empty content / gleaning / rate-limit / timeout), so retrying is what
|
|
685
|
+
# recovers them (the same fix as the contract clause extractor). A PERSISTENT failure re-raises --
|
|
686
|
+
# surfaced loudly, a chunk's claims are never silently dropped.
|
|
687
|
+
last_exc: Optional[BaseException] = None
|
|
688
|
+
for _attempt in range(_CLAIM_EXTRACT_ATTEMPTS):
|
|
689
|
+
try:
|
|
690
|
+
return await fn(text, model=extract_model, source_doc=source_doc)
|
|
691
|
+
except Exception as exc: # noqa: BLE001 - transient docling/LLM error -> retry; persistent -> raise
|
|
692
|
+
last_exc = exc
|
|
693
|
+
raise last_exc # type: ignore[misc] # persistent failure after retries (never None here)
|
|
694
|
+
|
|
695
|
+
per_chunk = await asyncio.gather(*(_one(c) for c in chunks))
|
|
696
|
+
claims = [c for group in per_chunk for c in group]
|
|
697
|
+
for i, c in enumerate(claims): # global re-index -> unique claim ids across chunks
|
|
698
|
+
c.fact_id = Claim.make_id(source_doc, i, c.assertion_text)
|
|
699
|
+
return claims
|
|
700
|
+
|
|
701
|
+
|
|
702
|
+
async def run_ad_compliance_check(
|
|
703
|
+
subject_text: Optional[str] = None, source_doc: str = "", *, store: Any, extract_model: Any,
|
|
704
|
+
judge_model_id: str, embedder: Any = None, k: int = 5, sources: Optional[list[str]] = None,
|
|
705
|
+
name: Optional[str] = None, data: Optional[bytes] = None, discoverer: Any = None, aclaim_fn: Any = None,
|
|
706
|
+
doc: Any = None,
|
|
707
|
+
) -> ComplianceReport:
|
|
708
|
+
"""The ADVERTISING compliance path (the subgraph behind the `check_ad_compliance` MCP tool; the generic
|
|
709
|
+
counterpart is `run_generic_compliance_verdict`) -> a cited `ComplianceReport`. SEG-7b: accepts EITHER a pasted
|
|
710
|
+
`subject_text` OR an uploaded ad (`name` + raw `data` bytes), and runs the SAME semantic front-end as the
|
|
711
|
+
generic path -- parse -> semantic chunk -> per-chunk typed-`Claim` extraction -> attach structural locators
|
|
712
|
+
(§/¶/bullet) -> judge. The typed-Claim tail (claim_type / disclosure routing) is UNCHANGED; a scanned ad's
|
|
713
|
+
unreadable pages surface on `report.ocr_unreadable_pages` (SEG-6). `sources` (0007) scopes to named policies;
|
|
714
|
+
`discoverer`/`aclaim_fn`/`doc` inject for tests."""
|
|
715
|
+
_validate_sources(store, sources) # reject an unknown policy BEFORE the expensive parse + extraction
|
|
716
|
+
parsed_doc, unreadable = await _aparse_subject_any(text=subject_text, name=name, data=data, doc=doc)
|
|
717
|
+
chunks = await subject_chunks(parsed_doc, discoverer=discoverer)
|
|
718
|
+
claims = await _aextract_ad_claims(chunks, source_doc, extract_model, aclaim_fn=aclaim_fn)
|
|
719
|
+
attach_structural_locators(claims, parsed_doc) # SEG-4: verbatim-match each claim to its docling element
|
|
720
|
+
|
|
721
|
+
async def _precomputed_claims_fn(_text: str, _source: str) -> list:
|
|
722
|
+
return claims # extracted once, per-chunk, above
|
|
723
|
+
|
|
724
|
+
graph = production_compliance_check(
|
|
725
|
+
store, extract_model=extract_model, judge_model_id=judge_model_id, embedder=embedder, k=k, sources=sources,
|
|
726
|
+
claims_fn=_precomputed_claims_fn)
|
|
727
|
+
subject_text_joined = "\n\n".join(c.assertion_text for c in claims)
|
|
728
|
+
out = await graph.ainvoke({"subject_text": subject_text_joined, "source_doc": source_doc})
|
|
729
|
+
report = out["report"]
|
|
730
|
+
report.ocr_unreadable_pages = unreadable # SEG-6: the ad path surfaces the OCR PARTIAL too
|
|
731
|
+
return report
|
|
732
|
+
|
|
733
|
+
|
|
734
|
+
def production_generic_compliance_check(store: Any, *, judge_model_id: str, embedder: Any, k: int = 8,
|
|
735
|
+
sources: Optional[list[str]] = None, facts_fn: Any):
|
|
736
|
+
"""COMP-VERDICT-GENERIC: wire the DOMAIN-AGNOSTIC verdict path -- generic subject facts (no claim_type),
|
|
737
|
+
SEMANTIC-ONLY requirement narrowing (`filter_applicability=False`, no domain applicability ontology needed),
|
|
738
|
+
and the GENERIC judge (text-only). Gives a cited LLM verdict in ANY compliance domain; enrichment
|
|
739
|
+
(COMP-APPLIC-1) only ADDS structured precision on top. `embedder` is required (semantic retrieval is the
|
|
740
|
+
narrowing here). `sources` (issue 0007) optionally scopes the check to named policy `source`s (None = the
|
|
741
|
+
whole store). `facts_fn` (required) is the async-adapted producer `(text, source) -> [CheckableFact]`; the
|
|
742
|
+
caller (`run_subject_compliance_verdict`) precomputes the SEMANTIC facts and injects them here (SEG-7a)."""
|
|
743
|
+
from rag_wright.capabilities.compliance_judgment import build_ageneric_judge_fn
|
|
744
|
+
|
|
745
|
+
requirements = _load_requirements(store, sources)
|
|
746
|
+
# DEON-6: prohibitions narrow by the dimension-agnostic constraint router -- a claim's inferred scope (its
|
|
747
|
+
# actor, DEON-5) matched against each requirement's effective scope. Recall-first: a claim with no scope, or a
|
|
748
|
+
# requirement with a generic actor, is not excluded.
|
|
749
|
+
constraint_scope_fn = lambda claim: getattr(claim, "scope", None) or [] # noqa: E731 (a claim's inferred scope)
|
|
750
|
+
select_fn = build_select_fn(embedder, requirements, k=k, filter_applicability=False,
|
|
751
|
+
constraint_scope_fn=constraint_scope_fn)
|
|
752
|
+
|
|
753
|
+
async def _claims_fn(text: str, source: str) -> list:
|
|
754
|
+
return facts_fn(text, source) # precomputed semantic facts, adapted to the async claims seam
|
|
755
|
+
|
|
756
|
+
return build_compliance_check(
|
|
757
|
+
claims_fn=_claims_fn,
|
|
758
|
+
requirements_fn=lambda: requirements,
|
|
759
|
+
judge_fn=build_ageneric_judge_fn(judge_model_id),
|
|
760
|
+
select_fn=select_fn,
|
|
761
|
+
obligation_pairs_fn=build_obligation_pairs_fn(embedder), # DEON-2: bounded per-obligation evidence
|
|
762
|
+
defense_linker=build_defense_linker(embedder, requirements), # DEON-9: permission carve-outs as defenses
|
|
763
|
+
constraint_scope_fn=constraint_scope_fn, # ADR-0068: report the actor-gated prohibition pairs (issue 0013)
|
|
764
|
+
)
|
|
765
|
+
|
|
766
|
+
|
|
767
|
+
async def run_generic_compliance_verdict(
|
|
768
|
+
subject_text: str, source_doc: str, *, store: Any, judge_model_id: str, embedder: Any, k: int = 8,
|
|
769
|
+
sources: Optional[list[str]] = None, extract_model: Any = None, discoverer: Any = None,
|
|
770
|
+
aextract_fn: Any = None,
|
|
771
|
+
) -> ComplianceReport:
|
|
772
|
+
"""COMP-VERDICT-GENERIC: a domain-agnostic compliance verdict for a free-text subject against the Requirement
|
|
773
|
+
KG -> a cited `ComplianceReport`. Works with NO domain applicability enrichment (the always-answer guarantee).
|
|
774
|
+
|
|
775
|
+
`sources` (issue 0007) optionally scopes the check to named policy `source`s -- None checks against the whole
|
|
776
|
+
store; a list checks against ONLY those policies; `[]` scopes to nothing; an unknown name raises
|
|
777
|
+
`UnknownComplianceSourceError`.
|
|
778
|
+
|
|
779
|
+
SEG-7a: a thin shim over `run_subject_compliance_verdict` (text mode). The paste is parsed through docling and
|
|
780
|
+
run through the SAME semantic pipeline as an upload (chunk -> verbatim assertion extraction -> locator ->
|
|
781
|
+
judge); a structureless paste yields per-assertion facts with NO `§` locator. `extract_model`/`discoverer`/
|
|
782
|
+
`aextract_fn` are passed through (the latter two inject for tests)."""
|
|
783
|
+
return await run_subject_compliance_verdict(
|
|
784
|
+
source_doc, store=store, judge_model_id=judge_model_id, embedder=embedder, k=k, sources=sources,
|
|
785
|
+
text=subject_text, extract_model=extract_model, discoverer=discoverer, aextract_fn=aextract_fn)
|
|
786
|
+
|
|
787
|
+
|
|
788
|
+
_HEADING_KINDS = frozenset({"section_header", "title", "field_heading"})
|
|
789
|
+
|
|
790
|
+
|
|
791
|
+
def _merge_wrapped_items(raw: list[tuple[str, str]]) -> list[tuple[str, str]]:
|
|
792
|
+
"""SEG-4: coalesce docling's line-split of a wrapped paragraph back into ONE logical element. A line-based
|
|
793
|
+
backend (markdown, a hard-wrapped .txt) emits each physical line as a separate `text` item; a mid-sentence
|
|
794
|
+
line break is a soft-wrap, NOT a paragraph boundary. Signal: the previous same-kind body item does NOT end
|
|
795
|
+
with sentence-terminal punctuation (`.`/`!`/`?`) -> the current item continues it, so merge. Headings never
|
|
796
|
+
merge; a line ending in terminal punctuation starts a new element (a genuine paragraph break)."""
|
|
797
|
+
merged: list[list[str]] = []
|
|
798
|
+
for kind, text in raw:
|
|
799
|
+
if (kind not in _HEADING_KINDS and text and merged
|
|
800
|
+
and merged[-1][0] == kind and merged[-1][1] and merged[-1][1][-1] not in ".!?"):
|
|
801
|
+
merged[-1][1] = f"{merged[-1][1]} {text}" # soft-wrap continuation of the same logical element
|
|
802
|
+
else:
|
|
803
|
+
merged.append([kind, text])
|
|
804
|
+
return [(k, t) for k, t in merged]
|
|
805
|
+
|
|
806
|
+
|
|
807
|
+
def _item_provenance(parsed_doc: Any) -> list[dict]:
|
|
808
|
+
"""SEG-4: docling items -> per-LOGICAL-ELEMENT structural provenance `{text, section, element_kind,
|
|
809
|
+
element_ordinal}`. Line-wrapped paragraphs are merged first (`_merge_wrapped_items`) so ordinals count real
|
|
810
|
+
paragraphs, not physical lines. `section` = the enclosing section number (shared `_section_number`; None
|
|
811
|
+
before the first heading -> a flat doc stays section-less). Ordinals count WITHIN a section, PER KIND (¶ for
|
|
812
|
+
body text, bullet for list items), reset at each heading."""
|
|
813
|
+
from rag_wright.corpus.document_parser import _section_number
|
|
814
|
+
|
|
815
|
+
raw: list[tuple[str, str]] = []
|
|
816
|
+
for item in getattr(parsed_doc, "texts", []) or []:
|
|
817
|
+
lab = getattr(item, "label", "")
|
|
818
|
+
kind = str(getattr(lab, "value", lab) or "") # DocItemLabel enum -> its value; a plain string stays as-is
|
|
819
|
+
text = (getattr(item, "text", "") or "").strip()
|
|
820
|
+
if not text and kind not in _HEADING_KINDS:
|
|
821
|
+
continue # empty body item: nothing to locate or count
|
|
822
|
+
raw.append((kind, text))
|
|
823
|
+
|
|
824
|
+
out: list[dict] = []
|
|
825
|
+
section: Optional[str] = None
|
|
826
|
+
n_sections = 0
|
|
827
|
+
para_ord = 0
|
|
828
|
+
bullet_ord = 0
|
|
829
|
+
for kind, text in _merge_wrapped_items(raw):
|
|
830
|
+
if kind in _HEADING_KINDS: # a heading opens a new section and resets the within-section ordinals
|
|
831
|
+
n_sections += 1
|
|
832
|
+
section = _section_number(text, n_sections)
|
|
833
|
+
para_ord = bullet_ord = 0
|
|
834
|
+
out.append({"text": text, "section": section, "element_kind": kind, "element_ordinal": None})
|
|
835
|
+
continue
|
|
836
|
+
if kind == "list_item":
|
|
837
|
+
bullet_ord += 1
|
|
838
|
+
ordinal: Optional[int] = bullet_ord
|
|
839
|
+
else:
|
|
840
|
+
para_ord += 1
|
|
841
|
+
ordinal = para_ord
|
|
842
|
+
out.append({"text": text, "section": section, "element_kind": kind, "element_ordinal": ordinal})
|
|
843
|
+
return out
|
|
844
|
+
|
|
845
|
+
|
|
846
|
+
def _norm_ws(s: str) -> str:
|
|
847
|
+
return " ".join((s or "").split())
|
|
848
|
+
|
|
849
|
+
|
|
850
|
+
def attach_structural_locators(facts: list, parsed_doc: Any) -> list:
|
|
851
|
+
"""SEG-4: stamp each `CheckableFact` with the structural locator of the docling element its VERBATIM assertion
|
|
852
|
+
came from (`section`, `element_kind`, `element_ordinal`) -> `locator()` renders `§ N ¶M` / `§ N · bullet M`.
|
|
853
|
+
|
|
854
|
+
Robust to real-world parsing: docling can split a soft-wrapped paragraph (or a hard-wrapped .txt) into several
|
|
855
|
+
consecutive `text` items, so an assertion may SPAN items. We therefore match against the whitespace-normalized
|
|
856
|
+
CONCATENATION of the body items (each item's char range recorded), find the assertion, and attribute it to the
|
|
857
|
+
item where it STARTS -- so a cross-item assertion is located, never dropped. Unmatched (genuinely absent /
|
|
858
|
+
heavily paraphrased) stays unlocated, still citable by its text. Line-wrapped paragraphs are merged in
|
|
859
|
+
`_item_provenance` so ordinals count real paragraphs, not physical lines. Mutates + returns `facts`."""
|
|
860
|
+
body = [p for p in _item_provenance(parsed_doc) if _norm_ws(p["text"])]
|
|
861
|
+
concat = ""
|
|
862
|
+
ranges: list[tuple[int, int, dict]] = [] # (start, end, provenance) in the normalized concatenation
|
|
863
|
+
for p in body:
|
|
864
|
+
t = _norm_ws(p["text"])
|
|
865
|
+
start = len(concat)
|
|
866
|
+
concat += t + " "
|
|
867
|
+
ranges.append((start, start + len(t), p))
|
|
868
|
+
for f in facts:
|
|
869
|
+
needle = _norm_ws(f.assertion_text)
|
|
870
|
+
if not needle:
|
|
871
|
+
continue
|
|
872
|
+
pos = concat.find(needle)
|
|
873
|
+
if pos < 0:
|
|
874
|
+
continue
|
|
875
|
+
for start, end, p in ranges: # attribute to the item where the assertion STARTS
|
|
876
|
+
if start <= pos < end:
|
|
877
|
+
f.section = p["section"]
|
|
878
|
+
f.element_kind = p["element_kind"]
|
|
879
|
+
f.element_ordinal = p["element_ordinal"]
|
|
880
|
+
break
|
|
881
|
+
return facts
|
|
882
|
+
|
|
883
|
+
|
|
884
|
+
async def aextract_subject_facts(chunks: list[str], *, source_doc: str, model: Any, aextract_fn: Any = None,
|
|
885
|
+
max_concurrency: int = 4) -> list:
|
|
886
|
+
"""SEG-3: extract the checkable assertions (verbatim) from each subject CHUNK CONCURRENTLY (semaphore, per the
|
|
887
|
+
parallel-LLM rule) -> `CheckableFact`s, re-indexed globally so `fact_id`s are unique across chunks.
|
|
888
|
+
Domain-neutral (`CheckableFact`, no `claim_type` -- that is the ad path). The structural locator
|
|
889
|
+
(section / ¶ / bullet) is attached later, in SEG-4. `aextract_fn` is injected for hermetic tests."""
|
|
890
|
+
from rag_wright.capabilities.assertion_extraction import aassertion_extraction
|
|
891
|
+
from rag_wright.capabilities.dg_extraction import aextract_parties
|
|
892
|
+
|
|
893
|
+
fn = aextract_fn or aextract_parties
|
|
894
|
+
sem = asyncio.Semaphore(max_concurrency)
|
|
895
|
+
|
|
896
|
+
async def _one(chunk: str) -> list:
|
|
897
|
+
async with sem:
|
|
898
|
+
return await aassertion_extraction(chunk, model=model, source_doc=source_doc, aextract_fn=fn)
|
|
899
|
+
|
|
900
|
+
per_chunk = await asyncio.gather(*(_one(c) for c in chunks))
|
|
901
|
+
facts = [f for group in per_chunk for f in group]
|
|
902
|
+
for i, f in enumerate(facts): # global re-index -> unique fact_ids across chunks
|
|
903
|
+
f.fact_id = CheckableFact.make_id(source_doc, i, f.assertion_text)
|
|
904
|
+
return facts
|
|
905
|
+
|
|
906
|
+
|
|
907
|
+
async def subject_chunks(parsed_doc: Any, *, discoverer: Any = None) -> list[str]:
|
|
908
|
+
"""SEG-2/SEG-5: semantically chunk a parsed subject document into coherent chunk texts, via the SAME shared
|
|
909
|
+
chunker as ingestion (`achunk_texts`; no cache/summarize -- the subject is transient). The default discoverer
|
|
910
|
+
is `StructuralModelFallbackDiscoverer` -- the exact discoverer PRODUCTION INGESTION uses: structural boundaries
|
|
911
|
+
first (no model for a structured doc), a BOUNDED per-section model refinement only for an over-cap section, so
|
|
912
|
+
cost never scales with document length (no size bottleneck) and there is NO subject-specific large-doc code.
|
|
913
|
+
`parsed_doc` is the docling document (exposes `.texts`). RLM is a future escalation, as on ingestion. SEG-3
|
|
914
|
+
extracts the checkable assertions from each returned chunk."""
|
|
915
|
+
from rag_wright.capabilities.rlm_chunking import achunk_texts
|
|
916
|
+
|
|
917
|
+
return await achunk_texts(parsed_doc, discoverer=discoverer)
|
|
918
|
+
|
|
919
|
+
|
|
920
|
+
async def semantic_subject_facts(parsed_doc: Any, *, source_doc: str, model: Any = None, discoverer: Any = None,
|
|
921
|
+
aextract_fn: Any = None) -> list:
|
|
922
|
+
"""SEG-7a: the SUBJECT fact producer -- the composed semantic pipeline that supersedes the old regex
|
|
923
|
+
sections producer. Chunk the parsed doc (SEG-2/SEG-5, the production ingestion discoverer) -> extract the
|
|
924
|
+
checkable assertions VERBATIM per chunk (SEG-3) -> attach each to its docling element for the structural
|
|
925
|
+
locator (SEG-4). Returns `CheckableFact`s cited "doc § {section} ¶{n}: {verbatim}" (or just the span for a
|
|
926
|
+
flat doc). `model` (assertion extractor) defaults to the production extraction model; `discoverer`/`aextract_fn`
|
|
927
|
+
are injected for hermetic tests (no model/parse)."""
|
|
928
|
+
m = model
|
|
929
|
+
if m is None and aextract_fn is None:
|
|
930
|
+
from rag_wright.capabilities.dg_extraction import default_extraction_model
|
|
931
|
+
|
|
932
|
+
m = default_extraction_model("subject-assert")
|
|
933
|
+
chunks = await subject_chunks(parsed_doc, discoverer=discoverer)
|
|
934
|
+
facts = await aextract_subject_facts(chunks, source_doc=source_doc, model=m, aextract_fn=aextract_fn)
|
|
935
|
+
attach_structural_locators(facts, parsed_doc)
|
|
936
|
+
return facts
|
|
937
|
+
|
|
938
|
+
|
|
939
|
+
async def _aparse_subject_any(*, text: Optional[str], name: Optional[str], data: Optional[bytes],
|
|
940
|
+
doc: Any = None) -> tuple[Any, list[int]]:
|
|
941
|
+
"""SEG-7a: parse ANY subject input to a docling document + OCR unreadable pages, for the uniform semantic
|
|
942
|
+
pipeline. `doc` (a test injection) is returned as-is. Upload (`data`): `aparse_subject` (tiered OCR, captures
|
|
943
|
+
unreadable pages). Paste (`text`): parsed through docling as `.txt` bytes (decision A -- uniform semantic
|
|
944
|
+
handling, no OCR pages), NOT the old short-circuit."""
|
|
945
|
+
if doc is not None:
|
|
946
|
+
return doc, []
|
|
947
|
+
if data is not None:
|
|
948
|
+
return await aparse_subject(name, data)
|
|
949
|
+
if text is not None:
|
|
950
|
+
from rag_wright.corpus.document_parser import aparse_document_bytes
|
|
951
|
+
|
|
952
|
+
return await aparse_document_bytes("subject.txt", text.encode("utf-8")), []
|
|
953
|
+
raise ValueError("run_subject_compliance_verdict needs either text= or (name=, data=)")
|
|
954
|
+
|
|
955
|
+
|
|
956
|
+
async def aparse_subject(name: str, data: bytes, *, parser: Any = None) -> tuple[Any, list[int]]:
|
|
957
|
+
"""SEG-6: parse an uploaded subject ONCE through the tiered OCR chokepoint (`TieredOCRParser`, the SAME OCR as
|
|
958
|
+
ingestion), returning `(docling_document, ocr_unreadable_pages)`. The unreadable pages (a degraded scan the
|
|
959
|
+
VLM still could not read) are captured from the tiered parser's report so they can surface on the
|
|
960
|
+
`ComplianceReport` -- a verdict is never silently based on half-read text. `parser` injected for tests."""
|
|
961
|
+
from rag_wright.capabilities.parsing import TieredOCRParser
|
|
962
|
+
from rag_wright.corpus.document_parser import aparse_document_bytes
|
|
963
|
+
|
|
964
|
+
tiered = parser if parser is not None else TieredOCRParser()
|
|
965
|
+
document = await aparse_document_bytes(name, data, parser=tiered)
|
|
966
|
+
unreadable = list(getattr(getattr(tiered, "report", None), "unreadable_pages", []) or [])
|
|
967
|
+
return document, unreadable
|
|
968
|
+
|
|
969
|
+
|
|
970
|
+
async def run_subject_compliance_verdict(
|
|
971
|
+
source_doc: str, *, store: Any, judge_model_id: str, embedder: Any, k: int = 8,
|
|
972
|
+
sources: Optional[list[str]] = None, text: Optional[str] = None, name: Optional[str] = None,
|
|
973
|
+
data: Optional[bytes] = None, extract_model: Any = None, discoverer: Any = None, aextract_fn: Any = None,
|
|
974
|
+
doc: Any = None,
|
|
975
|
+
) -> ComplianceReport:
|
|
976
|
+
"""SEG-7a: the ONE subject-compliance front-end. Accepts EITHER a pasted `text` OR an uploaded document
|
|
977
|
+
(`name` + raw `data` bytes: PDF/DOCX/HTML/TXT), and runs the SEMANTIC pipeline uniformly: parse -> semantic
|
|
978
|
+
chunk (the production ingestion discoverer) -> extract the checkable assertions VERBATIM per chunk -> attach
|
|
979
|
+
each to its docling element -> judge. Findings cite "doc § {section} ¶{n}: {verbatim}" (or just the span for a
|
|
980
|
+
structureless subject). The subject is TRANSIENT (parsed/checked, never written to the store); a scanned
|
|
981
|
+
subject's unreadable pages surface on `report.ocr_unreadable_pages` (SEG-6).
|
|
982
|
+
|
|
983
|
+
`extract_model` (assertion extractor) defaults to the production extraction model. `sources` (0007) scopes to
|
|
984
|
+
named policies (unknown -> `UnknownComplianceSourceError`). `discoverer`/`aextract_fn`/`doc` are injected for
|
|
985
|
+
hermetic tests. (SEG-7a replaced the old regex sections/granularity dial with the semantic producer.)"""
|
|
986
|
+
_validate_sources(store, sources) # SEG-7a: reject an unknown policy BEFORE the expensive parse + extraction
|
|
987
|
+
parsed_doc, unreadable = await _aparse_subject_any(text=text, name=name, data=data, doc=doc)
|
|
988
|
+
facts = await semantic_subject_facts(
|
|
989
|
+
parsed_doc, source_doc=source_doc, model=extract_model, discoverer=discoverer, aextract_fn=aextract_fn)
|
|
990
|
+
graph = production_generic_compliance_check(
|
|
991
|
+
store, judge_model_id=judge_model_id, embedder=embedder, k=k, sources=sources,
|
|
992
|
+
facts_fn=lambda _text, _source: facts) # precomputed semantic subject facts
|
|
993
|
+
subject_text = "\n\n".join(f.assertion_text for f in facts)
|
|
994
|
+
out = await graph.ainvoke({"subject_text": subject_text, "source_doc": source_doc})
|
|
995
|
+
report = out["report"]
|
|
996
|
+
report.ocr_unreadable_pages = unreadable # SEG-6: surface the OCR PARTIAL so a verdict is never silently partial
|
|
997
|
+
return report
|
|
998
|
+
|
|
999
|
+
|
|
1000
|
+
async def run_compliance_document_verdict(
|
|
1001
|
+
doc_name: str, data: bytes, *, store: Any, judge_model_id: str, embedder: Any, k: int = 8,
|
|
1002
|
+
sources: Optional[list[str]] = None, extract_model: Any = None, discoverer: Any = None,
|
|
1003
|
+
aextract_fn: Any = None, doc: Any = None,
|
|
1004
|
+
) -> ComplianceReport:
|
|
1005
|
+
"""Issue 0008 / SEG-7a: check an uploaded subject DOCUMENT (raw bytes: PDF/DOCX/HTML/TXT) for compliance --
|
|
1006
|
+
a thin shim over `run_subject_compliance_verdict` (the shared SEMANTIC front-end). Findings cite each verbatim
|
|
1007
|
+
assertion's "§ {section} ¶{n}" locator; a scanned subject's unreadable pages surface on the report (SEG-6).
|
|
1008
|
+
`sources` (0007) scopes to named policies; `discoverer`/`aextract_fn`/`doc` inject for tests."""
|
|
1009
|
+
return await run_subject_compliance_verdict(
|
|
1010
|
+
doc_name, store=store, judge_model_id=judge_model_id, embedder=embedder, k=k, sources=sources,
|
|
1011
|
+
name=doc_name, data=data, extract_model=extract_model, discoverer=discoverer, aextract_fn=aextract_fn,
|
|
1012
|
+
doc=doc)
|
|
1013
|
+
|
|
1014
|
+
|
|
1015
|
+
def register_compliance_check(registry) -> None:
|
|
1016
|
+
"""Register `compliance_check` (subgraph; CC-6). Contract = `ComplianceReport`."""
|
|
1017
|
+
registry.register(
|
|
1018
|
+
"compliance_check",
|
|
1019
|
+
contract=ComplianceReport,
|
|
1020
|
+
kind="subgraph",
|
|
1021
|
+
display_name="Compliance check (subject doc x requirements -> cited findings + gap matrix)",
|
|
1022
|
+
)
|
|
1023
|
+
|
|
1024
|
+
|
|
1025
|
+
async def ainvoke(resources, inputs: dict):
|
|
1026
|
+
"""EP-REF-1c (ADR-0118): the capability invoke factory (impl_ref target) for the GENERIC compliance check --
|
|
1027
|
+
NOT the FTC-tuned `run_ad_compliance_check` (the product owns that variant + its guardrails). The store, the
|
|
1028
|
+
judge model (STRUCTURED_REASONING), and the embedder come from the workspace handle; the subject + scope from
|
|
1029
|
+
`inputs`. Two subject shapes: `{subject_text, source_doc}` (text) or `{doc_name, data}` (raw document bytes).
|
|
1030
|
+
`inputs` may also carry `k` (retrieval depth) and `sources` (scope to named policies)."""
|
|
1031
|
+
from rag_wright.models.profiles import ModelRole
|
|
1032
|
+
|
|
1033
|
+
judge = resources.model_id(ModelRole.STRUCTURED_REASONING)
|
|
1034
|
+
k = inputs.get("k", 8)
|
|
1035
|
+
sources = inputs.get("sources")
|
|
1036
|
+
if inputs.get("data") is not None: # a subject DOCUMENT (bytes)
|
|
1037
|
+
return await run_compliance_document_verdict(
|
|
1038
|
+
inputs["doc_name"], inputs["data"], store=resources._store, judge_model_id=judge,
|
|
1039
|
+
embedder=resources._embedder, k=k, sources=sources)
|
|
1040
|
+
return await run_generic_compliance_verdict( # a subject TEXT
|
|
1041
|
+
inputs["subject_text"], inputs["source_doc"], store=resources._store, judge_model_id=judge,
|
|
1042
|
+
embedder=resources._embedder, k=k, sources=sources)
|