rag-wright 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- rag_wright/__init__.py +13 -0
- rag_wright/api/__init__.py +33 -0
- rag_wright/api/config.py +59 -0
- rag_wright/api/discover.py +70 -0
- rag_wright/api/documents.py +39 -0
- rag_wright/api/ids.py +31 -0
- rag_wright/api/invoke.py +99 -0
- rag_wright/api/kg.py +61 -0
- rag_wright/api/mcp.py +94 -0
- rag_wright/api/usage.py +30 -0
- rag_wright/api/workspace.py +85 -0
- rag_wright/capabilities/__init__.py +8 -0
- rag_wright/capabilities/answer_generator.py +427 -0
- rag_wright/capabilities/ard.py +286 -0
- rag_wright/capabilities/assertion_extraction.py +79 -0
- rag_wright/capabilities/chunk_read.py +58 -0
- rag_wright/capabilities/chunk_write.py +163 -0
- rag_wright/capabilities/claim_extraction.py +153 -0
- rag_wright/capabilities/clause_exception_linking.py +117 -0
- rag_wright/capabilities/compliance_judgment.py +322 -0
- rag_wright/capabilities/compliance_store.py +87 -0
- rag_wright/capabilities/contract_kg_serve.py +156 -0
- rag_wright/capabilities/contract_kg_store.py +251 -0
- rag_wright/capabilities/dg_extraction.py +585 -0
- rag_wright/capabilities/disambiguation.py +163 -0
- rag_wright/capabilities/document_parse.py +87 -0
- rag_wright/capabilities/document_scope.py +49 -0
- rag_wright/capabilities/embedding.py +164 -0
- rag_wright/capabilities/embedding_profiles.py +43 -0
- rag_wright/capabilities/entity_resolution.py +154 -0
- rag_wright/capabilities/fusion.py +64 -0
- rag_wright/capabilities/graph_extraction.py +243 -0
- rag_wright/capabilities/graph_query.py +73 -0
- rag_wright/capabilities/graph_storage.py +111 -0
- rag_wright/capabilities/highlight_serve.py +142 -0
- rag_wright/capabilities/hybrid_search.py +65 -0
- rag_wright/capabilities/invoke.py +31 -0
- rag_wright/capabilities/jev_decision.py +38 -0
- rag_wright/capabilities/manifests.py +872 -0
- rag_wright/capabilities/okf_navigate.py +456 -0
- rag_wright/capabilities/parsing.py +286 -0
- rag_wright/capabilities/property_boosted_retrieval.py +125 -0
- rag_wright/capabilities/query_function_classifier.py +94 -0
- rag_wright/capabilities/query_understanding.py +109 -0
- rag_wright/capabilities/registry.py +262 -0
- rag_wright/capabilities/remote_encoders.py +94 -0
- rag_wright/capabilities/requirement_extraction.py +247 -0
- rag_wright/capabilities/reranking.py +123 -0
- rag_wright/capabilities/retrieval_core.py +126 -0
- rag_wright/capabilities/rlm_chunking.py +808 -0
- rag_wright/capabilities/rlm_synthesis.py +316 -0
- rag_wright/capabilities/scan_quality.py +136 -0
- rag_wright/capabilities/span_relevance_judgment.py +191 -0
- rag_wright/capabilities/vision_to_text.py +85 -0
- rag_wright/capabilities/vlm_ocr.py +85 -0
- rag_wright/contracts/__init__.py +6 -0
- rag_wright/contracts/chunk.py +79 -0
- rag_wright/contracts/compliance.py +303 -0
- rag_wright/contracts/contract_meta.py +27 -0
- rag_wright/contracts/extraction.py +130 -0
- rag_wright/contracts/function.py +167 -0
- rag_wright/contracts/function_routing.py +91 -0
- rag_wright/contracts/highlight.py +74 -0
- rag_wright/contracts/identifiers.py +153 -0
- rag_wright/contracts/jurisdiction.py +96 -0
- rag_wright/contracts/ontology.py +142 -0
- rag_wright/contracts/property.py +201 -0
- rag_wright/contracts/provenance.py +78 -0
- rag_wright/contracts/query_intent.py +53 -0
- rag_wright/contracts/span.py +76 -0
- rag_wright/contracts/value_match.py +84 -0
- rag_wright/corpus/__init__.py +0 -0
- rag_wright/corpus/canonicalize.py +116 -0
- rag_wright/corpus/cuad.py +153 -0
- rag_wright/corpus/cuad_ingestion.py +72 -0
- rag_wright/corpus/document_parser.py +299 -0
- rag_wright/corpus/edgar.py +231 -0
- rag_wright/corpus/gcs_ingestion.py +120 -0
- rag_wright/corpus/http.py +110 -0
- rag_wright/corpus/selection.py +152 -0
- rag_wright/mcp/__init__.py +11 -0
- rag_wright/mcp/compliance_server.py +299 -0
- rag_wright/mcp/intra_document_qa_server.py +170 -0
- rag_wright/mcp/relational_qa_server.py +171 -0
- rag_wright/mcp/session_store.py +64 -0
- rag_wright/mcp/typed_property_retrieval_server.py +191 -0
- rag_wright/models/__init__.py +8 -0
- rag_wright/models/profiles.py +331 -0
- rag_wright/models/seam.py +497 -0
- rag_wright/models/tag_structured.py +285 -0
- rag_wright/models/tracing.py +179 -0
- rag_wright/models/usage.py +102 -0
- rag_wright/okf/__init__.py +11 -0
- rag_wright/okf/compile.py +292 -0
- rag_wright/okf/document.py +47 -0
- rag_wright/okf/enrich.py +176 -0
- rag_wright/okf/links.py +190 -0
- rag_wright/okf/lint.py +105 -0
- rag_wright/ontology/__init__.py +6 -0
- rag_wright/ontology/_generated_template_meta.py +60 -0
- rag_wright/ontology/_generated_vocab.py +52 -0
- rag_wright/ontology/clause_template.py +964 -0
- rag_wright/ontology/codegen.py +84 -0
- rag_wright/ontology/compliance_bridge.ttl +186 -0
- rag_wright/ontology/contract_bridge.ttl +2685 -0
- rag_wright/ontology/contract_taxonomy.py +24 -0
- rag_wright/ontology/derive.py +58 -0
- rag_wright/ontology/loader.py +435 -0
- rag_wright/ontology/packs/ftc_16cfr255.ttl +29 -0
- rag_wright/ontology/registry.py +87 -0
- rag_wright/ontology/template_introspect.py +100 -0
- rag_wright/py.typed +0 -0
- rag_wright/reference/__init__.py +2 -0
- rag_wright/reference/compliance.py +41 -0
- rag_wright/reference/contract_seam.py +123 -0
- rag_wright/skills/__init__.py +7 -0
- rag_wright/skills/claim_extraction/SKILL.md +47 -0
- rag_wright/skills/claim_extraction/__init__.py +1 -0
- rag_wright/skills/claim_extraction/template.py +50 -0
- rag_wright/skills/compliance_judgment/SKILL.md +59 -0
- rag_wright/skills/corpus_ingest/SKILL.md +106 -0
- rag_wright/skills/extraction_semantic_judge/SKILL.md +51 -0
- rag_wright/skills/extraction_semantic_judge/__init__.py +1 -0
- rag_wright/skills/generation/SKILL.md +64 -0
- rag_wright/skills/generation/__init__.py +1 -0
- rag_wright/skills/generic_compliance_judgment/SKILL.md +58 -0
- rag_wright/skills/okf_navigate/SKILL.md +137 -0
- rag_wright/skills/requirement_extraction/SKILL.md +47 -0
- rag_wright/skills/requirement_extraction/__init__.py +1 -0
- rag_wright/skills/requirement_extraction/template.py +50 -0
- rag_wright/skills/rlm/SKILL.md +186 -0
- rag_wright/skills/rlm/__init__.py +31 -0
- rag_wright/skills/rlm/agent.py +292 -0
- rag_wright/skills/span_relevance_judgment/SKILL.md +67 -0
- rag_wright/skills/vision_to_text/SKILL.md +36 -0
- rag_wright/skills/vision_to_text/__init__.py +1 -0
- rag_wright/spans/__init__.py +1 -0
- rag_wright/spans/boundary.py +78 -0
- rag_wright/spans/clause_function_classifier.py +490 -0
- rag_wright/spans/clause_kg_extractor.py +337 -0
- rag_wright/spans/cuad_labels.py +81 -0
- rag_wright/spans/dim_classifier.py +158 -0
- rag_wright/spans/dim_fleet.json +411 -0
- rag_wright/spans/function_classifier.py +77 -0
- rag_wright/spans/function_families.py +62 -0
- rag_wright/spans/hybrid_classifier.py +103 -0
- rag_wright/spans/legalbert_classifier.py +83 -0
- rag_wright/spans/model_capabilities.py +107 -0
- rag_wright/spans/new_function_labels.py +111 -0
- rag_wright/spans/page_map.py +68 -0
- rag_wright/spans/property_extractor.py +365 -0
- rag_wright/spans/property_grounding.py +182 -0
- rag_wright/spans/reclassify.py +77 -0
- rag_wright/spans/scarce_function_labels.py +105 -0
- rag_wright/spans/segment.py +341 -0
- rag_wright/spans/semantic_judge.py +197 -0
- rag_wright/spans/symbolic_validation.py +131 -0
- rag_wright/spans/tag_clause_extractor.py +182 -0
- rag_wright/store/__init__.py +6 -0
- rag_wright/store/arcadedb.py +1135 -0
- rag_wright/store/chunk_text.py +66 -0
- rag_wright/store/seam.py +213 -0
- rag_wright/subgraphs/__init__.py +0 -0
- rag_wright/subgraphs/async_ingestion.py +204 -0
- rag_wright/subgraphs/compliance_check.py +1042 -0
- rag_wright/subgraphs/compliance_ingestion.py +306 -0
- rag_wright/subgraphs/contract_ingestion_pipeline.py +999 -0
- rag_wright/subgraphs/graph_extraction.py +102 -0
- rag_wright/subgraphs/intra_document_qa.py +328 -0
- rag_wright/subgraphs/observability.py +140 -0
- rag_wright/subgraphs/query_constraint_extraction.py +73 -0
- rag_wright/subgraphs/relational_qa.py +165 -0
- rag_wright/subgraphs/requirement_extraction.py +137 -0
- rag_wright/subgraphs/scaffold.py +65 -0
- rag_wright/subgraphs/semantic_chunking.py +183 -0
- rag_wright/subgraphs/typed_clause_extraction.py +172 -0
- rag_wright/subgraphs/typed_property_retrieval.py +278 -0
- rag_wright/util/__init__.py +1 -0
- rag_wright/util/concurrent.py +153 -0
- rag_wright/util/spacy_model.py +45 -0
- rag_wright-0.1.0.dist-info/METADATA +168 -0
- rag_wright-0.1.0.dist-info/RECORD +184 -0
- rag_wright-0.1.0.dist-info/WHEEL +4 -0
- rag_wright-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,153 @@
|
|
|
1
|
+
"""CC-3 (compliance §13 C-3), SKILL-SPLIT: subject ad -> `Claim[]`, split into a SKILL + a FUNCTION.
|
|
2
|
+
|
|
3
|
+
Per the capability-architecture principle (`function` = deterministic/no-model; a single LLM act = an authored
|
|
4
|
+
`agent_skill`; a workflow = `subgraph`), this is two capabilities:
|
|
5
|
+
|
|
6
|
+
- **`claim_extraction` (agent_skill)** -- the claim-extraction METHOD, authored as `skills/claim_extraction/`
|
|
7
|
+
(SKILL.md + the `template.py` schema asset: `ExtractedAd`/`ExtractedClaim`). One docling-graph call (`direct`,
|
|
8
|
+
ads are short) fills the template. `extract_ad` is its runtime; `extract_fn` is injected for hermetic tests.
|
|
9
|
+
- **`claim_adaptation` (function)** -- `to_claims`: DETERMINISTIC, no model. Maps the raw `ExtractedAd` to the
|
|
10
|
+
closed CC-1 `Claim` vocab (off-vocab claim_type kept-but-AMBIGUOUS -- a checkable assertion is never dropped),
|
|
11
|
+
attaches the span provenance + content-hash `claim_id`.
|
|
12
|
+
|
|
13
|
+
`claim_extraction(...)` composes them (skill -> function) for callers. The schema lives with the skill (an
|
|
14
|
+
Agent-Skill asset), re-exported here for consumers/tests.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
from typing import Any, Callable
|
|
20
|
+
|
|
21
|
+
from rag_wright.capabilities.dg_extraction import aextract_parties, extract_parties
|
|
22
|
+
from rag_wright.contracts.compliance import Claim, ClaimType
|
|
23
|
+
from rag_wright.contracts.provenance import ConfidenceTag
|
|
24
|
+
from rag_wright.skills.claim_extraction.template import ExtractedAd, ExtractedClaim # the skill's schema asset
|
|
25
|
+
|
|
26
|
+
__all__ = ["ExtractedAd", "ExtractedClaim", "extract_ad", "aextract_ad", "to_claims", "claim_extraction",
|
|
27
|
+
"aclaim_extraction", "register_claim_extraction", "register_claim_adaptation"]
|
|
28
|
+
|
|
29
|
+
_CLAIM_TYPES = {c.value for c in ClaimType}
|
|
30
|
+
# off-vocab fallback: keep the claim (never drop a checkable assertion) but mark it AMBIGUOUS. EFFICACY is the
|
|
31
|
+
# broadest "the product works" type; the AMBIGUOUS tag is what actually signals the uncertainty downstream.
|
|
32
|
+
_FALLBACK_CLAIM_TYPE = ClaimType.EFFICACY
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _coerce_claim_type(raw: str) -> tuple[ClaimType, bool]:
|
|
36
|
+
"""(ClaimType, ambiguous?) -- map the model's string to the closed vocab; an off-vocab value keeps the
|
|
37
|
+
claim under the fallback type and flags AMBIGUOUS (never drop a real assertion)."""
|
|
38
|
+
value = (raw or "").strip().lower()
|
|
39
|
+
if value in _CLAIM_TYPES:
|
|
40
|
+
return ClaimType(value), False
|
|
41
|
+
return _FALLBACK_CLAIM_TYPE, True
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def _clean(v: str) -> str | None:
|
|
45
|
+
v = (v or "").strip()
|
|
46
|
+
return v or None
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def to_claims(extracted: ExtractedAd, *, source_doc: str) -> list[Claim]:
|
|
50
|
+
"""`claim_adaptation` (FUNCTION -- deterministic, no model): adapt an extracted ad to validated `Claim`s.
|
|
51
|
+
Off-vocab claim_type -> fallback + AMBIGUOUS; blank assertion skipped. `claim_id` uses the content-hash
|
|
52
|
+
scheme; the (source_doc, assertion) is the span cite."""
|
|
53
|
+
from rag_wright.contracts.compliance import Constraint
|
|
54
|
+
|
|
55
|
+
out: list[Claim] = []
|
|
56
|
+
for index, item in enumerate(extracted.claims):
|
|
57
|
+
text = (item.assertion_text or "").strip()
|
|
58
|
+
if not text:
|
|
59
|
+
continue
|
|
60
|
+
claim_type, ambiguous = _coerce_claim_type(item.claim_type)
|
|
61
|
+
actor = _clean(item.actor)
|
|
62
|
+
# DEON-8: surface the ROLE actor as a dimension-agnostic Constraint on `.scope` (mirroring the generic
|
|
63
|
+
# `to_facts`), so the obligation actor gate (DEON-7) works on the ad path too. The typed `.actor` is kept.
|
|
64
|
+
scope = [Constraint(dimension="actor", value=actor.lower())] if actor else []
|
|
65
|
+
out.append(Claim(
|
|
66
|
+
fact_id=Claim.make_id(source_doc, index, text),
|
|
67
|
+
source_doc=source_doc,
|
|
68
|
+
claim_type=claim_type,
|
|
69
|
+
assertion_text=text,
|
|
70
|
+
scope=scope,
|
|
71
|
+
actor=actor,
|
|
72
|
+
subject_product=_clean(item.subject_product),
|
|
73
|
+
quantitative_value=_clean(item.quantitative_value),
|
|
74
|
+
disclosures_present=[d.strip() for d in item.disclosures_present if d and d.strip()],
|
|
75
|
+
evidence_referenced=bool(item.evidence_referenced),
|
|
76
|
+
medium=_clean(item.medium),
|
|
77
|
+
confidence=ConfidenceTag.AMBIGUOUS if ambiguous else ConfidenceTag.EXTRACTED,
|
|
78
|
+
))
|
|
79
|
+
return out
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
ExtractFn = Callable[..., Any] # (text, model, *, template, **kw) -> ExtractedAd | None
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def extract_ad(
|
|
86
|
+
text: str, *, model: Any, extract_fn: ExtractFn = extract_parties,
|
|
87
|
+
max_tokens: int = 1500, preamble_chars: int = 8000, extraction_contract: str = "direct",
|
|
88
|
+
) -> ExtractedAd | None:
|
|
89
|
+
"""The `claim_extraction` SKILL's runtime: run the docling-graph extraction (the skill's `template.py`
|
|
90
|
+
schema) through the model seam and return the raw `ExtractedAd` (or None). Ads are short -> `direct`."""
|
|
91
|
+
return extract_fn(text, model, template=ExtractedAd,
|
|
92
|
+
max_tokens=max_tokens, preamble_chars=preamble_chars,
|
|
93
|
+
extraction_contract=extraction_contract)
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def claim_extraction(
|
|
97
|
+
text: str, *, model: Any, source_doc: str,
|
|
98
|
+
extract_fn: ExtractFn = extract_parties, max_tokens: int = 1500, preamble_chars: int = 8000,
|
|
99
|
+
extraction_contract: str = "direct",
|
|
100
|
+
) -> list[Claim]:
|
|
101
|
+
"""Compose the SKILL (extract_ad) and the FUNCTION (to_claims): extract the raw ad, then adapt to `Claim`s.
|
|
102
|
+
Returns [] if extraction yields nothing."""
|
|
103
|
+
extracted = extract_ad(text, model=model, extract_fn=extract_fn, max_tokens=max_tokens,
|
|
104
|
+
preamble_chars=preamble_chars, extraction_contract=extraction_contract)
|
|
105
|
+
if extracted is None:
|
|
106
|
+
return []
|
|
107
|
+
return to_claims(extracted, source_doc=source_doc)
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
async def aextract_ad(
|
|
111
|
+
text: str, *, model: Any, aextract_fn: Any = aextract_parties,
|
|
112
|
+
max_tokens: int = 1500, preamble_chars: int = 8000, extraction_contract: str = "direct",
|
|
113
|
+
) -> ExtractedAd | None:
|
|
114
|
+
"""ASYNC-C1 (ADR-0057): the async twin of `extract_ad` -- the claim-extraction docling-graph act on the async
|
|
115
|
+
seam (`aextract_parties`, true wall-clock deadline via the injected client). `aextract_fn` injected for tests."""
|
|
116
|
+
return await aextract_fn(text, model, template=ExtractedAd,
|
|
117
|
+
max_tokens=max_tokens, preamble_chars=preamble_chars,
|
|
118
|
+
extraction_contract=extraction_contract)
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
async def aclaim_extraction(
|
|
122
|
+
text: str, *, model: Any, source_doc: str, aextract_fn: Any = aextract_parties,
|
|
123
|
+
max_tokens: int = 1500, preamble_chars: int = 8000, extraction_contract: str = "direct",
|
|
124
|
+
) -> list[Claim]:
|
|
125
|
+
"""ASYNC-C1 (ADR-0057): the async twin of `claim_extraction` -- await the async extraction ACT, then the
|
|
126
|
+
deterministic `to_claims` adaptation. Same contract: [] if extraction yields nothing."""
|
|
127
|
+
extracted = await aextract_ad(text, model=model, aextract_fn=aextract_fn, max_tokens=max_tokens,
|
|
128
|
+
preamble_chars=preamble_chars, extraction_contract=extraction_contract)
|
|
129
|
+
if extracted is None:
|
|
130
|
+
return []
|
|
131
|
+
return to_claims(extracted, source_doc=source_doc)
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def register_claim_extraction(registry) -> None:
|
|
135
|
+
"""Register `claim_extraction` as an AGENT_SKILL (CC-3): a single docling-graph LLM extraction act, authored
|
|
136
|
+
as `skills/claim_extraction/` (SKILL.md + the template.py schema). Typed output = `ExtractedAd`."""
|
|
137
|
+
registry.register(
|
|
138
|
+
"claim_extraction",
|
|
139
|
+
contract=ExtractedAd,
|
|
140
|
+
kind="agent_skill",
|
|
141
|
+
display_name="Claim extraction (subject ad -> checkable claims; authored skill)",
|
|
142
|
+
)
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
def register_claim_adaptation(registry) -> None:
|
|
146
|
+
"""Register `claim_adaptation` (FUNCTION -- deterministic): the skill's raw `ExtractedAd` -> validated
|
|
147
|
+
`Claim[]` (closed vocab, off-vocab kept-but-AMBIGUOUS, span provenance, content-hash id)."""
|
|
148
|
+
registry.register(
|
|
149
|
+
"claim_adaptation",
|
|
150
|
+
contract=Claim,
|
|
151
|
+
kind="function",
|
|
152
|
+
display_name="Claim adaptation (extracted ad -> validated Claims)",
|
|
153
|
+
)
|
|
@@ -0,0 +1,117 @@
|
|
|
1
|
+
"""ADR-0044: the `IS_EXCEPTION_TO` derived carve-out relationship (exception clause -> the Cap clause it excepts).
|
|
2
|
+
|
|
3
|
+
A single contract's liability structure is usually stored as TWO disconnected clauses -- a `Cap On Liability`
|
|
4
|
+
clause (often without its carve-outs) and a separate, often property-less `Uncapped Liability` clause (the
|
|
5
|
+
carve-out, e.g. "uncapped for negligence") -- with NO relationship between them. So a query like "how is
|
|
6
|
+
liability capped, and under what conditions?" gets two contradictory-looking fragments and the generator
|
|
7
|
+
abstains, instead of "capped at X, EXCEPT uncapped for negligence."
|
|
8
|
+
|
|
9
|
+
This capability adds the missing link WITHOUT re-ingesting: a pure pass over the already-populated clauses (a derived-relationship
|
|
10
|
+
linking pass over the existing KG) that writes `IsExceptionTo` edges. The signal is SYMBOLIC co-occurrence +
|
|
11
|
+
POSITIONAL PROXIMITY: within a contract, an `Uncapped` clause whose operative span is within `window` characters
|
|
12
|
+
of a `Cap` clause's span (~ the same liability section) is that cap's carve-out. Distant co-occurrence is NOT
|
|
13
|
+
linked (probably an unrelated standalone uncapped clause). The link is **INFERRED** (a reasoned inference, not an
|
|
14
|
+
extracted fact, FR-S.4): surfaced at query time and human-validatable, never a silent hard claim.
|
|
15
|
+
|
|
16
|
+
Scope (ADR-0044): the edge type is general; only the cap<->uncapped pair is populated for now.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
from __future__ import annotations
|
|
20
|
+
|
|
21
|
+
from collections import defaultdict
|
|
22
|
+
from typing import Any
|
|
23
|
+
|
|
24
|
+
from pydantic import BaseModel
|
|
25
|
+
|
|
26
|
+
from rag_wright.capabilities.registry import CapabilityRegistry
|
|
27
|
+
from rag_wright.contracts.provenance import ConfidenceTag
|
|
28
|
+
|
|
29
|
+
CAP_FUNCTION = "Cap On Liability"
|
|
30
|
+
EXCEPTION_FUNCTION = "Uncapped Liability"
|
|
31
|
+
DEFAULT_PROXIMITY_WINDOW = 3000 # chars between the two spans' ranges (~ one liability section); tunable
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
class ClauseExceptionLink(BaseModel):
|
|
35
|
+
"""One `IsExceptionTo` edge: the `exception_clause_id` (an Uncapped clause) is a carve-out/exception to the
|
|
36
|
+
`cap_clause_id` (a Cap clause) in `contract_id`. Confidence INFERRED -- a derived, reasoned link (FR-S.4)."""
|
|
37
|
+
|
|
38
|
+
exception_clause_id: str
|
|
39
|
+
cap_clause_id: str
|
|
40
|
+
contract_id: str
|
|
41
|
+
confidence: ConfidenceTag = ConfidenceTag.INFERRED
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
class ClauseExceptionLinkResult(BaseModel):
|
|
45
|
+
"""The capability's output: the derived `IsExceptionTo` links + the honest unlinked count."""
|
|
46
|
+
|
|
47
|
+
links: list[ClauseExceptionLink]
|
|
48
|
+
unlinked_exceptions: int # Uncapped clauses with no in-window Cap clause (skipped, not linked)
|
|
49
|
+
contracts_processed: int
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def _gap(a: dict, b: dict) -> int:
|
|
53
|
+
"""Character gap between two spans' [doc_start, doc_end] ranges (0 if they overlap)."""
|
|
54
|
+
a0, a1 = int(a.get("doc_start") or 0), int(a.get("doc_end") or 0)
|
|
55
|
+
b0, b1 = int(b.get("doc_start") or 0), int(b.get("doc_end") or 0)
|
|
56
|
+
if a1 < b0:
|
|
57
|
+
return b0 - a1
|
|
58
|
+
if b1 < a0:
|
|
59
|
+
return a0 - b1
|
|
60
|
+
return 0
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def derive_exception_links(
|
|
64
|
+
positions: list[dict], *, window: int = DEFAULT_PROXIMITY_WINDOW,
|
|
65
|
+
cap_function: str = CAP_FUNCTION, exception_function: str = EXCEPTION_FUNCTION,
|
|
66
|
+
) -> ClauseExceptionLinkResult:
|
|
67
|
+
"""Pure: within each contract, link each `exception_function` clause to the NEAREST `cap_function` clause
|
|
68
|
+
whose operative span is within `window` chars (proximity ~ same section) -> an INFERRED `IsExceptionTo` link.
|
|
69
|
+
A distant or cap-less exception clause is counted `unlinked`, never linked to a far cap (that would be a
|
|
70
|
+
false carve-out). One link per (exception, cap) pair. `positions`: rows with {clause_id, function,
|
|
71
|
+
contract_id, doc_start, doc_end}."""
|
|
72
|
+
by_contract: dict[str, dict[str, list[dict]]] = defaultdict(lambda: {"cap": [], "exc": []})
|
|
73
|
+
for p in positions:
|
|
74
|
+
bucket = by_contract[p["contract_id"]]
|
|
75
|
+
if p["function"] == cap_function:
|
|
76
|
+
bucket["cap"].append(p)
|
|
77
|
+
elif p["function"] == exception_function:
|
|
78
|
+
bucket["exc"].append(p)
|
|
79
|
+
|
|
80
|
+
links: list[ClauseExceptionLink] = []
|
|
81
|
+
unlinked = 0
|
|
82
|
+
for contract_id, bucket in by_contract.items():
|
|
83
|
+
caps = bucket["cap"]
|
|
84
|
+
if not caps: # uncapped clauses but no cap clause in this contract -> nothing to except
|
|
85
|
+
unlinked += len(bucket["exc"])
|
|
86
|
+
continue
|
|
87
|
+
for exc in bucket["exc"]:
|
|
88
|
+
nearest = min(caps, key=lambda cap: _gap(exc, cap))
|
|
89
|
+
if _gap(exc, nearest) <= window:
|
|
90
|
+
links.append(ClauseExceptionLink(
|
|
91
|
+
exception_clause_id=exc["clause_id"], cap_clause_id=nearest["clause_id"],
|
|
92
|
+
contract_id=contract_id))
|
|
93
|
+
else:
|
|
94
|
+
unlinked += 1 # co-occurs but distant -> probably unrelated; do not link (no false carve-out)
|
|
95
|
+
|
|
96
|
+
return ClauseExceptionLinkResult(
|
|
97
|
+
links=links, unlinked_exceptions=unlinked, contracts_processed=len(by_contract))
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def clause_exception_linking(store: Any, *, window: int = DEFAULT_PROXIMITY_WINDOW) -> ClauseExceptionLinkResult:
|
|
101
|
+
"""The registered capability (ADR-0044): read the Cap + Uncapped clause positions, derive the proximity-based
|
|
102
|
+
`IsExceptionTo` links, write them (idempotent, clears the layer first), and return the result. No re-ingest --
|
|
103
|
+
a derived-relationship pass over the existing KG."""
|
|
104
|
+
positions = store.clause_positions([CAP_FUNCTION, EXCEPTION_FUNCTION])
|
|
105
|
+
result = derive_exception_links(positions, window=window)
|
|
106
|
+
store.write_clause_exception_links(result.links)
|
|
107
|
+
return result
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def register_clause_exception_linking(registry: CapabilityRegistry) -> None:
|
|
111
|
+
"""ADR-0044: register `clause_exception_linking` (function; the cap<->uncapped carve-out relationship)."""
|
|
112
|
+
registry.register(
|
|
113
|
+
"clause_exception_linking",
|
|
114
|
+
contract=ClauseExceptionLinkResult,
|
|
115
|
+
kind="function",
|
|
116
|
+
display_name="Clause exception linking (IsExceptionTo carve-out edges over the contract KG)",
|
|
117
|
+
)
|
|
@@ -0,0 +1,322 @@
|
|
|
1
|
+
"""CC-4 (compliance §13.2), SKILL-SPLIT: the judgment node, split into a SKILL + a deterministic FUNCTION.
|
|
2
|
+
|
|
3
|
+
Per the capability-architecture principle (a `function` is deterministic and takes no model; a single LLM act is
|
|
4
|
+
an authored `agent_skill`; a workflow is a `subgraph`), the judgment is two capabilities:
|
|
5
|
+
|
|
6
|
+
- **`compliance_judgment` (agent_skill)** -- the LLM judgment METHOD, authored as `skills/compliance_judgment/
|
|
7
|
+
SKILL.md` and applied through the model seam (product = Granite, ADR-0039). Given one claim + one requirement
|
|
8
|
+
and ONLY the ad text, it returns a raw `JudgeVerdict` (verdict / rationale / confidence). `build_compliance_
|
|
9
|
+
judge_fn` is its runtime; `structured_factory` is injected for hermetic tests.
|
|
10
|
+
- **`compliance_finding_assembly` (function)** -- `assemble_finding`: DETERMINISTIC, no model. Maps the raw
|
|
11
|
+
verdict to the closed vocab (unreadable/missing -> needs_review, the conservative default), attaches the
|
|
12
|
+
BOTH-SIDED citation FROM THE INPUTS (the model never authors a citation), and returns the `ComplianceFinding`.
|
|
13
|
+
|
|
14
|
+
`compliance_judgment(...)` composes them (skill -> function) for callers; `judge_pairs` runs the composition
|
|
15
|
+
concurrently (async + semaphore + LLM-CALL-TIMEOUT). The applying capability owns the guarantees the skill does
|
|
16
|
+
not (verdict vocab, conservative default, citation) -- the SKILL.md teaches only the reading.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
from __future__ import annotations
|
|
20
|
+
|
|
21
|
+
import asyncio
|
|
22
|
+
import os
|
|
23
|
+
from pathlib import Path
|
|
24
|
+
from typing import Awaitable, Callable, Optional
|
|
25
|
+
|
|
26
|
+
from pydantic import BaseModel
|
|
27
|
+
|
|
28
|
+
from rag_wright.contracts.compliance import (
|
|
29
|
+
CheckableFact,
|
|
30
|
+
Claim,
|
|
31
|
+
ComplianceFinding,
|
|
32
|
+
DeonticType,
|
|
33
|
+
Requirement,
|
|
34
|
+
Verdict,
|
|
35
|
+
)
|
|
36
|
+
from rag_wright.models.seam import build_structured
|
|
37
|
+
from rag_wright.util.concurrent import map_concurrent
|
|
38
|
+
|
|
39
|
+
_VERDICTS = {v.value for v in Verdict}
|
|
40
|
+
_SKILL_PATH = Path(__file__).parents[1] / "skills" / "compliance_judgment" / "SKILL.md" # advertising method
|
|
41
|
+
# COMP-VERDICT-GENERIC: the DOMAIN-AGNOSTIC judgment method -- the base the advertising SKILL specializes.
|
|
42
|
+
_GENERIC_SKILL_PATH = Path(__file__).parents[1] / "skills" / "generic_compliance_judgment" / "SKILL.md"
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
class JudgeVerdict(BaseModel):
|
|
46
|
+
"""The raw structured output of one judge call (the `compliance_judgment` SKILL's typed output). `verdict`
|
|
47
|
+
is a loose string mapped to the closed `Verdict` by the FUNCTION (an unreadable value -> needs_review);
|
|
48
|
+
citations are added from the inputs by the function, never authored here."""
|
|
49
|
+
|
|
50
|
+
verdict: str
|
|
51
|
+
rationale: str = ""
|
|
52
|
+
confidence: float = 0.0
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
# judge_fn: (subject_fact, requirement) -> JudgeVerdict, or None if the judge could not rule (-> conservative
|
|
56
|
+
# default). Accepts any `CheckableFact` (the advertising `Claim` is one).
|
|
57
|
+
JudgeFn = Callable[[CheckableFact, Requirement], Optional[JudgeVerdict]]
|
|
58
|
+
# ASYNC-C1 (ADR-0057): the async judge seam -- same signature, awaitable result (the model call gets a true
|
|
59
|
+
# wall-clock deadline via build_structured's .ainvoke).
|
|
60
|
+
AJudgeFn = Callable[[CheckableFact, Requirement], Awaitable[Optional[JudgeVerdict]]]
|
|
61
|
+
|
|
62
|
+
# The per-call appendix bound onto the SKILL method (the static method teaches the reading; the specific
|
|
63
|
+
# requirement + subject are appended at call time, the okf_navigate `_with_question` pattern).
|
|
64
|
+
# COMP-VERDICT-GENERIC: split into a domain-agnostic BASE tail (requirement + subject assertion -- works for ANY
|
|
65
|
+
# CheckableFact / domain) + an ADVERTISING enrichment line (claim_type / disclosures / evidence). The generic
|
|
66
|
+
# judge uses only the base; the advertising judge appends the enrichment (behavior unchanged).
|
|
67
|
+
_BASE_PROMPT_TAIL = (
|
|
68
|
+
"\n\nREQUIREMENT ({deontic}, {citation}; applies to {actor}):\n{requirement_text}\n\n"
|
|
69
|
+
"SUBJECT:\n{assertion}"
|
|
70
|
+
)
|
|
71
|
+
_AD_ENRICHMENT = "\n\nCLAIM SIGNALS: type={claim_type}; disclosures_present={disclosures}; evidence_referenced={evidence}"
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def _skill_body(path: Path) -> str:
|
|
75
|
+
"""A SKILL.md body with its YAML frontmatter stripped -- the judge's system/method prompt."""
|
|
76
|
+
text = path.read_text(encoding="utf-8")
|
|
77
|
+
if text.startswith("---"):
|
|
78
|
+
marker = text.find("\n---", 3)
|
|
79
|
+
if marker != -1:
|
|
80
|
+
text = text[marker + 4 :]
|
|
81
|
+
return text.strip()
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def judgment_method() -> str:
|
|
85
|
+
"""The ADVERTISING compliance-judgment method (skills/compliance_judgment/SKILL.md, FTC doctrine)."""
|
|
86
|
+
return _skill_body(_SKILL_PATH)
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def generic_judgment_method() -> str:
|
|
90
|
+
"""COMP-VERDICT-GENERIC: the DOMAIN-AGNOSTIC judgment method (skills/generic_compliance_judgment/SKILL.md) --
|
|
91
|
+
no advertising doctrine, so the generic judge reasons about "the subject" in any domain."""
|
|
92
|
+
return _skill_body(_GENERIC_SKILL_PATH)
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def _base_tail(fact: CheckableFact, requirement: Requirement) -> str:
|
|
96
|
+
"""The domain-agnostic judge appendix: the requirement + the subject assertion. Works for ANY CheckableFact.
|
|
97
|
+
issue 0044: the DEON-8 document signals (disclosure/evidence union) are appended here from `document_signals`
|
|
98
|
+
-- so the JUDGE still sees them (prompt unchanged), while `assertion_text` (and thus the citation) stays pure
|
|
99
|
+
document text. `document_signals` is '' for a plain fact, so the generic path is unaffected."""
|
|
100
|
+
return _BASE_PROMPT_TAIL.format(
|
|
101
|
+
deontic=requirement.deontic_type.value, citation=requirement.citation, actor=requirement.actor,
|
|
102
|
+
requirement_text=requirement.requirement_text, assertion=fact.assertion_text) \
|
|
103
|
+
+ (getattr(fact, "document_signals", "") or "")
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
# DEON-1 (issue 0012): the neuro-symbolic deontic framing -- the KG's deontic_type tells the judge HOW to reason.
|
|
107
|
+
# An OBLIGATION is breached by ABSENCE: the subject text above is the document's relevant content (DEON judges an
|
|
108
|
+
# obligation ONCE over the document, not per sentence), so the judge must DECIDE yes/no, not hedge to "unclear"
|
|
109
|
+
# just because the required element is missing -- absence IS the violation.
|
|
110
|
+
_OBLIGATION_FRAMING = (
|
|
111
|
+
"\n\nDEONTIC FRAMING -- this requirement is an OBLIGATION (it must be SATISFIED): the subject text above is the "
|
|
112
|
+
"document's relevant content, checked as a whole. Decide definitively: if the required element (e.g. the "
|
|
113
|
+
"disclosure/action) IS present, COMPLIANT; if it is ABSENT from this text, that is a VIOLATION -- a missing "
|
|
114
|
+
"required element is itself the breach. Do NOT answer needs_review merely because the element is absent.")
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def _deontic_framing(requirement: Requirement) -> str:
|
|
118
|
+
return _OBLIGATION_FRAMING if requirement.deontic_type is DeonticType.OBLIGATION else ""
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
# DEON-9 (issue 0012): permission-as-defense -- the same-source PERMISSIONS linked to this O/F rule (carve-outs /
|
|
122
|
+
# safe-harbors). The judge reasons over the rule AND its exceptions: a subject that falls within a permitted
|
|
123
|
+
# carve-out is NOT a violation. Only apply an exception that genuinely fits the subject (never a blanket excuse).
|
|
124
|
+
_DEFENSE_FRAMING = (
|
|
125
|
+
"\n\nEXCEPTIONS / DEFENSES (permitted carve-outs that may EXCUSE this rule):\n{defenses}\n"
|
|
126
|
+
"If the subject falls within one of these permitted exceptions, the rule is NOT violated -- rule COMPLIANT, "
|
|
127
|
+
"and say which exception applies. Only apply an exception that genuinely fits the subject.")
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def _defense_framing(requirement: Requirement) -> str:
|
|
131
|
+
defenses = getattr(requirement, "defenses", None) or []
|
|
132
|
+
if not defenses:
|
|
133
|
+
return ""
|
|
134
|
+
return _DEFENSE_FRAMING.format(defenses="\n".join(f"- {d}" for d in defenses))
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
def _ad_signals(fact: CheckableFact) -> str:
|
|
138
|
+
"""DEON-8: the ADVERTISING claim-signals line -- rendered only for a typed `Claim`. An obligation judged ONCE
|
|
139
|
+
(DEON-1) hands the judge a plain `CheckableFact` EVIDENCE BUNDLE with no claim_type; return '' so the ad judge
|
|
140
|
+
TOLERATES it (the bundle's disclosure signals ride in its text via DEON-8 Option 1) instead of crashing on a
|
|
141
|
+
missing attribute."""
|
|
142
|
+
ct = getattr(fact, "claim_type", None)
|
|
143
|
+
if ct is None:
|
|
144
|
+
return ""
|
|
145
|
+
return _AD_ENRICHMENT.format(
|
|
146
|
+
claim_type=ct.value, disclosures=fact.disclosures_present or "none", evidence=fact.evidence_referenced)
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
def build_generic_judge_fn(model_id: str, *, structured_factory=build_structured) -> JudgeFn:
|
|
150
|
+
"""COMP-VERDICT-GENERIC: the DOMAIN-AGNOSTIC judge -- rules a `(subject_fact, requirement)` pair on TEXT alone
|
|
151
|
+
using the GENERIC judgment method (no advertising doctrine; reasons about "the subject" in any domain), so it
|
|
152
|
+
gives a verdict in ANY compliance domain. Same conservative default as the advertising judge; the method +
|
|
153
|
+
the appendix (base tail only, no claim signals) differ."""
|
|
154
|
+
method = generic_judgment_method()
|
|
155
|
+
|
|
156
|
+
def judge(fact: CheckableFact, requirement: Requirement) -> Optional[JudgeVerdict]:
|
|
157
|
+
return structured_factory(model_id, JudgeVerdict).invoke(
|
|
158
|
+
method + _base_tail(fact, requirement) + _defense_framing(requirement) + _deontic_framing(requirement))
|
|
159
|
+
|
|
160
|
+
return judge
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
def build_ageneric_judge_fn(model_id: str, *, structured_factory=build_structured) -> AJudgeFn:
|
|
164
|
+
"""ASYNC-C1 (ADR-0057): the async twin of `build_generic_judge_fn` -- the DOMAIN-AGNOSTIC judge on the async
|
|
165
|
+
structured seam (`.ainvoke`, a true wall-clock deadline on the model call). Same method + base tail."""
|
|
166
|
+
method = generic_judgment_method()
|
|
167
|
+
|
|
168
|
+
async def judge(fact: CheckableFact, requirement: Requirement) -> Optional[JudgeVerdict]:
|
|
169
|
+
return await structured_factory(model_id, JudgeVerdict).ainvoke(
|
|
170
|
+
method + _base_tail(fact, requirement) + _defense_framing(requirement) + _deontic_framing(requirement))
|
|
171
|
+
|
|
172
|
+
return judge
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
def build_compliance_judge_fn(model_id: str, *, structured_factory=build_structured) -> JudgeFn:
|
|
176
|
+
"""The advertising `compliance_judgment` SKILL runtime: a Granite-backed judge `JudgeFn` through the model seam
|
|
177
|
+
(product = vLLM-Granite; ADR-0039). Base tail (requirement + subject) + the ADVERTISING claim signals
|
|
178
|
+
(claim_type / disclosures / evidence). Behavior unchanged from before the CheckableFact split.
|
|
179
|
+
`structured_factory` is injected for tests."""
|
|
180
|
+
method = judgment_method()
|
|
181
|
+
|
|
182
|
+
def judge(claim: Claim, requirement: Requirement) -> Optional[JudgeVerdict]:
|
|
183
|
+
prompt = (method + _base_tail(claim, requirement) + _ad_signals(claim)
|
|
184
|
+
+ _defense_framing(requirement) + _deontic_framing(requirement))
|
|
185
|
+
return structured_factory(model_id, JudgeVerdict).invoke(prompt)
|
|
186
|
+
|
|
187
|
+
return judge
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
def build_acompliance_judge_fn(model_id: str, *, structured_factory=build_structured) -> AJudgeFn:
|
|
191
|
+
"""ASYNC-C1 (ADR-0057): the async twin of `build_compliance_judge_fn` -- the advertising judge on the async
|
|
192
|
+
structured seam (`.ainvoke`, a true wall-clock deadline). Same method + base tail + claim signals."""
|
|
193
|
+
method = judgment_method()
|
|
194
|
+
|
|
195
|
+
async def judge(claim: Claim, requirement: Requirement) -> Optional[JudgeVerdict]:
|
|
196
|
+
prompt = (method + _base_tail(claim, requirement) + _ad_signals(claim)
|
|
197
|
+
+ _defense_framing(requirement) + _deontic_framing(requirement))
|
|
198
|
+
return await structured_factory(model_id, JudgeVerdict).ainvoke(prompt)
|
|
199
|
+
|
|
200
|
+
return judge
|
|
201
|
+
|
|
202
|
+
|
|
203
|
+
def _to_verdict(raw: str) -> Verdict:
|
|
204
|
+
"""Map the LLM's verdict string to the closed vocab; an unreadable value -> NEEDS_REVIEW (conservative)."""
|
|
205
|
+
value = (raw or "").strip().lower().replace("-", "_").replace(" ", "_")
|
|
206
|
+
return Verdict(value) if value in _VERDICTS else Verdict.NEEDS_REVIEW
|
|
207
|
+
|
|
208
|
+
|
|
209
|
+
def assemble_finding(
|
|
210
|
+
claim: Claim, requirement: Requirement, ruling: Optional[JudgeVerdict]
|
|
211
|
+
) -> ComplianceFinding:
|
|
212
|
+
"""`compliance_finding_assembly` (FUNCTION -- deterministic, no model): map the skill's raw `JudgeVerdict`
|
|
213
|
+
(or None) to a `ComplianceFinding`. None or an off-vocab verdict conservatively defaults to `needs_review`;
|
|
214
|
+
the BOTH-SIDED citation is taken from the INPUTS (the model never authors a citation)."""
|
|
215
|
+
if ruling is None:
|
|
216
|
+
verdict, rationale, confidence = Verdict.NEEDS_REVIEW, "judge did not return a ruling", 0.0
|
|
217
|
+
else:
|
|
218
|
+
verdict = _to_verdict(ruling.verdict)
|
|
219
|
+
rationale = ruling.rationale
|
|
220
|
+
confidence = min(1.0, max(0.0, ruling.confidence))
|
|
221
|
+
# UNIFY-A / SEG-1: cite "doc {locator}: {assertion}" where the locator is the fact's structural path
|
|
222
|
+
# ("§ 4.2", "§ 4.2 ¶3", "§ 4.2 · bullet 2") when present, else "" -> the old "doc: assertion" (back-compat).
|
|
223
|
+
# Still input-authored (the model never writes the citation).
|
|
224
|
+
_loc = claim.locator()
|
|
225
|
+
_sec = f" {_loc}" if _loc else ""
|
|
226
|
+
return ComplianceFinding(
|
|
227
|
+
claim_id=claim.fact_id,
|
|
228
|
+
requirement_id=requirement.requirement_id,
|
|
229
|
+
verdict=verdict,
|
|
230
|
+
rationale=rationale,
|
|
231
|
+
citation_claim=f"{claim.source_doc}{_sec}: {claim.assertion_text}", # issue 0044: doc text only, no signals
|
|
232
|
+
citation_requirement=f"{requirement.citation} ({requirement.requirement_id}): {requirement.requirement_text}",
|
|
233
|
+
citation_claim_kind=getattr(claim, "citation_kind", "verbatim"), # issue 0044: verbatim vs assembled
|
|
234
|
+
confidence=confidence,
|
|
235
|
+
)
|
|
236
|
+
|
|
237
|
+
|
|
238
|
+
def compliance_judgment(claim: Claim, requirement: Requirement, *, judge_fn: JudgeFn) -> ComplianceFinding:
|
|
239
|
+
"""Compose the SKILL (LLM judgment) and the FUNCTION (deterministic assembly): run `judge_fn` then
|
|
240
|
+
`assemble_finding`. A None ruling conservatively defaults to `needs_review`. Citations come from the inputs."""
|
|
241
|
+
return assemble_finding(claim, requirement, judge_fn(claim, requirement))
|
|
242
|
+
|
|
243
|
+
|
|
244
|
+
_JUDGE_TIMEOUT_S = float(os.environ.get("RAG_JUDGE_TIMEOUT_S", "90")) # per-pair wall-clock bound (LLM-CALL-TIMEOUT)
|
|
245
|
+
|
|
246
|
+
|
|
247
|
+
def judge_pairs(
|
|
248
|
+
pairs: list[tuple[Claim, Requirement]], *, judge_fn: JudgeFn, max_concurrency: int = 8,
|
|
249
|
+
timeout_s: float | None = _JUDGE_TIMEOUT_S,
|
|
250
|
+
) -> list[ComplianceFinding]:
|
|
251
|
+
"""Run the skill->function composition over many `(claim, requirement)` pairs concurrently (async +
|
|
252
|
+
semaphore, per the parallel-LLM rule). Order is preserved. CC-6 drives this over a subject doc's claims x
|
|
253
|
+
their applicable requirements.
|
|
254
|
+
|
|
255
|
+
`timeout_s` (LLM-CALL-TIMEOUT) bounds each judgment with a hard wall-clock deadline so a stalled provider
|
|
256
|
+
response never hangs the batch; a timed-out pair (map_concurrent -> None) becomes a conservative
|
|
257
|
+
needs_review finding (the same conservative default as a judge that could not rule)."""
|
|
258
|
+
results = map_concurrent(
|
|
259
|
+
pairs, lambda pair: compliance_judgment(pair[0], pair[1], judge_fn=judge_fn),
|
|
260
|
+
max_concurrency=max_concurrency, timeout_s=timeout_s,
|
|
261
|
+
)
|
|
262
|
+
return [
|
|
263
|
+
result if result is not None
|
|
264
|
+
else assemble_finding(claim, requirement, None) # timed out -> conservative needs_review
|
|
265
|
+
for (claim, requirement), result in zip(pairs, results)
|
|
266
|
+
]
|
|
267
|
+
|
|
268
|
+
|
|
269
|
+
async def ajudge_pairs(
|
|
270
|
+
pairs: list[tuple[Claim, Requirement]], *, ajudge_fn: AJudgeFn, max_concurrency: int = 8,
|
|
271
|
+
timeout_s: float | None = _JUDGE_TIMEOUT_S, timeout_retries: int = 1,
|
|
272
|
+
) -> list[ComplianceFinding]:
|
|
273
|
+
"""ASYNC-C1 (ADR-0057): the async twin of `judge_pairs` -- run the async judge over many `(claim,
|
|
274
|
+
requirement)` pairs concurrently (asyncio.gather + Semaphore, the parallel-LLM rule), order preserved.
|
|
275
|
+
|
|
276
|
+
Matches `judge_pairs`/`map_concurrent_async` semantics exactly: each judgment is bounded by `timeout_s` (a
|
|
277
|
+
hard wall-clock deadline via `asyncio.timeout`); on the deadline the call is retried up to `timeout_retries`
|
|
278
|
+
times, then the pair becomes `None` -> a conservative needs_review finding. A NON-timeout judge error
|
|
279
|
+
propagates (the same as the sync path), so a genuine bug is never masked as needs_review."""
|
|
280
|
+
semaphore = asyncio.Semaphore(max_concurrency)
|
|
281
|
+
|
|
282
|
+
async def _rule(claim: Claim, requirement: Requirement) -> Optional[JudgeVerdict]:
|
|
283
|
+
if timeout_s is None:
|
|
284
|
+
return await ajudge_fn(claim, requirement) # no wall-clock bound (the seam still bounds each call)
|
|
285
|
+
for attempt in range(timeout_retries + 1):
|
|
286
|
+
try:
|
|
287
|
+
async with asyncio.timeout(timeout_s):
|
|
288
|
+
return await ajudge_fn(claim, requirement)
|
|
289
|
+
except (asyncio.TimeoutError, TimeoutError):
|
|
290
|
+
if attempt >= timeout_retries:
|
|
291
|
+
return None # give up -> conservative needs_review (a stalled provider never hangs the batch)
|
|
292
|
+
return None
|
|
293
|
+
|
|
294
|
+
async def _one(pair: tuple[Claim, Requirement]) -> ComplianceFinding:
|
|
295
|
+
claim, requirement = pair
|
|
296
|
+
async with semaphore: # backpressure
|
|
297
|
+
ruling = await _rule(claim, requirement)
|
|
298
|
+
return assemble_finding(claim, requirement, ruling)
|
|
299
|
+
|
|
300
|
+
return list(await asyncio.gather(*(_one(pair) for pair in pairs)))
|
|
301
|
+
|
|
302
|
+
|
|
303
|
+
def register_compliance_judgment(registry) -> None:
|
|
304
|
+
"""Register `compliance_judgment` as an AGENT_SKILL (CC-4): a single grounded LLM judgment act, authored as
|
|
305
|
+
`skills/compliance_judgment/SKILL.md` and applied via the seam. Typed output = `JudgeVerdict`."""
|
|
306
|
+
registry.register(
|
|
307
|
+
"compliance_judgment",
|
|
308
|
+
contract=JudgeVerdict,
|
|
309
|
+
kind="agent_skill",
|
|
310
|
+
display_name="Compliance judgment (claim x requirement -> verdict; authored skill)",
|
|
311
|
+
)
|
|
312
|
+
|
|
313
|
+
|
|
314
|
+
def register_compliance_finding_assembly(registry) -> None:
|
|
315
|
+
"""Register `compliance_finding_assembly` (FUNCTION -- deterministic): the skill's raw verdict + the inputs
|
|
316
|
+
-> a cited `ComplianceFinding` (conservative default, both-sided citation from the inputs)."""
|
|
317
|
+
registry.register(
|
|
318
|
+
"compliance_finding_assembly",
|
|
319
|
+
contract=ComplianceFinding,
|
|
320
|
+
kind="function",
|
|
321
|
+
display_name="Compliance finding assembly (verdict + inputs -> cited finding)",
|
|
322
|
+
)
|