rag-wright 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- rag_wright/__init__.py +13 -0
- rag_wright/api/__init__.py +33 -0
- rag_wright/api/config.py +59 -0
- rag_wright/api/discover.py +70 -0
- rag_wright/api/documents.py +39 -0
- rag_wright/api/ids.py +31 -0
- rag_wright/api/invoke.py +99 -0
- rag_wright/api/kg.py +61 -0
- rag_wright/api/mcp.py +94 -0
- rag_wright/api/usage.py +30 -0
- rag_wright/api/workspace.py +85 -0
- rag_wright/capabilities/__init__.py +8 -0
- rag_wright/capabilities/answer_generator.py +427 -0
- rag_wright/capabilities/ard.py +286 -0
- rag_wright/capabilities/assertion_extraction.py +79 -0
- rag_wright/capabilities/chunk_read.py +58 -0
- rag_wright/capabilities/chunk_write.py +163 -0
- rag_wright/capabilities/claim_extraction.py +153 -0
- rag_wright/capabilities/clause_exception_linking.py +117 -0
- rag_wright/capabilities/compliance_judgment.py +322 -0
- rag_wright/capabilities/compliance_store.py +87 -0
- rag_wright/capabilities/contract_kg_serve.py +156 -0
- rag_wright/capabilities/contract_kg_store.py +251 -0
- rag_wright/capabilities/dg_extraction.py +585 -0
- rag_wright/capabilities/disambiguation.py +163 -0
- rag_wright/capabilities/document_parse.py +87 -0
- rag_wright/capabilities/document_scope.py +49 -0
- rag_wright/capabilities/embedding.py +164 -0
- rag_wright/capabilities/embedding_profiles.py +43 -0
- rag_wright/capabilities/entity_resolution.py +154 -0
- rag_wright/capabilities/fusion.py +64 -0
- rag_wright/capabilities/graph_extraction.py +243 -0
- rag_wright/capabilities/graph_query.py +73 -0
- rag_wright/capabilities/graph_storage.py +111 -0
- rag_wright/capabilities/highlight_serve.py +142 -0
- rag_wright/capabilities/hybrid_search.py +65 -0
- rag_wright/capabilities/invoke.py +31 -0
- rag_wright/capabilities/jev_decision.py +38 -0
- rag_wright/capabilities/manifests.py +872 -0
- rag_wright/capabilities/okf_navigate.py +456 -0
- rag_wright/capabilities/parsing.py +286 -0
- rag_wright/capabilities/property_boosted_retrieval.py +125 -0
- rag_wright/capabilities/query_function_classifier.py +94 -0
- rag_wright/capabilities/query_understanding.py +109 -0
- rag_wright/capabilities/registry.py +262 -0
- rag_wright/capabilities/remote_encoders.py +94 -0
- rag_wright/capabilities/requirement_extraction.py +247 -0
- rag_wright/capabilities/reranking.py +123 -0
- rag_wright/capabilities/retrieval_core.py +126 -0
- rag_wright/capabilities/rlm_chunking.py +808 -0
- rag_wright/capabilities/rlm_synthesis.py +316 -0
- rag_wright/capabilities/scan_quality.py +136 -0
- rag_wright/capabilities/span_relevance_judgment.py +191 -0
- rag_wright/capabilities/vision_to_text.py +85 -0
- rag_wright/capabilities/vlm_ocr.py +85 -0
- rag_wright/contracts/__init__.py +6 -0
- rag_wright/contracts/chunk.py +79 -0
- rag_wright/contracts/compliance.py +303 -0
- rag_wright/contracts/contract_meta.py +27 -0
- rag_wright/contracts/extraction.py +130 -0
- rag_wright/contracts/function.py +167 -0
- rag_wright/contracts/function_routing.py +91 -0
- rag_wright/contracts/highlight.py +74 -0
- rag_wright/contracts/identifiers.py +153 -0
- rag_wright/contracts/jurisdiction.py +96 -0
- rag_wright/contracts/ontology.py +142 -0
- rag_wright/contracts/property.py +201 -0
- rag_wright/contracts/provenance.py +78 -0
- rag_wright/contracts/query_intent.py +53 -0
- rag_wright/contracts/span.py +76 -0
- rag_wright/contracts/value_match.py +84 -0
- rag_wright/corpus/__init__.py +0 -0
- rag_wright/corpus/canonicalize.py +116 -0
- rag_wright/corpus/cuad.py +153 -0
- rag_wright/corpus/cuad_ingestion.py +72 -0
- rag_wright/corpus/document_parser.py +299 -0
- rag_wright/corpus/edgar.py +231 -0
- rag_wright/corpus/gcs_ingestion.py +120 -0
- rag_wright/corpus/http.py +110 -0
- rag_wright/corpus/selection.py +152 -0
- rag_wright/mcp/__init__.py +11 -0
- rag_wright/mcp/compliance_server.py +299 -0
- rag_wright/mcp/intra_document_qa_server.py +170 -0
- rag_wright/mcp/relational_qa_server.py +171 -0
- rag_wright/mcp/session_store.py +64 -0
- rag_wright/mcp/typed_property_retrieval_server.py +191 -0
- rag_wright/models/__init__.py +8 -0
- rag_wright/models/profiles.py +331 -0
- rag_wright/models/seam.py +497 -0
- rag_wright/models/tag_structured.py +285 -0
- rag_wright/models/tracing.py +179 -0
- rag_wright/models/usage.py +102 -0
- rag_wright/okf/__init__.py +11 -0
- rag_wright/okf/compile.py +292 -0
- rag_wright/okf/document.py +47 -0
- rag_wright/okf/enrich.py +176 -0
- rag_wright/okf/links.py +190 -0
- rag_wright/okf/lint.py +105 -0
- rag_wright/ontology/__init__.py +6 -0
- rag_wright/ontology/_generated_template_meta.py +60 -0
- rag_wright/ontology/_generated_vocab.py +52 -0
- rag_wright/ontology/clause_template.py +964 -0
- rag_wright/ontology/codegen.py +84 -0
- rag_wright/ontology/compliance_bridge.ttl +186 -0
- rag_wright/ontology/contract_bridge.ttl +2685 -0
- rag_wright/ontology/contract_taxonomy.py +24 -0
- rag_wright/ontology/derive.py +58 -0
- rag_wright/ontology/loader.py +435 -0
- rag_wright/ontology/packs/ftc_16cfr255.ttl +29 -0
- rag_wright/ontology/registry.py +87 -0
- rag_wright/ontology/template_introspect.py +100 -0
- rag_wright/py.typed +0 -0
- rag_wright/reference/__init__.py +2 -0
- rag_wright/reference/compliance.py +41 -0
- rag_wright/reference/contract_seam.py +123 -0
- rag_wright/skills/__init__.py +7 -0
- rag_wright/skills/claim_extraction/SKILL.md +47 -0
- rag_wright/skills/claim_extraction/__init__.py +1 -0
- rag_wright/skills/claim_extraction/template.py +50 -0
- rag_wright/skills/compliance_judgment/SKILL.md +59 -0
- rag_wright/skills/corpus_ingest/SKILL.md +106 -0
- rag_wright/skills/extraction_semantic_judge/SKILL.md +51 -0
- rag_wright/skills/extraction_semantic_judge/__init__.py +1 -0
- rag_wright/skills/generation/SKILL.md +64 -0
- rag_wright/skills/generation/__init__.py +1 -0
- rag_wright/skills/generic_compliance_judgment/SKILL.md +58 -0
- rag_wright/skills/okf_navigate/SKILL.md +137 -0
- rag_wright/skills/requirement_extraction/SKILL.md +47 -0
- rag_wright/skills/requirement_extraction/__init__.py +1 -0
- rag_wright/skills/requirement_extraction/template.py +50 -0
- rag_wright/skills/rlm/SKILL.md +186 -0
- rag_wright/skills/rlm/__init__.py +31 -0
- rag_wright/skills/rlm/agent.py +292 -0
- rag_wright/skills/span_relevance_judgment/SKILL.md +67 -0
- rag_wright/skills/vision_to_text/SKILL.md +36 -0
- rag_wright/skills/vision_to_text/__init__.py +1 -0
- rag_wright/spans/__init__.py +1 -0
- rag_wright/spans/boundary.py +78 -0
- rag_wright/spans/clause_function_classifier.py +490 -0
- rag_wright/spans/clause_kg_extractor.py +337 -0
- rag_wright/spans/cuad_labels.py +81 -0
- rag_wright/spans/dim_classifier.py +158 -0
- rag_wright/spans/dim_fleet.json +411 -0
- rag_wright/spans/function_classifier.py +77 -0
- rag_wright/spans/function_families.py +62 -0
- rag_wright/spans/hybrid_classifier.py +103 -0
- rag_wright/spans/legalbert_classifier.py +83 -0
- rag_wright/spans/model_capabilities.py +107 -0
- rag_wright/spans/new_function_labels.py +111 -0
- rag_wright/spans/page_map.py +68 -0
- rag_wright/spans/property_extractor.py +365 -0
- rag_wright/spans/property_grounding.py +182 -0
- rag_wright/spans/reclassify.py +77 -0
- rag_wright/spans/scarce_function_labels.py +105 -0
- rag_wright/spans/segment.py +341 -0
- rag_wright/spans/semantic_judge.py +197 -0
- rag_wright/spans/symbolic_validation.py +131 -0
- rag_wright/spans/tag_clause_extractor.py +182 -0
- rag_wright/store/__init__.py +6 -0
- rag_wright/store/arcadedb.py +1135 -0
- rag_wright/store/chunk_text.py +66 -0
- rag_wright/store/seam.py +213 -0
- rag_wright/subgraphs/__init__.py +0 -0
- rag_wright/subgraphs/async_ingestion.py +204 -0
- rag_wright/subgraphs/compliance_check.py +1042 -0
- rag_wright/subgraphs/compliance_ingestion.py +306 -0
- rag_wright/subgraphs/contract_ingestion_pipeline.py +999 -0
- rag_wright/subgraphs/graph_extraction.py +102 -0
- rag_wright/subgraphs/intra_document_qa.py +328 -0
- rag_wright/subgraphs/observability.py +140 -0
- rag_wright/subgraphs/query_constraint_extraction.py +73 -0
- rag_wright/subgraphs/relational_qa.py +165 -0
- rag_wright/subgraphs/requirement_extraction.py +137 -0
- rag_wright/subgraphs/scaffold.py +65 -0
- rag_wright/subgraphs/semantic_chunking.py +183 -0
- rag_wright/subgraphs/typed_clause_extraction.py +172 -0
- rag_wright/subgraphs/typed_property_retrieval.py +278 -0
- rag_wright/util/__init__.py +1 -0
- rag_wright/util/concurrent.py +153 -0
- rag_wright/util/spacy_model.py +45 -0
- rag_wright-0.1.0.dist-info/METADATA +168 -0
- rag_wright-0.1.0.dist-info/RECORD +184 -0
- rag_wright-0.1.0.dist-info/WHEEL +4 -0
- rag_wright-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,262 @@
|
|
|
1
|
+
"""The capability registration seam (FR-S.5, contract use).
|
|
2
|
+
|
|
3
|
+
Two registrations from one key, the capability's FR-C / FR-I / FR-Q name (see the "ARD registration"
|
|
4
|
+
note in tasks.md):
|
|
5
|
+
|
|
6
|
+
- **Internal registration:** a built capability is registered by its name with its contract, so the
|
|
7
|
+
MCP skill surface (T31) can expose it.
|
|
8
|
+
- **ARD registration:** registration also emits an ARD manifest *skeleton* (RegistryEntry-shaped,
|
|
9
|
+
the ADR-0005 mirror in `ard.py`) that GraphWright's compiler discovers and binds. The name is the
|
|
10
|
+
single shared key and the URN anchor, so the internal registry and the ARD manifest speak one
|
|
11
|
+
vocabulary. The `name` must be identical to the capability's name in the shared spec (the
|
|
12
|
+
cross-spec join key the Orchestration Spec binds against).
|
|
13
|
+
|
|
14
|
+
**Kind is explicit per capability** (no default), following the binding rule (docs/adr/0003):
|
|
15
|
+
`mcp_tool` crosses the MCP boundary (query-side governed skills, FR-S.5); `function` is an in-process
|
|
16
|
+
graph-node call (parser, embedder, reranker, fusion); `agent_skill` is loaded knowledge (RLM
|
|
17
|
+
chunking / synthesis, the RLM skill), not callable. A capability that fits none of the six kinds is
|
|
18
|
+
flagged, not forced.
|
|
19
|
+
|
|
20
|
+
**The emitted skeleton is a DRAFT.** Its representative queries (2-5, required by the schema) and
|
|
21
|
+
trust attestations are authored at the capability's own task via `ManifestSkeleton.author(...)`,
|
|
22
|
+
which returns the complete, validated `RegistryEntry` for the loadable path. A draft is never a
|
|
23
|
+
`RegistryEntry` and must be kept out of the loadable registry directory: GraphWright's
|
|
24
|
+
`RegistryStore` globs `*.json` non-recursively at the root, so drafts belong in a `staging/`
|
|
25
|
+
subdirectory (never globbed) until authored. Nothing, including the compiler's glob, ever loads a
|
|
26
|
+
partial as if it were registered.
|
|
27
|
+
"""
|
|
28
|
+
|
|
29
|
+
from __future__ import annotations
|
|
30
|
+
|
|
31
|
+
from dataclasses import dataclass
|
|
32
|
+
from typing import Optional
|
|
33
|
+
|
|
34
|
+
from pydantic import BaseModel
|
|
35
|
+
|
|
36
|
+
from rag_wright.capabilities.ard import (
|
|
37
|
+
CALLABLE_KINDS,
|
|
38
|
+
MEDIA_TYPE_BY_KIND,
|
|
39
|
+
URN_NAMESPACE,
|
|
40
|
+
URN_PUBLISHER,
|
|
41
|
+
ArdEnvelope,
|
|
42
|
+
Attestation,
|
|
43
|
+
CapabilityInterface,
|
|
44
|
+
EntryKind,
|
|
45
|
+
GovernanceBlock,
|
|
46
|
+
RegistryEntry,
|
|
47
|
+
ResponseBounds,
|
|
48
|
+
SkillRuntime,
|
|
49
|
+
TrustManifest,
|
|
50
|
+
)
|
|
51
|
+
|
|
52
|
+
# The ARD URN is domain-anchored (ARD v0.9 4.2.1): urn:air:<publisher>:<namespace>:<name>. The
|
|
53
|
+
# publisher and namespace are the single source in ard.py (dreamai.io / rag_wright).
|
|
54
|
+
_DEFAULT_OWNER = "dreamai.io"
|
|
55
|
+
|
|
56
|
+
# The canonical capability slugs are the cross-spec join keys (SPEC.md section 5, "FR-C canonical
|
|
57
|
+
# slugs"). The slug is the semantic capability name, never a requirement id (`hybrid_search`, never
|
|
58
|
+
# `fr-c-3`), so it survives spec renumbering. SPEC.md is authoritative; this frozenset mirrors it so
|
|
59
|
+
# registration rejects a name that is not a canonical capability (a typo, a requirement id, or an
|
|
60
|
+
# invented name) — the same shared-vocabulary discipline as the ARD schema mirror. Adding a
|
|
61
|
+
# capability updates both. FR-C.9 is a single slug (`generation`): reasoning, generation, and
|
|
62
|
+
# vision-to-text are one capability, bound at whichever node needs them.
|
|
63
|
+
CANONICAL_CAPABILITY_SLUGS: frozenset[str] = frozenset(
|
|
64
|
+
{
|
|
65
|
+
"graph_extraction", # FR-C.6
|
|
66
|
+
"entity_disambiguation", # FR-C.7 (canonicalize: normalize/reject/cluster; the T23b stage)
|
|
67
|
+
"entity_resolution", # FR-C.7 (closed-world linking to EDGAR CIK)
|
|
68
|
+
"ontology_registry_derivation", # FR-C.8 (foundation derivation: slug, but no ARD manifest)
|
|
69
|
+
"generation", # FR-C.9 (grounded/cited/abstaining answer generation)
|
|
70
|
+
"vision_to_text", # agent_skill: FR-C.9 single vision-language act (SKILL.md); split from generation (ADR-0014), SKILL-SPLIT
|
|
71
|
+
"rlm_chunking", # FR-I.1 (applies the RLM skill; dynamic RLM discoverer -> agent_skill)
|
|
72
|
+
"rlm_synthesis", # FR-Q.5 (applies the RLM skill)
|
|
73
|
+
"rlm_method", # FR-C.10 (the shared RLM method skill, if registered)
|
|
74
|
+
"okf_compile", # FR-K.1-K.4 (foundation derivation: slug, but no ARD manifest; T46)
|
|
75
|
+
"okf_navigate", # FR-K.6 (query-discovered traversal: slug + ARD manifest; T50)
|
|
76
|
+
# --- CAP-REG-2: the built contract-KG capabilities ---
|
|
77
|
+
"typed_value_normalization", # function: canonicalization + subsumption (KG-5a)
|
|
78
|
+
"extraction_grounding_judge", # function: ADR-0028 lexical grounding gate
|
|
79
|
+
"extraction_semantic_judge", # agent_skill: ADR-0040 Layer 3 verify-or-refute reading (SKILL.md); SKILL-SPLIT
|
|
80
|
+
"extraction_semantic_gate", # function: applies the semantic-judge skill + AMBIGUOUS downgrade (deterministic)
|
|
81
|
+
"intra_document_scoped_query", # function: intra-contract scoped KG serving (Leg A)
|
|
82
|
+
"clause_disambiguation", # function: disambiguation by property (Leg A)
|
|
83
|
+
"clause_function_classification", # model: LegalBERT function classifier (T56)
|
|
84
|
+
"clause_property_classification", # model: the 29-dim best-of-both property classifier fleet (ADR-0115/0116, EP-RT-1)
|
|
85
|
+
"jev_decision", # model: generic Jev System-1 typed-decision client (OpenRouter Decisions API), ADR-0119; async/IO-bound
|
|
86
|
+
"query_function_classification", # agent_skill: taxonomy-constrained query->function (KG-5e)
|
|
87
|
+
# --- CAP-REG-3: the KG-primary retrieval core (packaged out of eval/kg_primary.py) ---
|
|
88
|
+
"span_relevance_judgment", # agent_skill: per-span relevance VERDICT (span x condition -> relevant/not/uncertain); issue 0023, SKILL-SPLIT
|
|
89
|
+
# --- KG-7: the Party<->Contract unifying link over the one contract KG (ADR-0036) ---
|
|
90
|
+
"clause_exception_linking", # function: IsExceptionTo edges (Uncapped -> Cap carve-out) by proximity (ADR-0044)
|
|
91
|
+
# --- LG-1/LG-2: hardened LangGraph subgraphs ---
|
|
92
|
+
"typed_clause_extraction", # subgraph: extract -> adapt -> reground -> escalate -> [HITL] -> dead-letter
|
|
93
|
+
"query_constraint_extraction", # subgraph: query-side typed constraint extraction (graceful-empty)
|
|
94
|
+
# --- LG-3: composite pipeline subgraphs (compose the component subgraphs + registered capabilities) ---
|
|
95
|
+
"relational_qa", # subgraph: graph_query -> chunk_read -> generate_answer (entity question -> cited answer)
|
|
96
|
+
"intra_document_qa", # subgraph: scoped KG query -> generate_answer (contract + question -> cited answer)
|
|
97
|
+
"typed_property_retrieval", # subgraph: front-door + property_boosted_retrieval (Leg B, LEGB-SUBGRAPH). THE
|
|
98
|
+
# corpus-wide function+property retrieval leg; retired the redundant cross_corpus_retrieval (inferior pool)
|
|
99
|
+
"contract_ingestion_pipeline", # subgraph: generic corpus ingest (chunk->extract->resolve->write->link)
|
|
100
|
+
# --- Compliance module rung 1 (roadmap §13): the ad-compliance engine ---
|
|
101
|
+
"requirement_extraction", # subgraph: extract(docling-graph, multi-call) -> adapt; regulatory section -> Requirement[] (CC-2, SKILL-SPLIT)
|
|
102
|
+
"requirement_adaptation", # function: ExtractedRegulationSection -> validated Requirement[] (deterministic)
|
|
103
|
+
"claim_extraction", # agent_skill: single LLM extraction act (ad -> ExtractedAd); SKILL-SPLIT
|
|
104
|
+
"claim_adaptation", # function: ExtractedAd -> validated Claim[] (deterministic)
|
|
105
|
+
"compliance_judgment", # agent_skill: single LLM judgment act ((claim, requirement) -> verdict); SKILL-SPLIT
|
|
106
|
+
"compliance_finding_assembly", # function: raw verdict + inputs -> cited ComplianceFinding (deterministic)
|
|
107
|
+
"compliance_ingestion", # subgraph: regulatory corpus -> Requirement KG (CC-5)
|
|
108
|
+
"compliance_check", # subgraph: subject doc x requirements -> cited findings + gap matrix (CC-6)
|
|
109
|
+
"compliance_check_mcp", # mcp_tool: the discoverable MCP-tool surface of compliance_check (MCP-PROTO)
|
|
110
|
+
"intra_document_qa_mcp", # mcp_tool: the discoverable MCP-tool surface of intra_document_qa (MCP-PROTO B1)
|
|
111
|
+
"relational_qa_mcp", # mcp_tool: the discoverable MCP-tool surface of relational_qa (MCP-PROTO B2)
|
|
112
|
+
"typed_property_retrieval_mcp", # mcp_tool: the MCP-tool surface of typed_property_retrieval (MCP-PROTO B3)
|
|
113
|
+
}
|
|
114
|
+
)
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def capability_urn(name: str) -> str:
|
|
118
|
+
"""The domain-anchored ARD URN for a capability name (urn:air:dreamai.io:rag_wright:<name>)."""
|
|
119
|
+
return f"urn:air:{URN_PUBLISHER}:{URN_NAMESPACE}:{name}"
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
class ManifestSkeleton(BaseModel):
|
|
123
|
+
"""A DRAFT ARD manifest: the fields registration can derive, minus the authored fields.
|
|
124
|
+
|
|
125
|
+
Not loadable as a `RegistryEntry` (it has no representative queries yet). `author(...)` fills the
|
|
126
|
+
authored fields and returns the complete, validated `RegistryEntry`.
|
|
127
|
+
"""
|
|
128
|
+
|
|
129
|
+
name: str
|
|
130
|
+
kind: EntryKind
|
|
131
|
+
identifier: str # the URN
|
|
132
|
+
media_type: str
|
|
133
|
+
display_name: str
|
|
134
|
+
response_bounds: Optional[ResponseBounds] = None # present for callable kinds only
|
|
135
|
+
owner: str = _DEFAULT_OWNER
|
|
136
|
+
description: Optional[str] = None
|
|
137
|
+
tags: list[str] = []
|
|
138
|
+
|
|
139
|
+
def author(
|
|
140
|
+
self,
|
|
141
|
+
representative_queries: list[str],
|
|
142
|
+
*,
|
|
143
|
+
attestations: Optional[list[Attestation]] = None,
|
|
144
|
+
identity: Optional[str] = None,
|
|
145
|
+
identity_type: str = "domain",
|
|
146
|
+
golden_eval_ref: Optional[str] = None,
|
|
147
|
+
requires: Optional[list[str]] = None,
|
|
148
|
+
skill_runtime: Optional[SkillRuntime] = None,
|
|
149
|
+
capability_interface: Optional[CapabilityInterface] = None,
|
|
150
|
+
description: Optional[str] = None,
|
|
151
|
+
tags: Optional[list[str]] = None,
|
|
152
|
+
) -> RegistryEntry:
|
|
153
|
+
"""Author the draft into a complete, validated `RegistryEntry` for the loadable path.
|
|
154
|
+
|
|
155
|
+
`representative_queries` (2-5) and any trust `attestations` are the fields that could not be
|
|
156
|
+
derived at registration; the schema validators enforce them. `capability_interface` is the
|
|
157
|
+
optional GraphWright vendor extension (ADR-0030), declared per capability at its own task.
|
|
158
|
+
"""
|
|
159
|
+
envelope = ArdEnvelope(
|
|
160
|
+
identifier=self.identifier,
|
|
161
|
+
display_name=self.display_name,
|
|
162
|
+
type=self.media_type,
|
|
163
|
+
representative_queries=representative_queries,
|
|
164
|
+
trust_manifest=TrustManifest(
|
|
165
|
+
identity=identity or self.identifier,
|
|
166
|
+
identity_type=identity_type,
|
|
167
|
+
attestations=attestations or [],
|
|
168
|
+
),
|
|
169
|
+
description=description if description is not None else self.description,
|
|
170
|
+
tags=tags if tags is not None else self.tags,
|
|
171
|
+
)
|
|
172
|
+
return RegistryEntry(
|
|
173
|
+
kind=self.kind,
|
|
174
|
+
envelope=envelope,
|
|
175
|
+
response_bounds=self.response_bounds,
|
|
176
|
+
requires=requires or [],
|
|
177
|
+
skill_runtime=skill_runtime,
|
|
178
|
+
capability_interface=capability_interface,
|
|
179
|
+
golden_eval_ref=golden_eval_ref,
|
|
180
|
+
governance=GovernanceBlock(owner=self.owner),
|
|
181
|
+
)
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
@dataclass(frozen=True)
|
|
185
|
+
class CapabilityRegistration:
|
|
186
|
+
"""One registered capability: its name, kind, contract, and its draft ARD manifest skeleton."""
|
|
187
|
+
|
|
188
|
+
name: str
|
|
189
|
+
kind: EntryKind
|
|
190
|
+
contract: type[BaseModel]
|
|
191
|
+
skeleton: ManifestSkeleton
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
class CapabilityRegistry:
|
|
195
|
+
"""An in-process registry of built capabilities, keyed by the capability name (FR-S.5)."""
|
|
196
|
+
|
|
197
|
+
def __init__(self) -> None:
|
|
198
|
+
self._by_name: dict[str, CapabilityRegistration] = {}
|
|
199
|
+
|
|
200
|
+
def register(
|
|
201
|
+
self,
|
|
202
|
+
name: str,
|
|
203
|
+
*,
|
|
204
|
+
contract: type[BaseModel],
|
|
205
|
+
kind: EntryKind,
|
|
206
|
+
display_name: Optional[str] = None,
|
|
207
|
+
description: Optional[str] = None,
|
|
208
|
+
tags: Optional[list[str]] = None,
|
|
209
|
+
response_bounds: Optional[ResponseBounds] = None,
|
|
210
|
+
) -> CapabilityRegistration:
|
|
211
|
+
"""Register a capability by name with its contract and (explicit) kind, emitting its ARD
|
|
212
|
+
skeleton. Rejects a malformed name, an unknown kind, or a duplicate registration."""
|
|
213
|
+
if name not in CANONICAL_CAPABILITY_SLUGS:
|
|
214
|
+
raise ValueError(
|
|
215
|
+
f"{name!r} is not a canonical capability slug (SPEC.md section 5, FR-C canonical "
|
|
216
|
+
f"slugs); register under the exact slug, one of {sorted(CANONICAL_CAPABILITY_SLUGS)}"
|
|
217
|
+
)
|
|
218
|
+
if kind not in MEDIA_TYPE_BY_KIND:
|
|
219
|
+
raise ValueError(
|
|
220
|
+
f"unknown capability kind {kind!r}; must be one of {sorted(MEDIA_TYPE_BY_KIND)} "
|
|
221
|
+
"(flag a capability that fits none of the six rather than forcing it)"
|
|
222
|
+
)
|
|
223
|
+
if name in self._by_name:
|
|
224
|
+
raise ValueError(f"capability {name!r} is already registered (duplicate)")
|
|
225
|
+
|
|
226
|
+
if kind in CALLABLE_KINDS:
|
|
227
|
+
bounds = response_bounds or ResponseBounds()
|
|
228
|
+
else: # agent_skill is loaded, not called
|
|
229
|
+
if response_bounds is not None:
|
|
230
|
+
raise ValueError(f"kind {kind!r} is not callable and must not declare response_bounds")
|
|
231
|
+
bounds = None
|
|
232
|
+
|
|
233
|
+
skeleton = ManifestSkeleton(
|
|
234
|
+
name=name,
|
|
235
|
+
kind=kind,
|
|
236
|
+
identifier=capability_urn(name),
|
|
237
|
+
media_type=MEDIA_TYPE_BY_KIND[kind],
|
|
238
|
+
display_name=display_name or name,
|
|
239
|
+
response_bounds=bounds,
|
|
240
|
+
description=description,
|
|
241
|
+
tags=tags or [],
|
|
242
|
+
)
|
|
243
|
+
registration = CapabilityRegistration(
|
|
244
|
+
name=name, kind=kind, contract=contract, skeleton=skeleton
|
|
245
|
+
)
|
|
246
|
+
self._by_name[name] = registration
|
|
247
|
+
return registration
|
|
248
|
+
|
|
249
|
+
def get(self, name: str) -> CapabilityRegistration:
|
|
250
|
+
"""The registration for a name, or raise `KeyError` if the name is unknown."""
|
|
251
|
+
if name not in self._by_name:
|
|
252
|
+
raise KeyError(f"no capability registered under {name!r}")
|
|
253
|
+
return self._by_name[name]
|
|
254
|
+
|
|
255
|
+
def __contains__(self, name: object) -> bool:
|
|
256
|
+
return name in self._by_name
|
|
257
|
+
|
|
258
|
+
def __len__(self) -> int:
|
|
259
|
+
return len(self._by_name)
|
|
260
|
+
|
|
261
|
+
def names(self) -> list[str]:
|
|
262
|
+
return sorted(self._by_name)
|
|
@@ -0,0 +1,94 @@
|
|
|
1
|
+
"""MS1-6 (MODAL-STACK-1, ADR-0039): HTTP-client adapters that route the query-side encoders to the co-located
|
|
2
|
+
A100 GPU-services stack (`scripts/modal_stack_a100.py`).
|
|
3
|
+
|
|
4
|
+
The LLM surfaces route to vLLM-Granite via the model seam (`RAG_SERVING`, MS1-1/3); BGE-M3 embed + LegalBERT
|
|
5
|
+
classify are DIRECT classes, not through the seam, so they get their own remote adapters here. Each matches
|
|
6
|
+
the local interface it replaces (`embedding.Embedder` / `LegalBertFunctionClassifier`), so it drops into the
|
|
7
|
+
same query seam -- the product then runs the query's non-LLM GPU work (embed + classify) on the SAME A100 as
|
|
8
|
+
vLLM, the whole point of D2. Selected by the `STACK_URL` env (mirroring the LLM's `RAG_SERVING` switch); unset
|
|
9
|
+
-> the local in-process encoders. `post` is injected so the adapters are hermetically testable (no network).
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import json
|
|
15
|
+
import os
|
|
16
|
+
import urllib.request
|
|
17
|
+
from typing import Any, Callable
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _post_json(url: str, payload: dict, timeout: int = 120) -> dict:
|
|
21
|
+
req = urllib.request.Request(
|
|
22
|
+
url, data=json.dumps(payload).encode(),
|
|
23
|
+
headers={"Content-Type": "application/json"}, method="POST")
|
|
24
|
+
with urllib.request.urlopen(req, timeout=timeout) as r:
|
|
25
|
+
return json.loads(r.read())
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
class RemoteBGEEmbedder:
|
|
29
|
+
"""BGE-M3 dense/sparse encode via the A100 `/embed` endpoint. Matches `embedding.BGEM3Embedder`
|
|
30
|
+
(encode_dense / encode_sparse / encode_batch), producing vectors in the SAME space as the KG's stored
|
|
31
|
+
span vectors (verified byte-identical in MS1-5b)."""
|
|
32
|
+
|
|
33
|
+
def __init__(self, base_url: str, *, post: Callable[..., dict] = _post_json) -> None:
|
|
34
|
+
self._url = base_url.rstrip("/") + "/embed"
|
|
35
|
+
self._post = post
|
|
36
|
+
|
|
37
|
+
def encode_dense(self, text: str) -> list[float]:
|
|
38
|
+
return self._post(self._url, {"text": text})["dense"][0]
|
|
39
|
+
|
|
40
|
+
def encode_sparse(self, text: str) -> dict[int, float]:
|
|
41
|
+
return {int(k): float(v) for k, v in self._post(self._url, {"text": text})["sparse"][0].items()}
|
|
42
|
+
|
|
43
|
+
def encode_batch(self, texts: list[str]) -> tuple[list[list[float]], list[dict[int, float]]]:
|
|
44
|
+
out = self._post(self._url, {"texts": list(texts)})
|
|
45
|
+
sparse = [{int(k): float(v) for k, v in s.items()} for s in out["sparse"]]
|
|
46
|
+
return out["dense"], sparse
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
class RemoteLegalBertClassifier:
|
|
50
|
+
"""LegalBERT function classification via the A100 `/classify` endpoint. Matches
|
|
51
|
+
`LegalBertFunctionClassifier` (classify / classify_topk)."""
|
|
52
|
+
|
|
53
|
+
def __init__(self, base_url: str, *, post: Callable[..., dict] = _post_json) -> None:
|
|
54
|
+
self._url = base_url.rstrip("/") + "/classify"
|
|
55
|
+
self._post = post
|
|
56
|
+
|
|
57
|
+
def classify(self, texts: list[str], *, batch_size: int = 32) -> list[str]:
|
|
58
|
+
if not texts:
|
|
59
|
+
return []
|
|
60
|
+
return self._post(self._url, {"texts": list(texts)})["labels"]
|
|
61
|
+
|
|
62
|
+
def classify_topk(self, texts: list[str], *, k: int = 2, batch_size: int = 32) -> list[list[str]]:
|
|
63
|
+
if not texts:
|
|
64
|
+
return []
|
|
65
|
+
return self._post(self._url, {"texts": list(texts), "k": k})["topk"]
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def stack_url() -> str | None:
|
|
69
|
+
"""The co-located A100 stack base URL from `STACK_URL` (the query-encoder serving switch), or None (local)."""
|
|
70
|
+
return os.getenv("STACK_URL")
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def query_embedder(*, post: Callable[..., dict] = _post_json) -> Any:
|
|
74
|
+
"""The query-side embedder: the remote A100 `/embed` adapter when `STACK_URL` is set, else the local
|
|
75
|
+
in-process `BGEM3Embedder` (dev/default). One env (`STACK_URL`) moves the query's embed onto the A100."""
|
|
76
|
+
url = stack_url()
|
|
77
|
+
if url:
|
|
78
|
+
return RemoteBGEEmbedder(url, post=post)
|
|
79
|
+
from rag_wright.capabilities.embedding import BGEM3Embedder
|
|
80
|
+
|
|
81
|
+
return BGEM3Embedder()
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def query_classifier(model_path: Any = None, *, post: Callable[..., dict] = _post_json) -> Any:
|
|
85
|
+
"""The query-side function classifier: the remote A100 `/classify` adapter when `STACK_URL` is set, else
|
|
86
|
+
the local in-process `LegalBertFunctionClassifier` loaded from `model_path`."""
|
|
87
|
+
url = stack_url()
|
|
88
|
+
if url:
|
|
89
|
+
return RemoteLegalBertClassifier(url, post=post)
|
|
90
|
+
from pathlib import Path
|
|
91
|
+
|
|
92
|
+
from rag_wright.spans.legalbert_classifier import LegalBertFunctionClassifier
|
|
93
|
+
|
|
94
|
+
return LegalBertFunctionClassifier.load(Path(model_path or "data/models/legalbert_function"))
|
|
@@ -0,0 +1,247 @@
|
|
|
1
|
+
"""CC-2 (compliance §13), SKILL-SPLIT: the extraction ACT + the adaptation FUNCTION for `requirement_extraction`.
|
|
2
|
+
|
|
3
|
+
Per the capability-architecture rubric, `requirement_extraction` is a SUBGRAPH (its `auto/dense` extraction is
|
|
4
|
+
MULTI-LLM-call; the extract -> adapt chaining is the deterministic workflow). This module holds the subgraph's
|
|
5
|
+
two pieces:
|
|
6
|
+
|
|
7
|
+
- **the extraction ACT** (`extract_regulation_section`) -- the docling-graph schema-driven extraction using the
|
|
8
|
+
co-located skill asset `skills/requirement_extraction/template.py` (ExtractedRegulationSection). Runs `"auto"`
|
|
9
|
+
(dense on long sections, skeleton-then-fill). The subgraph's extract node.
|
|
10
|
+
- **`requirement_adaptation` (function)** -- `to_requirements`: DETERMINISTIC, no model. Maps the raw extraction
|
|
11
|
+
to validated `Requirement`s (deontic vocab coercion -> AMBIGUOUS; off-vocab claim_type dropped; blank skipped;
|
|
12
|
+
citation = the section; content-hash id). The subgraph's adapt node.
|
|
13
|
+
|
|
14
|
+
The subgraph itself (extract -> adapt, hardened) is `subgraphs/requirement_extraction.py`. The template lives
|
|
15
|
+
with the skill (an Agent-Skill asset), re-exported here for consumers/tests.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
import asyncio
|
|
21
|
+
from typing import Any, Callable
|
|
22
|
+
|
|
23
|
+
from rag_wright.capabilities.dg_extraction import aextract_parties, extract_parties
|
|
24
|
+
from rag_wright.contracts.compliance import ClaimType, Constraint, DeonticType, Requirement
|
|
25
|
+
from rag_wright.contracts.provenance import ConfidenceTag
|
|
26
|
+
from rag_wright.skills.requirement_extraction.template import ( # the skill's schema asset
|
|
27
|
+
ExtractedRegulationSection,
|
|
28
|
+
ExtractedRequirement,
|
|
29
|
+
)
|
|
30
|
+
|
|
31
|
+
__all__ = ["ExtractedRegulationSection", "ExtractedRequirement", "extract_regulation_section",
|
|
32
|
+
"ajev_extract_regulation_section", "operative_rule_spans", "to_requirements",
|
|
33
|
+
"register_requirement_adaptation"]
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def operative_rule_spans(text: str) -> list[tuple[str, str]]:
|
|
37
|
+
"""CIC (ADR-0119 scaffolding): deterministically split a section's text into candidate rule spans (reusing the
|
|
38
|
+
engine's byte-faithful `segment_clause`), keep the OPERATIVE ones (carrying a deontic cue), and derive each
|
|
39
|
+
span's deontic_type from its cue (ttl-driven `deontic_type_of`). Returns `[(verbatim_span_text,
|
|
40
|
+
deontic_local_name)]`. NO LLM. This is the span-producer front end a decision model (`jev_decision`, ADR-0119)
|
|
41
|
+
judges per span; it is NOT wired into the live ingest extraction, which is unchanged."""
|
|
42
|
+
from rag_wright.ontology.loader import deontic_type_of
|
|
43
|
+
from rag_wright.spans.segment import segment_clause
|
|
44
|
+
|
|
45
|
+
out: list[tuple[str, str]] = []
|
|
46
|
+
for span in segment_clause("", text or ""):
|
|
47
|
+
span_text = span.text.strip()
|
|
48
|
+
deontic = deontic_type_of(span_text)
|
|
49
|
+
if deontic is not None:
|
|
50
|
+
out.append((span_text, deontic))
|
|
51
|
+
return out
|
|
52
|
+
|
|
53
|
+
_CLAIM_TYPES = {c.value for c in ClaimType}
|
|
54
|
+
_DEONTIC = {d.value for d in DeonticType}
|
|
55
|
+
|
|
56
|
+
# --- CIC (ADR-0119): the Jev-decision extraction act -- deterministic spans + ONE Jev call/span (operative gate +
|
|
57
|
+
# claim_types + actor), cue-deontic, verbatim text. The decision KNOWLEDGE (the operative rubric, the per-ClaimType
|
|
58
|
+
# and per-ActorRole criteria) is authored in compliance_bridge.ttl and LOADED here (ADR-0066/0119), never hardcoded.
|
|
59
|
+
# The few-shot GUIDANCE below is prompt-engineering overlay (code), not domain vocab. The open fields applicability
|
|
60
|
+
# + evidence_standard are NOT produced here (open-vocab; a residual-LLM/ttl follow-up) -- for FTC the claim_types
|
|
61
|
+
# ARE the applicability (claim_type constraints via `to_requirements`).
|
|
62
|
+
_JEV_GUIDANCE = (
|
|
63
|
+
"Guidance: 'This supplement cures insomnia' -> efficacy, health; 'lasts 3x longer than Brand X' -> "
|
|
64
|
+
"performance, comparative; 'reduced to $9.99' -> pricing; 'as recommended by Dr. Smith' -> endorsement. "
|
|
65
|
+
"Actor: who the rule binds.\n\nRule:\n")
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def _jev_questions() -> dict:
|
|
69
|
+
"""Build the Jev question-set from the ttl decision knowledge (ADR-0119): the operative rubric
|
|
70
|
+
(`load_operative_rubric`), the actor `choice` (`load_actor_role_criteria` + an 'other' catch-all), and one
|
|
71
|
+
`noul` per ClaimType (`load_claim_type_criteria`)."""
|
|
72
|
+
from rag_wright.ontology.loader import (
|
|
73
|
+
load_actor_role_criteria, load_claim_type_criteria, load_operative_rubric,
|
|
74
|
+
)
|
|
75
|
+
|
|
76
|
+
rub = load_operative_rubric()
|
|
77
|
+
actor_criteria = {**load_actor_role_criteria(), "other": "none of the above, multiple parties, or unspecified"}
|
|
78
|
+
q = {"operative": {"type": "noul", "instructions": rub["instructions"],
|
|
79
|
+
"criteria": {"true": rub["true"], "false": rub["false"]}},
|
|
80
|
+
"actor": {"type": "choice", "instructions": "Who does this rule primarily bind?", "criteria": actor_criteria}}
|
|
81
|
+
for ct, desc in load_claim_type_criteria().items():
|
|
82
|
+
q[f"ct_{ct}"] = {"type": "noul", "instructions": f"Does this rule apply to {ct} advertising claims?",
|
|
83
|
+
"criteria": {"true": desc, "false": f"not specifically about {ct}"}}
|
|
84
|
+
return q
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
# --- ADR-0119: the GATED residual extraction for the OPEN fields (applicability + evidence_standard). These are
|
|
88
|
+
# open-text, not closed decisions, so Jev cannot produce them; a tiny structured LLM call fills them ONLY for the
|
|
89
|
+
# rules that carry a conditional/evidence cue (most rules skip it -> per-rule LLM stays near-zero).
|
|
90
|
+
_COND_CUES = ("if ", "where ", "unless", "provided", "only if", "when ", "except")
|
|
91
|
+
_EVID_CUES = ("substantiat", "evidence", "competent and reliable", "scientific", "proof")
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def _needs_residual(span_text: str) -> bool:
|
|
95
|
+
low = span_text.lower()
|
|
96
|
+
return any(c in low for c in _COND_CUES) or any(c in low for c in _EVID_CUES)
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
async def _residual_open_fields(span_text: str, model_id: str) -> tuple[list[str], str]:
|
|
100
|
+
"""Extract the OPEN fields (applicability conditions + evidence standard) of one rule via a tiny structured
|
|
101
|
+
LLM call (the model-profile seam). Gated by `_needs_residual`, so this runs only on the few rules that have a
|
|
102
|
+
conditional/evidence cue. Degrades to ([], '') on any failure (recall-first: never drop the rule)."""
|
|
103
|
+
from pydantic import BaseModel, Field
|
|
104
|
+
|
|
105
|
+
from rag_wright.models.seam import build_structured
|
|
106
|
+
|
|
107
|
+
class _Open(BaseModel):
|
|
108
|
+
applicability: list[str] = Field(
|
|
109
|
+
default_factory=list,
|
|
110
|
+
description="conditions the rule applies under, as 'dimension: value' (e.g. 'jurisdiction: California', "
|
|
111
|
+
"'employee_class: hourly'); empty if the rule applies unconditionally")
|
|
112
|
+
evidence_standard: str = Field(
|
|
113
|
+
default="", description="the substantiation the rule requires, if any (e.g. 'competent and reliable scientific evidence')")
|
|
114
|
+
|
|
115
|
+
try:
|
|
116
|
+
out = await build_structured(model_id, _Open).ainvoke(
|
|
117
|
+
f"Extract the applicability conditions and the evidence standard of this regulatory rule, if any:\n{span_text}")
|
|
118
|
+
return list(out.applicability or []), (out.evidence_standard or "").strip()
|
|
119
|
+
except Exception: # noqa: BLE001 - recall-first: open fields are best-effort, never fail the rule
|
|
120
|
+
return [], ""
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
async def ajev_extract_regulation_section(
|
|
124
|
+
text: str, *, concurrency: int = 8, op_threshold: float | None = None, ct_threshold: float | None = None,
|
|
125
|
+
residual_model_id: str | None = None,
|
|
126
|
+
) -> ExtractedRegulationSection:
|
|
127
|
+
"""CIC (ADR-0119): the Jev-decision extraction act. Deterministic `operative_rule_spans` produces candidate
|
|
128
|
+
spans; ONE Jev call per span (routed through the capability layer, `jev_decision`) decides operative-gate +
|
|
129
|
+
actor + claim_types; deontic_type is the cue-rule; requirement_text is the verbatim span. Drops spans the Jev
|
|
130
|
+
operative gate rejects. Thresholds default from the decision-model profile (`op_threshold`/`multilabel`),
|
|
131
|
+
overridable here. applicability/evidence_standard are left empty here (see note)."""
|
|
132
|
+
from rag_wright.capabilities.invoke import capability_impl
|
|
133
|
+
from rag_wright.models.profiles import decision_profile
|
|
134
|
+
|
|
135
|
+
spans = operative_rule_spans(text)
|
|
136
|
+
if not spans:
|
|
137
|
+
return ExtractedRegulationSection(section="", requirements=[])
|
|
138
|
+
prof = decision_profile()
|
|
139
|
+
op_thr = prof.op_threshold if op_threshold is None else op_threshold
|
|
140
|
+
ct_thr = prof.multilabel_threshold if ct_threshold is None else ct_threshold
|
|
141
|
+
if residual_model_id is None:
|
|
142
|
+
from rag_wright.models.profiles import ModelRole, model_for
|
|
143
|
+
residual_model_id = model_for(ModelRole.STRUCTURED_REASONING)
|
|
144
|
+
jev = capability_impl("jev_decision") # async (resources, inputs) -> decision body; store-independent
|
|
145
|
+
questions = _jev_questions()
|
|
146
|
+
ct_keys = [k[3:] for k in questions if k.startswith("ct_")] # the ttl-driven ClaimType set
|
|
147
|
+
sem = asyncio.Semaphore(max(1, concurrency))
|
|
148
|
+
|
|
149
|
+
async def _one(span_text: str, deontic: str) -> ExtractedRequirement | None:
|
|
150
|
+
async with sem:
|
|
151
|
+
d = await jev(None, {"state": _JEV_GUIDANCE + span_text, "questions": questions})
|
|
152
|
+
ans = d["answers"]
|
|
153
|
+
if float(ans["operative"].get("noul", 0.0)) < op_thr:
|
|
154
|
+
return None # Jev gate: not a binding rule -> drop (refines the cue-presence gate's precision)
|
|
155
|
+
claim_types = [ct for ct in ct_keys if float(ans[f"ct_{ct}"].get("noul", 0.0)) >= ct_thr]
|
|
156
|
+
# GATED residual: open fields only for rules with a conditional/evidence cue (most skip -> no LLM)
|
|
157
|
+
applicability, evidence = ([], "")
|
|
158
|
+
if _needs_residual(span_text):
|
|
159
|
+
applicability, evidence = await _residual_open_fields(span_text, residual_model_id)
|
|
160
|
+
return ExtractedRequirement(requirement_text=span_text, deontic_type=deontic,
|
|
161
|
+
actor=str(ans["actor"].get("choice", "")), claim_types=claim_types,
|
|
162
|
+
applicability=applicability, evidence_standard=evidence)
|
|
163
|
+
reqs = [r for r in await asyncio.gather(*(_one(t, d) for t, d in spans)) if r is not None]
|
|
164
|
+
return ExtractedRegulationSection(section="", requirements=reqs)
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
def _coerce_deontic(raw: str) -> tuple[DeonticType, bool]:
|
|
168
|
+
"""(DeonticType, ambiguous?) -- map the model's string to the closed vocab; an off-vocab value coerces to
|
|
169
|
+
OBLIGATION and flags AMBIGUOUS (a rule with an unreadable force is kept but marked, never dropped)."""
|
|
170
|
+
value = (raw or "").strip().lower()
|
|
171
|
+
if value in _DEONTIC:
|
|
172
|
+
return DeonticType(value), False
|
|
173
|
+
return DeonticType.OBLIGATION, True
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
ExtractFn = Callable[..., Any] # (text, model, *, template, **kw) -> ExtractedRegulationSection | None
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
def extract_regulation_section(
|
|
180
|
+
text: str, *, model: Any, extract_fn: ExtractFn = extract_parties,
|
|
181
|
+
max_tokens: int = 2000, preamble_chars: int = 24_000, extraction_contract: str = "auto",
|
|
182
|
+
) -> ExtractedRegulationSection | None:
|
|
183
|
+
"""The extraction ACT (the requirement_extraction skill, docling-graph runtime): fill the skill's
|
|
184
|
+
`template.py` schema from a § section's text. `extraction_contract="auto"` -> dense (multi-call) on long
|
|
185
|
+
sections so rules are not silently self-rationed. `extract_fn` is injected for hermetic tests."""
|
|
186
|
+
return extract_fn(text, model, template=ExtractedRegulationSection,
|
|
187
|
+
max_tokens=max_tokens, preamble_chars=preamble_chars,
|
|
188
|
+
extraction_contract=extraction_contract)
|
|
189
|
+
|
|
190
|
+
|
|
191
|
+
async def aextract_regulation_section(
|
|
192
|
+
text: str, *, model: Any, aextract_fn: Any = aextract_parties,
|
|
193
|
+
max_tokens: int = 2000, preamble_chars: int = 24_000, extraction_contract: str = "auto",
|
|
194
|
+
) -> ExtractedRegulationSection | None:
|
|
195
|
+
"""ASYNC (ADR-0057): the async twin of `extract_regulation_section` -- docling-graph extraction on the async
|
|
196
|
+
seam (`aextract_parties`, true wall-clock deadline via the injected client). `aextract_fn` injected for tests."""
|
|
197
|
+
return await aextract_fn(text, model, template=ExtractedRegulationSection,
|
|
198
|
+
max_tokens=max_tokens, preamble_chars=preamble_chars,
|
|
199
|
+
extraction_contract=extraction_contract)
|
|
200
|
+
|
|
201
|
+
|
|
202
|
+
def to_requirements(extracted: ExtractedRegulationSection, *, source: str, section: str) -> list[Requirement]:
|
|
203
|
+
"""`requirement_adaptation` (FUNCTION -- deterministic, no model): adapt an extracted section to validated
|
|
204
|
+
`Requirement`s. Off-vocab deontic -> AMBIGUOUS; off-vocab claim_type dropped; blank rule text skipped.
|
|
205
|
+
Citation = the section; id = the content-hash scheme."""
|
|
206
|
+
out: list[Requirement] = []
|
|
207
|
+
for item in extracted.requirements:
|
|
208
|
+
text = (item.requirement_text or "").strip()
|
|
209
|
+
if not text:
|
|
210
|
+
continue
|
|
211
|
+
deontic, ambiguous = _coerce_deontic(item.deontic_type)
|
|
212
|
+
# claim_type is the ADVERTISING reference (closed FTC vocab -> off-vocab dropped, unchanged).
|
|
213
|
+
scope = [Constraint(dimension="claim_type", value=ct.strip().lower())
|
|
214
|
+
for ct in item.claim_types if ct.strip().lower() in _CLAIM_TYPES]
|
|
215
|
+
# P3a (Gap 2): generic 'dimension: value' applicability conditions for ANY policy domain -- kept
|
|
216
|
+
# RECALL-FIRST (an unknown customer dimension is NOT dropped; the query-side matcher is dimension-agnostic).
|
|
217
|
+
seen = {c.as_tuple() for c in scope}
|
|
218
|
+
for cond in getattr(item, "applicability", None) or []:
|
|
219
|
+
dim, sep, val = (cond or "").partition(":")
|
|
220
|
+
dim, val = dim.strip().lower(), val.strip().lower()
|
|
221
|
+
if not (sep and dim and val) or (dim, val) in seen:
|
|
222
|
+
continue
|
|
223
|
+
seen.add((dim, val))
|
|
224
|
+
scope.append(Constraint(dimension=dim, value=val))
|
|
225
|
+
out.append(Requirement(
|
|
226
|
+
requirement_id=Requirement.make_id(source, section, text),
|
|
227
|
+
source=source,
|
|
228
|
+
citation=f"§ {section}",
|
|
229
|
+
deontic_type=deontic,
|
|
230
|
+
actor=(item.actor or "").strip() or "unspecified",
|
|
231
|
+
applicability_scope=scope,
|
|
232
|
+
requirement_text=text,
|
|
233
|
+
evidence_standard=(item.evidence_standard or "").strip() or None,
|
|
234
|
+
confidence=ConfidenceTag.AMBIGUOUS if ambiguous else ConfidenceTag.EXTRACTED,
|
|
235
|
+
))
|
|
236
|
+
return out
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
def register_requirement_adaptation(registry) -> None:
|
|
240
|
+
"""Register `requirement_adaptation` (FUNCTION -- deterministic): the raw ExtractedRegulationSection ->
|
|
241
|
+
validated `Requirement[]` (deontic coercion, off-vocab handling, citation, content-hash id). No model."""
|
|
242
|
+
registry.register(
|
|
243
|
+
"requirement_adaptation",
|
|
244
|
+
contract=Requirement,
|
|
245
|
+
kind="function",
|
|
246
|
+
display_name="Requirement adaptation (extracted section -> validated Requirements)",
|
|
247
|
+
)
|