rag-wright 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- rag_wright/__init__.py +13 -0
- rag_wright/api/__init__.py +33 -0
- rag_wright/api/config.py +59 -0
- rag_wright/api/discover.py +70 -0
- rag_wright/api/documents.py +39 -0
- rag_wright/api/ids.py +31 -0
- rag_wright/api/invoke.py +99 -0
- rag_wright/api/kg.py +61 -0
- rag_wright/api/mcp.py +94 -0
- rag_wright/api/usage.py +30 -0
- rag_wright/api/workspace.py +85 -0
- rag_wright/capabilities/__init__.py +8 -0
- rag_wright/capabilities/answer_generator.py +427 -0
- rag_wright/capabilities/ard.py +286 -0
- rag_wright/capabilities/assertion_extraction.py +79 -0
- rag_wright/capabilities/chunk_read.py +58 -0
- rag_wright/capabilities/chunk_write.py +163 -0
- rag_wright/capabilities/claim_extraction.py +153 -0
- rag_wright/capabilities/clause_exception_linking.py +117 -0
- rag_wright/capabilities/compliance_judgment.py +322 -0
- rag_wright/capabilities/compliance_store.py +87 -0
- rag_wright/capabilities/contract_kg_serve.py +156 -0
- rag_wright/capabilities/contract_kg_store.py +251 -0
- rag_wright/capabilities/dg_extraction.py +585 -0
- rag_wright/capabilities/disambiguation.py +163 -0
- rag_wright/capabilities/document_parse.py +87 -0
- rag_wright/capabilities/document_scope.py +49 -0
- rag_wright/capabilities/embedding.py +164 -0
- rag_wright/capabilities/embedding_profiles.py +43 -0
- rag_wright/capabilities/entity_resolution.py +154 -0
- rag_wright/capabilities/fusion.py +64 -0
- rag_wright/capabilities/graph_extraction.py +243 -0
- rag_wright/capabilities/graph_query.py +73 -0
- rag_wright/capabilities/graph_storage.py +111 -0
- rag_wright/capabilities/highlight_serve.py +142 -0
- rag_wright/capabilities/hybrid_search.py +65 -0
- rag_wright/capabilities/invoke.py +31 -0
- rag_wright/capabilities/jev_decision.py +38 -0
- rag_wright/capabilities/manifests.py +872 -0
- rag_wright/capabilities/okf_navigate.py +456 -0
- rag_wright/capabilities/parsing.py +286 -0
- rag_wright/capabilities/property_boosted_retrieval.py +125 -0
- rag_wright/capabilities/query_function_classifier.py +94 -0
- rag_wright/capabilities/query_understanding.py +109 -0
- rag_wright/capabilities/registry.py +262 -0
- rag_wright/capabilities/remote_encoders.py +94 -0
- rag_wright/capabilities/requirement_extraction.py +247 -0
- rag_wright/capabilities/reranking.py +123 -0
- rag_wright/capabilities/retrieval_core.py +126 -0
- rag_wright/capabilities/rlm_chunking.py +808 -0
- rag_wright/capabilities/rlm_synthesis.py +316 -0
- rag_wright/capabilities/scan_quality.py +136 -0
- rag_wright/capabilities/span_relevance_judgment.py +191 -0
- rag_wright/capabilities/vision_to_text.py +85 -0
- rag_wright/capabilities/vlm_ocr.py +85 -0
- rag_wright/contracts/__init__.py +6 -0
- rag_wright/contracts/chunk.py +79 -0
- rag_wright/contracts/compliance.py +303 -0
- rag_wright/contracts/contract_meta.py +27 -0
- rag_wright/contracts/extraction.py +130 -0
- rag_wright/contracts/function.py +167 -0
- rag_wright/contracts/function_routing.py +91 -0
- rag_wright/contracts/highlight.py +74 -0
- rag_wright/contracts/identifiers.py +153 -0
- rag_wright/contracts/jurisdiction.py +96 -0
- rag_wright/contracts/ontology.py +142 -0
- rag_wright/contracts/property.py +201 -0
- rag_wright/contracts/provenance.py +78 -0
- rag_wright/contracts/query_intent.py +53 -0
- rag_wright/contracts/span.py +76 -0
- rag_wright/contracts/value_match.py +84 -0
- rag_wright/corpus/__init__.py +0 -0
- rag_wright/corpus/canonicalize.py +116 -0
- rag_wright/corpus/cuad.py +153 -0
- rag_wright/corpus/cuad_ingestion.py +72 -0
- rag_wright/corpus/document_parser.py +299 -0
- rag_wright/corpus/edgar.py +231 -0
- rag_wright/corpus/gcs_ingestion.py +120 -0
- rag_wright/corpus/http.py +110 -0
- rag_wright/corpus/selection.py +152 -0
- rag_wright/mcp/__init__.py +11 -0
- rag_wright/mcp/compliance_server.py +299 -0
- rag_wright/mcp/intra_document_qa_server.py +170 -0
- rag_wright/mcp/relational_qa_server.py +171 -0
- rag_wright/mcp/session_store.py +64 -0
- rag_wright/mcp/typed_property_retrieval_server.py +191 -0
- rag_wright/models/__init__.py +8 -0
- rag_wright/models/profiles.py +331 -0
- rag_wright/models/seam.py +497 -0
- rag_wright/models/tag_structured.py +285 -0
- rag_wright/models/tracing.py +179 -0
- rag_wright/models/usage.py +102 -0
- rag_wright/okf/__init__.py +11 -0
- rag_wright/okf/compile.py +292 -0
- rag_wright/okf/document.py +47 -0
- rag_wright/okf/enrich.py +176 -0
- rag_wright/okf/links.py +190 -0
- rag_wright/okf/lint.py +105 -0
- rag_wright/ontology/__init__.py +6 -0
- rag_wright/ontology/_generated_template_meta.py +60 -0
- rag_wright/ontology/_generated_vocab.py +52 -0
- rag_wright/ontology/clause_template.py +964 -0
- rag_wright/ontology/codegen.py +84 -0
- rag_wright/ontology/compliance_bridge.ttl +186 -0
- rag_wright/ontology/contract_bridge.ttl +2685 -0
- rag_wright/ontology/contract_taxonomy.py +24 -0
- rag_wright/ontology/derive.py +58 -0
- rag_wright/ontology/loader.py +435 -0
- rag_wright/ontology/packs/ftc_16cfr255.ttl +29 -0
- rag_wright/ontology/registry.py +87 -0
- rag_wright/ontology/template_introspect.py +100 -0
- rag_wright/py.typed +0 -0
- rag_wright/reference/__init__.py +2 -0
- rag_wright/reference/compliance.py +41 -0
- rag_wright/reference/contract_seam.py +123 -0
- rag_wright/skills/__init__.py +7 -0
- rag_wright/skills/claim_extraction/SKILL.md +47 -0
- rag_wright/skills/claim_extraction/__init__.py +1 -0
- rag_wright/skills/claim_extraction/template.py +50 -0
- rag_wright/skills/compliance_judgment/SKILL.md +59 -0
- rag_wright/skills/corpus_ingest/SKILL.md +106 -0
- rag_wright/skills/extraction_semantic_judge/SKILL.md +51 -0
- rag_wright/skills/extraction_semantic_judge/__init__.py +1 -0
- rag_wright/skills/generation/SKILL.md +64 -0
- rag_wright/skills/generation/__init__.py +1 -0
- rag_wright/skills/generic_compliance_judgment/SKILL.md +58 -0
- rag_wright/skills/okf_navigate/SKILL.md +137 -0
- rag_wright/skills/requirement_extraction/SKILL.md +47 -0
- rag_wright/skills/requirement_extraction/__init__.py +1 -0
- rag_wright/skills/requirement_extraction/template.py +50 -0
- rag_wright/skills/rlm/SKILL.md +186 -0
- rag_wright/skills/rlm/__init__.py +31 -0
- rag_wright/skills/rlm/agent.py +292 -0
- rag_wright/skills/span_relevance_judgment/SKILL.md +67 -0
- rag_wright/skills/vision_to_text/SKILL.md +36 -0
- rag_wright/skills/vision_to_text/__init__.py +1 -0
- rag_wright/spans/__init__.py +1 -0
- rag_wright/spans/boundary.py +78 -0
- rag_wright/spans/clause_function_classifier.py +490 -0
- rag_wright/spans/clause_kg_extractor.py +337 -0
- rag_wright/spans/cuad_labels.py +81 -0
- rag_wright/spans/dim_classifier.py +158 -0
- rag_wright/spans/dim_fleet.json +411 -0
- rag_wright/spans/function_classifier.py +77 -0
- rag_wright/spans/function_families.py +62 -0
- rag_wright/spans/hybrid_classifier.py +103 -0
- rag_wright/spans/legalbert_classifier.py +83 -0
- rag_wright/spans/model_capabilities.py +107 -0
- rag_wright/spans/new_function_labels.py +111 -0
- rag_wright/spans/page_map.py +68 -0
- rag_wright/spans/property_extractor.py +365 -0
- rag_wright/spans/property_grounding.py +182 -0
- rag_wright/spans/reclassify.py +77 -0
- rag_wright/spans/scarce_function_labels.py +105 -0
- rag_wright/spans/segment.py +341 -0
- rag_wright/spans/semantic_judge.py +197 -0
- rag_wright/spans/symbolic_validation.py +131 -0
- rag_wright/spans/tag_clause_extractor.py +182 -0
- rag_wright/store/__init__.py +6 -0
- rag_wright/store/arcadedb.py +1135 -0
- rag_wright/store/chunk_text.py +66 -0
- rag_wright/store/seam.py +213 -0
- rag_wright/subgraphs/__init__.py +0 -0
- rag_wright/subgraphs/async_ingestion.py +204 -0
- rag_wright/subgraphs/compliance_check.py +1042 -0
- rag_wright/subgraphs/compliance_ingestion.py +306 -0
- rag_wright/subgraphs/contract_ingestion_pipeline.py +999 -0
- rag_wright/subgraphs/graph_extraction.py +102 -0
- rag_wright/subgraphs/intra_document_qa.py +328 -0
- rag_wright/subgraphs/observability.py +140 -0
- rag_wright/subgraphs/query_constraint_extraction.py +73 -0
- rag_wright/subgraphs/relational_qa.py +165 -0
- rag_wright/subgraphs/requirement_extraction.py +137 -0
- rag_wright/subgraphs/scaffold.py +65 -0
- rag_wright/subgraphs/semantic_chunking.py +183 -0
- rag_wright/subgraphs/typed_clause_extraction.py +172 -0
- rag_wright/subgraphs/typed_property_retrieval.py +278 -0
- rag_wright/util/__init__.py +1 -0
- rag_wright/util/concurrent.py +153 -0
- rag_wright/util/spacy_model.py +45 -0
- rag_wright-0.1.0.dist-info/METADATA +168 -0
- rag_wright-0.1.0.dist-info/RECORD +184 -0
- rag_wright-0.1.0.dist-info/WHEEL +4 -0
- rag_wright-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,100 @@
|
|
|
1
|
+
"""ADR-0066 P1b-1: introspect the hand-maintained clause extraction template (`clause_template.py`) into a
|
|
2
|
+
structured field spec -- the SHARED source used by both the bootstrap emitter (writes the specs into the ttl) and
|
|
3
|
+
the drift test (asserts the ttl captured them faithfully). One introspection, so emitter and test cannot diverge.
|
|
4
|
+
|
|
5
|
+
Keyed by TEMPLATE FIELD (`<Model>.<field>`), because the template is not a flat dimension->field map: it has
|
|
6
|
+
nested constraint models (CapConstraint/TemporalConstraint/Jurisdiction) and non-dimension fields (clause_type,
|
|
7
|
+
document_reference), and field names do not uniformly match dimension values (excepts<->carve_out).
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import enum
|
|
13
|
+
import typing
|
|
14
|
+
from dataclasses import dataclass, field
|
|
15
|
+
|
|
16
|
+
from pydantic import BaseModel
|
|
17
|
+
from pydantic_core import PydanticUndefined
|
|
18
|
+
|
|
19
|
+
from rag_wright.ontology import clause_template as ct
|
|
20
|
+
|
|
21
|
+
# The models whose fields make up the extraction template (root + nested constraint models).
|
|
22
|
+
_MODELS: tuple[tuple[str, type[BaseModel]], ...] = (
|
|
23
|
+
("Clause", ct.Clause),
|
|
24
|
+
("CapConstraint", ct.CapConstraint),
|
|
25
|
+
("TemporalConstraint", ct.TemporalConstraint),
|
|
26
|
+
("Jurisdiction", ct.Jurisdiction),
|
|
27
|
+
)
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
@dataclass(frozen=True)
|
|
31
|
+
class TemplateFieldSpec:
|
|
32
|
+
"""One field of the extraction template -- everything needed to regenerate it (P1b-2) + its knowledge."""
|
|
33
|
+
|
|
34
|
+
model: str
|
|
35
|
+
name: str
|
|
36
|
+
kind: str # scalar_enum | list_enum | list_str | optional_str | str | model_ref
|
|
37
|
+
default_token: str # required | none | list | enum:<value> | model_none
|
|
38
|
+
definition: str = "" # the field's LOOK-FOR description (verbatim; "" for the TODO gaps)
|
|
39
|
+
enum_class: str | None = None # the Enum class name (scalar_enum / list_enum)
|
|
40
|
+
model_ref: str | None = None # the nested model name (model_ref)
|
|
41
|
+
edge_label: str | None = None # the docling-graph edge label (nested refs)
|
|
42
|
+
max_length: int | None = None
|
|
43
|
+
examples: tuple[str, ...] = field(default_factory=tuple)
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def _enum_class(ann: typing.Any) -> type[enum.Enum] | None:
|
|
47
|
+
for a in [ann, *typing.get_args(ann)]:
|
|
48
|
+
for b in [a, *typing.get_args(a)]:
|
|
49
|
+
if isinstance(b, type) and issubclass(b, enum.Enum):
|
|
50
|
+
return b
|
|
51
|
+
return None
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def _model_class(ann: typing.Any) -> type[BaseModel] | None:
|
|
55
|
+
for a in [ann, *typing.get_args(ann)]:
|
|
56
|
+
if isinstance(a, type) and issubclass(a, BaseModel):
|
|
57
|
+
return a
|
|
58
|
+
return None
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def _spec(model_name: str, name: str, f: typing.Any) -> TemplateFieldSpec:
|
|
62
|
+
ann = f.annotation
|
|
63
|
+
enum_cls = _enum_class(ann)
|
|
64
|
+
model_cls = _model_class(ann)
|
|
65
|
+
is_list = typing.get_origin(ann) is list
|
|
66
|
+
js = f.json_schema_extra if isinstance(f.json_schema_extra, dict) else {}
|
|
67
|
+
max_length = next((getattr(m, "max_length", None) for m in (f.metadata or [])
|
|
68
|
+
if getattr(m, "max_length", None) is not None), None)
|
|
69
|
+
|
|
70
|
+
if model_cls is not None:
|
|
71
|
+
kind, default_token = "model_ref", "model_none"
|
|
72
|
+
elif is_list and enum_cls is not None:
|
|
73
|
+
kind, default_token = "list_enum", "list"
|
|
74
|
+
elif is_list: # issue 0037: List[str] -- an OPEN descriptive list dim (verbatim capture, canonicalized at KG)
|
|
75
|
+
kind, default_token = "list_str", "list"
|
|
76
|
+
elif enum_cls is not None:
|
|
77
|
+
kind = "scalar_enum"
|
|
78
|
+
default_token = f"enum:{f.default.value}" if isinstance(f.default, enum.Enum) else "required"
|
|
79
|
+
else: # str / Optional[str]
|
|
80
|
+
kind = "optional_str" if typing.get_origin(ann) is typing.Union else "str"
|
|
81
|
+
default_token = "required" if f.default is PydanticUndefined else "none"
|
|
82
|
+
|
|
83
|
+
return TemplateFieldSpec(
|
|
84
|
+
model=model_name, name=name, kind=kind, default_token=default_token,
|
|
85
|
+
definition=f.description or "",
|
|
86
|
+
enum_class=enum_cls.__name__ if enum_cls is not None else None,
|
|
87
|
+
model_ref=model_cls.__name__ if model_cls is not None else None,
|
|
88
|
+
edge_label=js.get("edge_label"),
|
|
89
|
+
max_length=max_length,
|
|
90
|
+
examples=tuple(f.examples or ()),
|
|
91
|
+
)
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def introspect_template_fields() -> list[TemplateFieldSpec]:
|
|
95
|
+
"""The template's fields as structured specs (stable order: model order, then field-declaration order)."""
|
|
96
|
+
specs: list[TemplateFieldSpec] = []
|
|
97
|
+
for model_name, model in _MODELS:
|
|
98
|
+
for name, f in model.model_fields.items():
|
|
99
|
+
specs.append(_spec(model_name, name, f))
|
|
100
|
+
return specs
|
rag_wright/py.typed
ADDED
|
File without changes
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
"""EP-REF-1c: thin REFERENCE invoker wrappers for the GENERIC compliance leg -- worked examples showing a product
|
|
2
|
+
the exact call shape: open a workspace, then invoke the capability BY NAME via `ainvoke_subgraph`. Each is a
|
|
3
|
+
one-liner over the engine API -- no store/embedder/model-id/id-parsing. Reference-ONLY: a product owns the
|
|
4
|
+
FTC-routing / ad-compliance variants (`run_ad_compliance_check`) and its guardrails/human-gate policy.
|
|
5
|
+
"""
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
from typing import Any, Optional
|
|
9
|
+
|
|
10
|
+
from rag_wright.api import ainvoke_subgraph
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
async def invoke_compliance_check(ws: Any, *, subject_text: str, source_doc: str, k: int = 8,
|
|
14
|
+
sources: Optional[list[str]] = None):
|
|
15
|
+
"""Check a subject TEXT against the Requirement KG -> a `ComplianceReport` (cited findings + gap matrix).
|
|
16
|
+
`sources` scopes to named policies (None = store-wide)."""
|
|
17
|
+
return await ainvoke_subgraph(
|
|
18
|
+
"compliance_check",
|
|
19
|
+
{"subject_text": subject_text, "source_doc": source_doc, "k": k, "sources": sources},
|
|
20
|
+
resources=ws)
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
async def invoke_document_check(ws: Any, *, doc_name: str, data: bytes, k: int = 8,
|
|
24
|
+
sources: Optional[list[str]] = None):
|
|
25
|
+
"""Check a subject DOCUMENT (raw bytes: PDF/DOCX/HTML/TXT) against the Requirement KG -> a `ComplianceReport`."""
|
|
26
|
+
return await ainvoke_subgraph(
|
|
27
|
+
"compliance_check",
|
|
28
|
+
{"doc_name": doc_name, "data": data, "k": k, "sources": sources},
|
|
29
|
+
resources=ws)
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
async def invoke_policy_ingest(ws: Any, *, source: str, sections_path: Any = None,
|
|
33
|
+
doc_name: Optional[str] = None, data: Optional[bytes] = None):
|
|
34
|
+
"""Ingest a policy into the Requirement KG -> an `IngestionReport`. Give EITHER `sections_path` (a pre-sectioned
|
|
35
|
+
eCFR-style `sections.json`) OR `doc_name` + `data` (a policy DOCUMENT's raw bytes, split at its headings)."""
|
|
36
|
+
inputs: dict = {"source": source}
|
|
37
|
+
if data is not None:
|
|
38
|
+
inputs |= {"doc_name": doc_name, "data": data}
|
|
39
|
+
else:
|
|
40
|
+
inputs["sections_path"] = sections_path
|
|
41
|
+
return await ainvoke_subgraph("compliance_ingestion", inputs, resources=ws)
|
|
@@ -0,0 +1,123 @@
|
|
|
1
|
+
"""EP-REF-1d: a REFERENCE product seam for the engine's CONTRACT/COMPLIANCE reference domain -- a worked example
|
|
2
|
+
of what a product seam looks like AFTER the domain-agnostic separation. It is the shape EP-SEAM-3 refactors
|
|
3
|
+
RuleWright's real `engine/seam.py` toward.
|
|
4
|
+
|
|
5
|
+
Every method is a thin composition over the engine: `open_workspace` (tenancy = one call), the invokers
|
|
6
|
+
(`ainvoke_subgraph`), the generic API reads (`entities_by_name`), and the reference-pack store extensions
|
|
7
|
+
(`ContractKGStore` / `ComplianceStore`) + the compliance invoker wrappers. It holds NO `ArcadeDBStore`, embedder,
|
|
8
|
+
model id, or id-string parsing -- those are engine calls.
|
|
9
|
+
|
|
10
|
+
This is a worked EXAMPLE, deliberately thin. A real product adds, around these calls, the concerns marked
|
|
11
|
+
`# PRODUCT OWNS:` below -- tenancy policy, scoping (`ScopeViolation`), the FTC/ad-compliance variants, the
|
|
12
|
+
unknown-policy guard, presentation/citation types, caching, and routing the engine's usage/progress to its own
|
|
13
|
+
telemetry. Those stay product-side; see `docs/product/seam-adaptation-guide.md`.
|
|
14
|
+
"""
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
from typing import Any, Optional
|
|
18
|
+
|
|
19
|
+
from rag_wright.api import (
|
|
20
|
+
EngineConfig,
|
|
21
|
+
ainvoke_subgraph,
|
|
22
|
+
aparse_document,
|
|
23
|
+
entities_by_name,
|
|
24
|
+
open_workspace,
|
|
25
|
+
source_document,
|
|
26
|
+
)
|
|
27
|
+
from rag_wright.capabilities.compliance_store import ComplianceStore
|
|
28
|
+
from rag_wright.capabilities.contract_kg_store import ContractKGStore
|
|
29
|
+
from rag_wright.reference.compliance import invoke_compliance_check, invoke_policy_ingest
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
class ContractComplianceSeam:
|
|
33
|
+
"""A thin reference seam over the engine for the contract/compliance reference domain."""
|
|
34
|
+
|
|
35
|
+
def __init__(self, config: EngineConfig) -> None:
|
|
36
|
+
self._config = config
|
|
37
|
+
|
|
38
|
+
# --- tenancy -------------------------------------------------------------------------------------
|
|
39
|
+
def open(self, corpus: str, *, reset: bool = False):
|
|
40
|
+
"""Open a workspace for a corpus (one engine call). PRODUCT OWNS: per-tenant corpus selection + auth policy."""
|
|
41
|
+
return open_workspace(self._config, corpus=corpus, reset=reset)
|
|
42
|
+
|
|
43
|
+
# --- ingestion + query (compose the invokers) ----------------------------------------------------
|
|
44
|
+
async def ingest_contract(self, ws, doc_id: str, *, cache_dir: str, text: Optional[str] = None,
|
|
45
|
+
path: Any = None, metadata: Optional[dict] = None):
|
|
46
|
+
"""Ingest one contract: build a `SourceDocument` (docling-parse a file PATH, or wrap TEXT) then run the
|
|
47
|
+
ingestion capability. Returns the engine's `IngestionReport`. PRODUCT OWNS: the document source + progress UI
|
|
48
|
+
(the engine emits progress/usage; route it to your telemetry)."""
|
|
49
|
+
sd = (await aparse_document(doc_id, path, cache_dir=cache_dir, metadata=metadata) if path is not None
|
|
50
|
+
else source_document(doc_id, text=text or ""))
|
|
51
|
+
return await ainvoke_subgraph("contract_ingestion_pipeline",
|
|
52
|
+
{"document": sd, "cache_dir": cache_dir}, resources=ws)
|
|
53
|
+
|
|
54
|
+
async def ask_contract(self, ws, contract_id: str, question: str):
|
|
55
|
+
"""Single-document Q&A over one contract. Returns the engine's cited answer (raw -- PRODUCT OWNS: what
|
|
56
|
+
counts as 'answered' / how to render the citations)."""
|
|
57
|
+
return await ainvoke_subgraph("intra_document_qa",
|
|
58
|
+
{"contract_id": contract_id, "question": question}, resources=ws)
|
|
59
|
+
|
|
60
|
+
async def search_corpus(self, ws, query: str, *, k: int = 8, documents: Optional[list[str]] = None):
|
|
61
|
+
"""Corpus-wide typed/similarity retrieval (Leg B). The leg extracts the query's typed constraints itself;
|
|
62
|
+
pass `documents` to scope to a workspace's docs. PRODUCT OWNS: ranking/selection presentation."""
|
|
63
|
+
return await ainvoke_subgraph("typed_property_retrieval",
|
|
64
|
+
{"query": query, "k": k, "documents": documents}, resources=ws)
|
|
65
|
+
|
|
66
|
+
# --- relational / terms / citations (reference-pack store extensions over the workspace store) ----
|
|
67
|
+
def find_party(self, ws, name: str) -> list[tuple[str, str]]:
|
|
68
|
+
"""A party NAME -> every `(entity_id, stored_name)` it resolves to (engine-normalized; one name can match
|
|
69
|
+
several nodes -- all are returned, never the first only)."""
|
|
70
|
+
return sorted((str(e["entity_id"]), str(e.get("name") or "")) for e in entities_by_name(ws, name))
|
|
71
|
+
|
|
72
|
+
def counterparties(self, ws, entity_id: str, *, max_hops: int = 1, documents: Optional[list[str]] = None):
|
|
73
|
+
"""The parties this one has a CONTRACTS_WITH edge to (one hop by default)."""
|
|
74
|
+
return ContractKGStore(ws._store).party_counterparties(entity_id, max_hops=max_hops, documents=documents)
|
|
75
|
+
|
|
76
|
+
def affiliates(self, ws, entity_id: str, *, documents: Optional[list[str]] = None):
|
|
77
|
+
"""The parties this one has an AFFILIATE_OF edge to (same corporate group -- a separate traversal)."""
|
|
78
|
+
return ContractKGStore(ws._store).party_affiliates(entity_id, documents=documents)
|
|
79
|
+
|
|
80
|
+
def contract_terms(self, ws, contract_id: str) -> list:
|
|
81
|
+
"""The typed clauses of one contract (the full view -- keeps AMBIGUOUS out-of-vocab values)."""
|
|
82
|
+
return ContractKGStore(ws._store).contract_terms(contract_id)
|
|
83
|
+
|
|
84
|
+
def span_locations(self, ws, contract_id: str) -> list:
|
|
85
|
+
"""Every span's position (pages/bbox/offsets) + the clause ids on it -- for a citation preview."""
|
|
86
|
+
return ContractKGStore(ws._store).span_locations(contract_id)
|
|
87
|
+
|
|
88
|
+
@staticmethod
|
|
89
|
+
def canonical_clause_type(label: str) -> Optional[str]:
|
|
90
|
+
"""Map a user's clause label onto the taxonomy (alias-resolving), or None -- to validate a correction."""
|
|
91
|
+
return ContractKGStore.canonical_clause_type(label)
|
|
92
|
+
|
|
93
|
+
@staticmethod
|
|
94
|
+
def clause_type_vocabulary() -> tuple[str, ...]:
|
|
95
|
+
"""Every clause type a sweep/correction UI can be scoped to."""
|
|
96
|
+
return ContractKGStore.clause_type_vocabulary()
|
|
97
|
+
|
|
98
|
+
# --- compliance (reference invoker wrappers + the read facade) -----------------------------------
|
|
99
|
+
async def ingest_policy(self, ws, *, source: str, sections_path: Any = None,
|
|
100
|
+
doc_name: Optional[str] = None, data: Optional[bytes] = None):
|
|
101
|
+
"""Ingest a policy into the Requirement KG (sections.json OR a document's bytes). Returns an
|
|
102
|
+
`IngestionReport`. PRODUCT OWNS: the unknown-policy guard / curation workflow."""
|
|
103
|
+
return await invoke_policy_ingest(ws, source=source, sections_path=sections_path,
|
|
104
|
+
doc_name=doc_name, data=data)
|
|
105
|
+
|
|
106
|
+
async def check(self, ws, *, subject_text: str, source_doc: str, k: int = 8,
|
|
107
|
+
sources: Optional[list[str]] = None):
|
|
108
|
+
"""Check a subject against the Requirement KG -> a `ComplianceReport` (the GENERIC verdict). PRODUCT OWNS:
|
|
109
|
+
the FTC/ad-compliance tuned variant (`run_ad_compliance_check`), the unknown-policy guard, the human gate."""
|
|
110
|
+
return await invoke_compliance_check(ws, subject_text=subject_text, source_doc=source_doc,
|
|
111
|
+
k=k, sources=sources)
|
|
112
|
+
|
|
113
|
+
def requirements_for(self, ws, source: str) -> list[dict]:
|
|
114
|
+
"""The curated requirement rows under one policy (the proof an ingest landed)."""
|
|
115
|
+
return ComplianceStore(ws._store).requirements_for(source)
|
|
116
|
+
|
|
117
|
+
def requirement_locations(self, ws, source: str) -> list:
|
|
118
|
+
"""Where each requirement of one policy sits in its document (pages + bbox) -- for a citation preview."""
|
|
119
|
+
return ComplianceStore(ws._store).requirement_locations(source)
|
|
120
|
+
|
|
121
|
+
def curated_requirement_count(self, ws, sources: Optional[list[str]] = None) -> int:
|
|
122
|
+
"""How many requirements are in scope -- the denominator of an honest coverage statement."""
|
|
123
|
+
return ComplianceStore(ws._store).curated_requirement_count(sources)
|
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
"""Authored skill content built as ordinary software.
|
|
2
|
+
|
|
3
|
+
Holds the RLM skill (FR-C.10): an authored SKILL.md teaching the divide-and-conquer method
|
|
4
|
+
(load a working set into an interpreter, slice and dispatch the work in code, synthesize the
|
|
5
|
+
results). Used by the RLM chunking capability (ingestion) and the RLM synthesis capability
|
|
6
|
+
(query). It is not provided by any build tool and is not a compiler feature in this repo.
|
|
7
|
+
"""
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: claim_extraction
|
|
3
|
+
description: >
|
|
4
|
+
The subject-document claim-extraction method: read an advertisement and pull out its distinct CHECKABLE
|
|
5
|
+
claims (the assertions a regulator could test), each with its kind, the disclosures present near it, and
|
|
6
|
+
whether the ad references evidence. Applied by the compliance_check subgraph (subject side). The schema is
|
|
7
|
+
the co-located asset `template.py` (ExtractedAd / ExtractedClaim); the deterministic mapping to the closed
|
|
8
|
+
Claim vocab is the claim_adaptation FUNCTION's job, not this skill's.
|
|
9
|
+
---
|
|
10
|
+
|
|
11
|
+
# Claim extraction: what checkable claims does this ad make?
|
|
12
|
+
|
|
13
|
+
This skill teaches a **method**, not a behavior. It turns a subject advertisement into the checkable claims the
|
|
14
|
+
compliance check will judge. It is authored software (an Agent Skill), with its extraction **schema** as the
|
|
15
|
+
co-located asset **`template.py`** (`ExtractedAd` → `ExtractedClaim[]`), referenced here and filled by the model.
|
|
16
|
+
|
|
17
|
+
## What to extract (the schema — `template.py`)
|
|
18
|
+
|
|
19
|
+
Per ad, produce an `ExtractedAd` whose `claims` are the distinct **checkable** assertions. For each claim:
|
|
20
|
+
|
|
21
|
+
- **assertion_text** — the claim itself, quoted or closely paraphrased; one claim per entry.
|
|
22
|
+
- **claim_type** — the kind, from the closed vocab: `efficacy, comparative, pricing, health, environmental,
|
|
23
|
+
endorsement, performance, guarantee`.
|
|
24
|
+
- **disclosures_present** — any disclaimers/qualifiers near the claim: `#ad`, `paid partnership`, `results vary`.
|
|
25
|
+
- **evidence_referenced** — whether the ad points to a study/data for the claim.
|
|
26
|
+
- **actor / subject_product / quantitative_value / medium** — when present.
|
|
27
|
+
|
|
28
|
+
Extract the **checkable** assertions (a regulator could test them), not pure subjective flourish — but when in
|
|
29
|
+
doubt, extract it; the judgment step decides puffery vs objective claim.
|
|
30
|
+
|
|
31
|
+
## The reliability method (docling-graph, from GP-1B)
|
|
32
|
+
|
|
33
|
+
The extractor runs through docling-graph in API mode. Three settings are load-bearing and must not drift:
|
|
34
|
+
|
|
35
|
+
1. **source must be a file path**, not a raw string — docling-graph `stat()`s it (write the text to a temp file).
|
|
36
|
+
2. **`structured_output=False`** (json_object) — the strict nested json_schema returns nothing on some models;
|
|
37
|
+
json_object is reliable across DeepSeek / Gemma / Granite.
|
|
38
|
+
3. **a `max_tokens` cap** — an unknown provider else gets a generic 8192 context window and SKIPS the LLM.
|
|
39
|
+
|
|
40
|
+
Ads are short, so `extraction_contract="direct"` (one call) is correct here.
|
|
41
|
+
|
|
42
|
+
## What this skill does NOT own (the applying capability's job)
|
|
43
|
+
|
|
44
|
+
- mapping `claim_type` to the closed `ClaimType` vocab and coercing an off-vocab value (kept-but-AMBIGUOUS, a
|
|
45
|
+
checkable assertion is never dropped) — the `claim_adaptation` FUNCTION;
|
|
46
|
+
- the `claim_id` content-hash identity and the span provenance — the FUNCTION;
|
|
47
|
+
- concurrency, timeouts, and the compliance judgment that follows — the subgraph.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""claim_extraction agent-skill folder: SKILL.md + the template.py schema asset."""
|
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
"""Schema ASSET for the `claim_extraction` agent skill (referenced by this folder's SKILL.md).
|
|
2
|
+
|
|
3
|
+
The docling-graph extraction template the skill fills: `ExtractedAd` (the subject ad) with its checkable
|
|
4
|
+
`ExtractedClaim`s. Loose strings by design (robust to model output); the deterministic `claim_adaptation`
|
|
5
|
+
FUNCTION maps them to the closed CC-1 `Claim` vocab. Co-located with the skill because the schema IS part of
|
|
6
|
+
the authored extraction method (an Agent-Skill asset), not a hidden implementation detail.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from pydantic import BaseModel, ConfigDict, Field
|
|
12
|
+
|
|
13
|
+
from rag_wright.capabilities.dg_extraction import edge
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class ExtractedClaim(BaseModel):
|
|
17
|
+
"""One checkable assertion the LLM reads out of a subject ad (a docling-graph child entity)."""
|
|
18
|
+
|
|
19
|
+
model_config = ConfigDict(graph_id_fields=["assertion_text"], extra="ignore", populate_by_name=True)
|
|
20
|
+
|
|
21
|
+
assertion_text: str = Field(
|
|
22
|
+
description="One checkable factual claim the ad makes, quoted or closely paraphrased (one claim per entry)")
|
|
23
|
+
claim_type: str = Field(
|
|
24
|
+
default="",
|
|
25
|
+
description=("The kind of claim, chosen from: efficacy, comparative, pricing, health, environmental, "
|
|
26
|
+
"endorsement, performance, guarantee"))
|
|
27
|
+
actor: str = Field(default="", description=(
|
|
28
|
+
"DEON-8: the ROLE of the party this claim involves -- a role word, NOT a person's or company's name. "
|
|
29
|
+
"Choose the general role: advertiser, endorser, expert, manufacturer, seller. (E.g. 'Dr. Miller "
|
|
30
|
+
"recommends ...' -> endorser, not 'Dr. Miller'.) Empty if no clear actor."))
|
|
31
|
+
subject_product: str = Field(default="", description="The product or brand the claim is about")
|
|
32
|
+
quantitative_value: str = Field(
|
|
33
|
+
default="", description="Any specific number/quantity claimed, e.g. '30 pounds in one month', '2x faster'")
|
|
34
|
+
disclosures_present: list[str] = Field(
|
|
35
|
+
default_factory=list,
|
|
36
|
+
description="Disclaimers/qualifiers present near the claim, e.g. '#ad', 'paid partnership', 'results vary'")
|
|
37
|
+
evidence_referenced: bool = Field(
|
|
38
|
+
default=False, description="Whether the ad references evidence/substantiation for the claim (a study, data)")
|
|
39
|
+
medium: str = Field(default="", description="The medium, e.g. social, tv, print, podcast, web")
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
class ExtractedAd(BaseModel):
|
|
43
|
+
"""The subject document and the distinct checkable claims it makes (the docling-graph root entity)."""
|
|
44
|
+
|
|
45
|
+
model_config = ConfigDict(graph_id_fields=["subject"], extra="ignore", populate_by_name=True)
|
|
46
|
+
|
|
47
|
+
subject: str = Field(description="A short label for the subject ad (the brand/product or a headline phrase)")
|
|
48
|
+
claims: list[ExtractedClaim] = edge(
|
|
49
|
+
"MAKES_CLAIM", default_factory=list,
|
|
50
|
+
description="The distinct checkable claims the ad makes (one entry per claim)")
|
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: compliance_judgment
|
|
3
|
+
description: >
|
|
4
|
+
The advertising-compliance judgment method: given ONE advertising claim and ONE applicable regulatory
|
|
5
|
+
requirement, and seeing only the ad text (never the advertiser's evidence files), decide whether the claim
|
|
6
|
+
clearly violates the requirement, clearly satisfies it, or cannot be judged from the text and must be
|
|
7
|
+
escalated for human review. Applied by the compliance_check subgraph (query side). The applying capability
|
|
8
|
+
owns the deterministic guarantees (verdict vocab, conservative default, both-sided citation) -- this skill
|
|
9
|
+
teaches only the reading.
|
|
10
|
+
---
|
|
11
|
+
|
|
12
|
+
# Compliance judgment: does this claim satisfy or violate this requirement?
|
|
13
|
+
|
|
14
|
+
This skill teaches a **method**, not a behavior. It extends the grounding-judge idea (ADR-0028) from
|
|
15
|
+
"is X supported by cue Y?" to "does claim X satisfy or violate requirement Y?". A capability applies it with
|
|
16
|
+
its own contract (`ComplianceFinding`) and its own guarantees; those guarantees are the **applying
|
|
17
|
+
capability's** job, not the method's (see "What this skill does NOT own").
|
|
18
|
+
|
|
19
|
+
## The one hard constraint: you see only the ad text
|
|
20
|
+
|
|
21
|
+
You are given the requirement and the claim. You **cannot** see the advertiser's studies, substantiation
|
|
22
|
+
files, or evidence. So you can only judge what the *text itself* shows. This constraint is the whole reason the
|
|
23
|
+
verdict is three-way, not two-way.
|
|
24
|
+
|
|
25
|
+
## The three verdicts
|
|
26
|
+
|
|
27
|
+
- **violation** — reserve this for what is **clearly wrong from the text itself**:
|
|
28
|
+
1. the claim **overclaims proof** — it asserts it is "clinically proven", "scientifically proven", "science
|
|
29
|
+
backed", "doctor proven", or "guaranteed" **without pointing to an actual study or data** (the
|
|
30
|
+
proof-language *is* the unsubstantiated claim, not evidence for it);
|
|
31
|
+
2. an endorsement is **missing a required disclosure** — no "#ad" / "paid partnership" is present when a
|
|
32
|
+
material connection would need disclosing;
|
|
33
|
+
3. a review or testimonial is **fake or deceptive**.
|
|
34
|
+
|
|
35
|
+
- **needs_review** — an **objective** efficacy / health / performance / factual claim that may well be true, but
|
|
36
|
+
the ad shows **no evidence** and makes **no overclaim**. You cannot verify its substantiation from the text
|
|
37
|
+
alone, so **escalate** it: a human will check the advertiser's substantiation file. Do **not** call this a
|
|
38
|
+
violation (you don't have the evidence), and do **not** clear it as compliant (you can't confirm it either).
|
|
39
|
+
|
|
40
|
+
- **compliant** — the requirement does not bite, because one of:
|
|
41
|
+
- the claim is mere **subjective opinion or taste/experience puffery** ("smooth flavor", "relaxing", "I like
|
|
42
|
+
it") — there is nothing objective to substantiate;
|
|
43
|
+
- the required **disclosure is present** (see the claim's `disclosures_present`, e.g. "#ad");
|
|
44
|
+
- the ad **actually points to real evidence** (a specific study / data / citation) for the objective claim;
|
|
45
|
+
- there is **no objective claim** to substantiate.
|
|
46
|
+
|
|
47
|
+
## The discipline
|
|
48
|
+
|
|
49
|
+
Only say **violation** when the text clearly shows the breach; only say **compliant** when the text clearly
|
|
50
|
+
clears it; **otherwise `needs_review`**. Uncertainty is escalation, never a silent pass. Give a one-sentence
|
|
51
|
+
rationale and a confidence in [0,1].
|
|
52
|
+
|
|
53
|
+
## What this skill does NOT own (the applying capability's job)
|
|
54
|
+
|
|
55
|
+
- the verdict **vocabulary** and the **conservative default** — an unreadable or missing verdict maps to
|
|
56
|
+
`needs_review` deterministically, in the capability, not here;
|
|
57
|
+
- the **both-sided citation** — the exact claim span and the exact requirement clause are attached from the
|
|
58
|
+
INPUTS by the capability, never authored by this skill (the model rules; it never fabricates a citation);
|
|
59
|
+
- the **ad-level rollup** (how many findings make an ad a violation) and the **human gate**.
|
|
@@ -0,0 +1,106 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: corpus_ingest
|
|
3
|
+
description: >
|
|
4
|
+
The repeatable method for ingesting ANY new corpus of contracts into the one contract KG and
|
|
5
|
+
auto-connecting it (parties <-> clauses). Point the GENERIC ingestion pipeline
|
|
6
|
+
(contract_ingestion_pipeline) at the corpus through a single thin CorpusAdapter -- never a
|
|
7
|
+
re-implemented ingest_xyz(). Write the adapter (parse + canonical source_doc_id + optional metadata),
|
|
8
|
+
point the entity registry at the corpus's parties, and run run_corpus_ingestion; the pipeline chunks,
|
|
9
|
+
segments, function-classifies, extracts clauses + the party graph, resolves entities, and writes both. Applied over src/rag_wright/subgraphs/contract_ingestion_pipeline.py.
|
|
10
|
+
---
|
|
11
|
+
|
|
12
|
+
# Ingesting a new corpus into the contract KG
|
|
13
|
+
|
|
14
|
+
This skill teaches a **method**, not a behavior. The design that makes it cheap: ONE generic, corpus-agnostic
|
|
15
|
+
ingestion pipeline (LG-3d, `contract_ingestion_pipeline`) plus a thin per-corpus **`CorpusAdapter`**. Adding a
|
|
16
|
+
corpus is one adapter — **never** a re-implemented `ingest_xyz()` that duplicates the flow. See ADR-0033
|
|
17
|
+
(unified KG), HYG-1 (canonical identity), ADR-0037 (the clause template is authoritative code), and
|
|
18
|
+
`docs/corpus_ingest_recipe.md` (the prose recipe this skill formalizes).
|
|
19
|
+
|
|
20
|
+
## The method: one generic pipeline + one thin adapter
|
|
21
|
+
|
|
22
|
+
The pipeline is fixed and shared. Everything corpus-specific lives behind one seam,
|
|
23
|
+
`CorpusAdapter.documents() -> Iterable[SourceDocument]`. `SourceDocument` is `{source_doc_id, text, metadata}`.
|
|
24
|
+
The pipeline, per document, runs: **chunk (semantic_chunking) → segment → LegalBERT function-classify →
|
|
25
|
+
clause-extract ∥ graph-extract → entity_resolution → write (clause KG + entity graph)**. (The KG-7
|
|
26
|
+
`party_clause_linking`/PartyTo post-step was retired — issue 0028 / ADR-0091 — since party→clause is reached
|
|
27
|
+
via CONTRACTS_WITH provenance + the contract-scoped clause KG.)
|
|
28
|
+
|
|
29
|
+
## A new *contract* corpus — 3 steps
|
|
30
|
+
|
|
31
|
+
### 1. Write one `CorpusAdapter` — the ONLY new code
|
|
32
|
+
`documents()` owns everything corpus-specific:
|
|
33
|
+
- enumerate the corpus's files/records;
|
|
34
|
+
- **parse each to text** (PDF → docling; JSON → read; …) — parsing lives here, so the pipeline is parse-agnostic;
|
|
35
|
+
- assign the id via `canonical_source_doc_id(...)` (HYG-1) — this is **load-bearing**: it is what lets the new
|
|
36
|
+
corpus's clauses, spans, entities, and contracts share ONE id scheme and auto-connect. A slug that disagrees
|
|
37
|
+
with the rest (e.g. `-` for spaces instead of `_`) silently breaks the cross-graph join;
|
|
38
|
+
- optionally attach corpus quirks on `SourceDocument.metadata` (annotated parties, pre-segmented spans, …).
|
|
39
|
+
|
|
40
|
+
`CuadAdapter` (in `contract_ingestion_pipeline.py`) is the reference implementation.
|
|
41
|
+
|
|
42
|
+
### 2. Point the entity registry at the corpus's parties
|
|
43
|
+
Extend the EDGAR verified registry (`build_verified_registry`) for the corpus's public companies, or accept
|
|
44
|
+
`UNLINKED` / `PRIVATE` for parties not in the registry (the honest closed-world gap).
|
|
45
|
+
|
|
46
|
+
### 3. Run it (monitored)
|
|
47
|
+
|
|
48
|
+
```python
|
|
49
|
+
from rag_wright.subgraphs.contract_ingestion_pipeline import (
|
|
50
|
+
arun_corpus_ingestion, aproduction_document_ingest,
|
|
51
|
+
)
|
|
52
|
+
|
|
53
|
+
report = await arun_corpus_ingestion(
|
|
54
|
+
YourAdapter(path, limit=N), # test on a FEW docs first; never a full re-ingest without intent
|
|
55
|
+
aproduction_document_ingest(store, cache_dir=..., registry=...),
|
|
56
|
+
)
|
|
57
|
+
# report: documents_ingested, dead_lettered (per-doc), per_document
|
|
58
|
+
```
|
|
59
|
+
|
|
60
|
+
The ingest **extraction models are caller-configurable** (like the query/compliance entrypoints; env vars stay
|
|
61
|
+
the fallback): `aproduction_document_ingest(store, cache_dir=..., registry=..., extract_model=..., list_model=...,
|
|
62
|
+
samples=...)`. `extract_model` is the primary clause-property model (an `ExtractionModel` or a bare model-id
|
|
63
|
+
string; default = the `RAG_SERVING` backend model, granite). `list_model` is the SECOND model for the cross-model
|
|
64
|
+
UNION on the LIST-bearing dims only (carve_out / covered_subject / damage_type) — granite and gemma
|
|
65
|
+
under-enumerate different list items, so their union is more complete; `"off"` disables it, default = gemma. If
|
|
66
|
+
you override `extract_model` (e.g. to qwen), set `list_model` deliberately — the union only helps if the two
|
|
67
|
+
models are complementary. `graph_extract_model` sets the party+affiliation extraction model (they share one),
|
|
68
|
+
`judge_model` the ingest semantic-judge model, and `chunk_model` the chunker's boundary-refinement model (used
|
|
69
|
+
only for over-cap sections). All accept a bare id or an `ExtractionModel` and default to their backend/env value,
|
|
70
|
+
so every ingest LLM surface (clause extract + list union, party/affiliation, judge, chunk boundary, and the
|
|
71
|
+
pre-existing `classify_fn`) is now a call-site argument.
|
|
72
|
+
|
|
73
|
+
`run_corpus_ingestion` streams `X/N` progress; a bad document dead-letters and is skipped (one bad doc never
|
|
74
|
+
kills the corpus). Write to a SCRATCH database first (`from_env(database=..., reset=True)`) to keep it
|
|
75
|
+
non-destructive while proving it out.
|
|
76
|
+
|
|
77
|
+
## Rules (not optional)
|
|
78
|
+
|
|
79
|
+
1. **Never write an `ingest_xyz()` that re-implements the flow.** A new corpus = one `CorpusAdapter`, then
|
|
80
|
+
`run_corpus_ingestion(adapter, ...)`. If you find yourself copying the pipeline, stop.
|
|
81
|
+
2. **The canonical `source_doc_id` is load-bearing.** Always derive it via `canonical_source_doc_id`; a
|
|
82
|
+
divergent slug breaks the cross-graph join (HYG-1). Every graph must share one id scheme.
|
|
83
|
+
3. **Extraction is concurrent, per-item tolerant, and cached.** Clause/graph extraction runs under
|
|
84
|
+
`map_concurrent` (granite is ~10s/single call); a truncated/failed span is SKIPPED, not fatal to the
|
|
85
|
+
document; and each successful extraction is cached by clause-id + template-schema-version, so a re-run or a
|
|
86
|
+
template change re-extracts only what it must.
|
|
87
|
+
4. **Long-running runs stream `X/N` and are actively monitored** (CLAUDE.md) — never launch-and-forget.
|
|
88
|
+
5. **The clause template is authoritative code, not regenerated** (ADR-0037); tune extraction by editing
|
|
89
|
+
`clause_template.py`, never by chasing a regeneration from the `.ttl`/spec.
|
|
90
|
+
|
|
91
|
+
## A new *domain* (non-contract)
|
|
92
|
+
The pipeline *structure* stays; the capabilities it binds change: a **new extraction template** (bootstrap a
|
|
93
|
+
fresh `.py` from a new ontology, then hand-maintain it — ADR-0037), a **retrained/replaced function classifier**
|
|
94
|
+
(new taxonomy), and possibly a different entity registry.
|
|
95
|
+
|
|
96
|
+
## What this skill does NOT own (deferred to the pipeline / capabilities)
|
|
97
|
+
- **The pipeline internals** (`build_document_ingest` graph, dead-letter, the extraction seams) — LG-3d.
|
|
98
|
+
- **The extraction capabilities** — semantic_chunking, the function classifier, clause extraction (docling-graph
|
|
99
|
+
+ granite, ADR-0037 template), GP-1B graph_extraction (ADR-0035), entity_resolution.
|
|
100
|
+
- **The span/embedding retrieval index** — wired into the pipeline as a parallel `index_spans` branch off the
|
|
101
|
+
shared `segment` node (INGEST-REFACTOR phase 2a); a corpus now gets the typed KG + entity graph + the
|
|
102
|
+
dense/sparse retrieval index in one pass. Indexing is best-effort (a failed index degrades to 0 spans, never
|
|
103
|
+
dead-letters the document's KG).
|
|
104
|
+
|
|
105
|
+
The method is: parse behind the adapter, one canonical id, run the generic pipeline, connect once. Keep this
|
|
106
|
+
file about that shape; the pipeline supplies the flow, the capabilities, and the tests.
|
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: extraction_semantic_judge
|
|
3
|
+
description: >
|
|
4
|
+
The clause-property faithfulness method: given ONE extracted property (a dimension = a value) and the clause
|
|
5
|
+
text it was extracted from, decide whether a careful reading of THIS clause genuinely SUPPORTS that property.
|
|
6
|
+
For the closed SEMANTIC dimensions (mutuality, favorability, party_asymmetry, cap_basis, the consent regimes)
|
|
7
|
+
whose value is a reading with no surface token, this is what the lexical and symbolic gates cannot reach.
|
|
8
|
+
Layer 3 of the neuro-symbolic extraction-fidelity cascade (ADR-0040). The applying capability owns the
|
|
9
|
+
deterministic guarantees (which dimensions are semantic, the AMBIGUOUS downgrade, concurrency) -- this skill
|
|
10
|
+
teaches only the reading.
|
|
11
|
+
---
|
|
12
|
+
|
|
13
|
+
# Extraction semantic judge: does this clause support this property?
|
|
14
|
+
|
|
15
|
+
This skill teaches a **method**, not a behavior. It extends the grounding-judge idea (ADR-0028) from
|
|
16
|
+
"is X supported by cue Y in the text?" to the harder **semantic** case: "does a faithful reading of THIS clause
|
|
17
|
+
support the property (dimension = value) that was extracted from it?" A capability applies it with its own
|
|
18
|
+
contract (`ClausePropertyRecord`) and its own guarantees; those guarantees are the **applying capability's**
|
|
19
|
+
job, not the method's (see "What this skill does NOT own").
|
|
20
|
+
|
|
21
|
+
## What you are judging
|
|
22
|
+
|
|
23
|
+
You are given one extracted **property** as `dimension = value` (with a short plain-language meaning of what
|
|
24
|
+
that property claims about the clause) and the **clause text**. These are the closed SEMANTIC dimensions:
|
|
25
|
+
`mutuality` (obligation runs both ways vs one), `favorability` (which side a term favors), `party_asymmetry`,
|
|
26
|
+
`cap_basis` (fixed_fee vs multiple_of_fees), the governing-law multiplicity, IP ownership, the non-solicit
|
|
27
|
+
target, the renewal mechanism, the change-of-control / assignment consent regimes, the MFN scope, the
|
|
28
|
+
termination right. Their value is a **reading** of the clause, not a token you can grep for -- which is exactly
|
|
29
|
+
why a model is spent here and nowhere else in the cascade.
|
|
30
|
+
|
|
31
|
+
## The one hard rule: strictness
|
|
32
|
+
|
|
33
|
+
Decide **supported=true** only if the clause **genuinely** supports the property. Decide **supported=false**
|
|
34
|
+
if the clause does not support it or contradicts it. Be strict:
|
|
35
|
+
|
|
36
|
+
- **mere plausibility is not support** -- that a mutual reading is *possible* is not enough; the clause must
|
|
37
|
+
actually bear it;
|
|
38
|
+
- **absence of support in this clause means supported=false** -- if the text is silent on what the property
|
|
39
|
+
asserts, it is not supported. Do not import world knowledge or the "usual" drafting; judge only THIS clause.
|
|
40
|
+
|
|
41
|
+
A `false` verdict means the reading is unsupported by the text; the applying capability will downgrade that
|
|
42
|
+
assertion to AMBIGUOUS (kept but flagged), never delete it. Give a one-sentence reason.
|
|
43
|
+
|
|
44
|
+
## What this skill does NOT own (the applying capability's job)
|
|
45
|
+
|
|
46
|
+
- **which dimensions are semantic** -- the set of dimensions this judge runs on (closed vocabulary, no lexical
|
|
47
|
+
cue) is selected deterministically by the capability, not decided here;
|
|
48
|
+
- the **AMBIGUOUS downgrade** and the "leave untouched on a judge error / no ruling" conservative rule --
|
|
49
|
+
applied by the capability, not by this method;
|
|
50
|
+
- the **per-assertion dispatch and concurrency** -- the capability runs this reading over each surviving
|
|
51
|
+
(non-AMBIGUOUS) semantic assertion; the method judges exactly one at a time.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""extraction_semantic_judge agent-skill folder: SKILL.md (the clause-property faithfulness method)."""
|