rag-wright 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (184) hide show
  1. rag_wright/__init__.py +13 -0
  2. rag_wright/api/__init__.py +33 -0
  3. rag_wright/api/config.py +59 -0
  4. rag_wright/api/discover.py +70 -0
  5. rag_wright/api/documents.py +39 -0
  6. rag_wright/api/ids.py +31 -0
  7. rag_wright/api/invoke.py +99 -0
  8. rag_wright/api/kg.py +61 -0
  9. rag_wright/api/mcp.py +94 -0
  10. rag_wright/api/usage.py +30 -0
  11. rag_wright/api/workspace.py +85 -0
  12. rag_wright/capabilities/__init__.py +8 -0
  13. rag_wright/capabilities/answer_generator.py +427 -0
  14. rag_wright/capabilities/ard.py +286 -0
  15. rag_wright/capabilities/assertion_extraction.py +79 -0
  16. rag_wright/capabilities/chunk_read.py +58 -0
  17. rag_wright/capabilities/chunk_write.py +163 -0
  18. rag_wright/capabilities/claim_extraction.py +153 -0
  19. rag_wright/capabilities/clause_exception_linking.py +117 -0
  20. rag_wright/capabilities/compliance_judgment.py +322 -0
  21. rag_wright/capabilities/compliance_store.py +87 -0
  22. rag_wright/capabilities/contract_kg_serve.py +156 -0
  23. rag_wright/capabilities/contract_kg_store.py +251 -0
  24. rag_wright/capabilities/dg_extraction.py +585 -0
  25. rag_wright/capabilities/disambiguation.py +163 -0
  26. rag_wright/capabilities/document_parse.py +87 -0
  27. rag_wright/capabilities/document_scope.py +49 -0
  28. rag_wright/capabilities/embedding.py +164 -0
  29. rag_wright/capabilities/embedding_profiles.py +43 -0
  30. rag_wright/capabilities/entity_resolution.py +154 -0
  31. rag_wright/capabilities/fusion.py +64 -0
  32. rag_wright/capabilities/graph_extraction.py +243 -0
  33. rag_wright/capabilities/graph_query.py +73 -0
  34. rag_wright/capabilities/graph_storage.py +111 -0
  35. rag_wright/capabilities/highlight_serve.py +142 -0
  36. rag_wright/capabilities/hybrid_search.py +65 -0
  37. rag_wright/capabilities/invoke.py +31 -0
  38. rag_wright/capabilities/jev_decision.py +38 -0
  39. rag_wright/capabilities/manifests.py +872 -0
  40. rag_wright/capabilities/okf_navigate.py +456 -0
  41. rag_wright/capabilities/parsing.py +286 -0
  42. rag_wright/capabilities/property_boosted_retrieval.py +125 -0
  43. rag_wright/capabilities/query_function_classifier.py +94 -0
  44. rag_wright/capabilities/query_understanding.py +109 -0
  45. rag_wright/capabilities/registry.py +262 -0
  46. rag_wright/capabilities/remote_encoders.py +94 -0
  47. rag_wright/capabilities/requirement_extraction.py +247 -0
  48. rag_wright/capabilities/reranking.py +123 -0
  49. rag_wright/capabilities/retrieval_core.py +126 -0
  50. rag_wright/capabilities/rlm_chunking.py +808 -0
  51. rag_wright/capabilities/rlm_synthesis.py +316 -0
  52. rag_wright/capabilities/scan_quality.py +136 -0
  53. rag_wright/capabilities/span_relevance_judgment.py +191 -0
  54. rag_wright/capabilities/vision_to_text.py +85 -0
  55. rag_wright/capabilities/vlm_ocr.py +85 -0
  56. rag_wright/contracts/__init__.py +6 -0
  57. rag_wright/contracts/chunk.py +79 -0
  58. rag_wright/contracts/compliance.py +303 -0
  59. rag_wright/contracts/contract_meta.py +27 -0
  60. rag_wright/contracts/extraction.py +130 -0
  61. rag_wright/contracts/function.py +167 -0
  62. rag_wright/contracts/function_routing.py +91 -0
  63. rag_wright/contracts/highlight.py +74 -0
  64. rag_wright/contracts/identifiers.py +153 -0
  65. rag_wright/contracts/jurisdiction.py +96 -0
  66. rag_wright/contracts/ontology.py +142 -0
  67. rag_wright/contracts/property.py +201 -0
  68. rag_wright/contracts/provenance.py +78 -0
  69. rag_wright/contracts/query_intent.py +53 -0
  70. rag_wright/contracts/span.py +76 -0
  71. rag_wright/contracts/value_match.py +84 -0
  72. rag_wright/corpus/__init__.py +0 -0
  73. rag_wright/corpus/canonicalize.py +116 -0
  74. rag_wright/corpus/cuad.py +153 -0
  75. rag_wright/corpus/cuad_ingestion.py +72 -0
  76. rag_wright/corpus/document_parser.py +299 -0
  77. rag_wright/corpus/edgar.py +231 -0
  78. rag_wright/corpus/gcs_ingestion.py +120 -0
  79. rag_wright/corpus/http.py +110 -0
  80. rag_wright/corpus/selection.py +152 -0
  81. rag_wright/mcp/__init__.py +11 -0
  82. rag_wright/mcp/compliance_server.py +299 -0
  83. rag_wright/mcp/intra_document_qa_server.py +170 -0
  84. rag_wright/mcp/relational_qa_server.py +171 -0
  85. rag_wright/mcp/session_store.py +64 -0
  86. rag_wright/mcp/typed_property_retrieval_server.py +191 -0
  87. rag_wright/models/__init__.py +8 -0
  88. rag_wright/models/profiles.py +331 -0
  89. rag_wright/models/seam.py +497 -0
  90. rag_wright/models/tag_structured.py +285 -0
  91. rag_wright/models/tracing.py +179 -0
  92. rag_wright/models/usage.py +102 -0
  93. rag_wright/okf/__init__.py +11 -0
  94. rag_wright/okf/compile.py +292 -0
  95. rag_wright/okf/document.py +47 -0
  96. rag_wright/okf/enrich.py +176 -0
  97. rag_wright/okf/links.py +190 -0
  98. rag_wright/okf/lint.py +105 -0
  99. rag_wright/ontology/__init__.py +6 -0
  100. rag_wright/ontology/_generated_template_meta.py +60 -0
  101. rag_wright/ontology/_generated_vocab.py +52 -0
  102. rag_wright/ontology/clause_template.py +964 -0
  103. rag_wright/ontology/codegen.py +84 -0
  104. rag_wright/ontology/compliance_bridge.ttl +186 -0
  105. rag_wright/ontology/contract_bridge.ttl +2685 -0
  106. rag_wright/ontology/contract_taxonomy.py +24 -0
  107. rag_wright/ontology/derive.py +58 -0
  108. rag_wright/ontology/loader.py +435 -0
  109. rag_wright/ontology/packs/ftc_16cfr255.ttl +29 -0
  110. rag_wright/ontology/registry.py +87 -0
  111. rag_wright/ontology/template_introspect.py +100 -0
  112. rag_wright/py.typed +0 -0
  113. rag_wright/reference/__init__.py +2 -0
  114. rag_wright/reference/compliance.py +41 -0
  115. rag_wright/reference/contract_seam.py +123 -0
  116. rag_wright/skills/__init__.py +7 -0
  117. rag_wright/skills/claim_extraction/SKILL.md +47 -0
  118. rag_wright/skills/claim_extraction/__init__.py +1 -0
  119. rag_wright/skills/claim_extraction/template.py +50 -0
  120. rag_wright/skills/compliance_judgment/SKILL.md +59 -0
  121. rag_wright/skills/corpus_ingest/SKILL.md +106 -0
  122. rag_wright/skills/extraction_semantic_judge/SKILL.md +51 -0
  123. rag_wright/skills/extraction_semantic_judge/__init__.py +1 -0
  124. rag_wright/skills/generation/SKILL.md +64 -0
  125. rag_wright/skills/generation/__init__.py +1 -0
  126. rag_wright/skills/generic_compliance_judgment/SKILL.md +58 -0
  127. rag_wright/skills/okf_navigate/SKILL.md +137 -0
  128. rag_wright/skills/requirement_extraction/SKILL.md +47 -0
  129. rag_wright/skills/requirement_extraction/__init__.py +1 -0
  130. rag_wright/skills/requirement_extraction/template.py +50 -0
  131. rag_wright/skills/rlm/SKILL.md +186 -0
  132. rag_wright/skills/rlm/__init__.py +31 -0
  133. rag_wright/skills/rlm/agent.py +292 -0
  134. rag_wright/skills/span_relevance_judgment/SKILL.md +67 -0
  135. rag_wright/skills/vision_to_text/SKILL.md +36 -0
  136. rag_wright/skills/vision_to_text/__init__.py +1 -0
  137. rag_wright/spans/__init__.py +1 -0
  138. rag_wright/spans/boundary.py +78 -0
  139. rag_wright/spans/clause_function_classifier.py +490 -0
  140. rag_wright/spans/clause_kg_extractor.py +337 -0
  141. rag_wright/spans/cuad_labels.py +81 -0
  142. rag_wright/spans/dim_classifier.py +158 -0
  143. rag_wright/spans/dim_fleet.json +411 -0
  144. rag_wright/spans/function_classifier.py +77 -0
  145. rag_wright/spans/function_families.py +62 -0
  146. rag_wright/spans/hybrid_classifier.py +103 -0
  147. rag_wright/spans/legalbert_classifier.py +83 -0
  148. rag_wright/spans/model_capabilities.py +107 -0
  149. rag_wright/spans/new_function_labels.py +111 -0
  150. rag_wright/spans/page_map.py +68 -0
  151. rag_wright/spans/property_extractor.py +365 -0
  152. rag_wright/spans/property_grounding.py +182 -0
  153. rag_wright/spans/reclassify.py +77 -0
  154. rag_wright/spans/scarce_function_labels.py +105 -0
  155. rag_wright/spans/segment.py +341 -0
  156. rag_wright/spans/semantic_judge.py +197 -0
  157. rag_wright/spans/symbolic_validation.py +131 -0
  158. rag_wright/spans/tag_clause_extractor.py +182 -0
  159. rag_wright/store/__init__.py +6 -0
  160. rag_wright/store/arcadedb.py +1135 -0
  161. rag_wright/store/chunk_text.py +66 -0
  162. rag_wright/store/seam.py +213 -0
  163. rag_wright/subgraphs/__init__.py +0 -0
  164. rag_wright/subgraphs/async_ingestion.py +204 -0
  165. rag_wright/subgraphs/compliance_check.py +1042 -0
  166. rag_wright/subgraphs/compliance_ingestion.py +306 -0
  167. rag_wright/subgraphs/contract_ingestion_pipeline.py +999 -0
  168. rag_wright/subgraphs/graph_extraction.py +102 -0
  169. rag_wright/subgraphs/intra_document_qa.py +328 -0
  170. rag_wright/subgraphs/observability.py +140 -0
  171. rag_wright/subgraphs/query_constraint_extraction.py +73 -0
  172. rag_wright/subgraphs/relational_qa.py +165 -0
  173. rag_wright/subgraphs/requirement_extraction.py +137 -0
  174. rag_wright/subgraphs/scaffold.py +65 -0
  175. rag_wright/subgraphs/semantic_chunking.py +183 -0
  176. rag_wright/subgraphs/typed_clause_extraction.py +172 -0
  177. rag_wright/subgraphs/typed_property_retrieval.py +278 -0
  178. rag_wright/util/__init__.py +1 -0
  179. rag_wright/util/concurrent.py +153 -0
  180. rag_wright/util/spacy_model.py +45 -0
  181. rag_wright-0.1.0.dist-info/METADATA +168 -0
  182. rag_wright-0.1.0.dist-info/RECORD +184 -0
  183. rag_wright-0.1.0.dist-info/WHEEL +4 -0
  184. rag_wright-0.1.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,100 @@
1
+ """ADR-0066 P1b-1: introspect the hand-maintained clause extraction template (`clause_template.py`) into a
2
+ structured field spec -- the SHARED source used by both the bootstrap emitter (writes the specs into the ttl) and
3
+ the drift test (asserts the ttl captured them faithfully). One introspection, so emitter and test cannot diverge.
4
+
5
+ Keyed by TEMPLATE FIELD (`<Model>.<field>`), because the template is not a flat dimension->field map: it has
6
+ nested constraint models (CapConstraint/TemporalConstraint/Jurisdiction) and non-dimension fields (clause_type,
7
+ document_reference), and field names do not uniformly match dimension values (excepts<->carve_out).
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ import enum
13
+ import typing
14
+ from dataclasses import dataclass, field
15
+
16
+ from pydantic import BaseModel
17
+ from pydantic_core import PydanticUndefined
18
+
19
+ from rag_wright.ontology import clause_template as ct
20
+
21
+ # The models whose fields make up the extraction template (root + nested constraint models).
22
+ _MODELS: tuple[tuple[str, type[BaseModel]], ...] = (
23
+ ("Clause", ct.Clause),
24
+ ("CapConstraint", ct.CapConstraint),
25
+ ("TemporalConstraint", ct.TemporalConstraint),
26
+ ("Jurisdiction", ct.Jurisdiction),
27
+ )
28
+
29
+
30
+ @dataclass(frozen=True)
31
+ class TemplateFieldSpec:
32
+ """One field of the extraction template -- everything needed to regenerate it (P1b-2) + its knowledge."""
33
+
34
+ model: str
35
+ name: str
36
+ kind: str # scalar_enum | list_enum | list_str | optional_str | str | model_ref
37
+ default_token: str # required | none | list | enum:<value> | model_none
38
+ definition: str = "" # the field's LOOK-FOR description (verbatim; "" for the TODO gaps)
39
+ enum_class: str | None = None # the Enum class name (scalar_enum / list_enum)
40
+ model_ref: str | None = None # the nested model name (model_ref)
41
+ edge_label: str | None = None # the docling-graph edge label (nested refs)
42
+ max_length: int | None = None
43
+ examples: tuple[str, ...] = field(default_factory=tuple)
44
+
45
+
46
+ def _enum_class(ann: typing.Any) -> type[enum.Enum] | None:
47
+ for a in [ann, *typing.get_args(ann)]:
48
+ for b in [a, *typing.get_args(a)]:
49
+ if isinstance(b, type) and issubclass(b, enum.Enum):
50
+ return b
51
+ return None
52
+
53
+
54
+ def _model_class(ann: typing.Any) -> type[BaseModel] | None:
55
+ for a in [ann, *typing.get_args(ann)]:
56
+ if isinstance(a, type) and issubclass(a, BaseModel):
57
+ return a
58
+ return None
59
+
60
+
61
+ def _spec(model_name: str, name: str, f: typing.Any) -> TemplateFieldSpec:
62
+ ann = f.annotation
63
+ enum_cls = _enum_class(ann)
64
+ model_cls = _model_class(ann)
65
+ is_list = typing.get_origin(ann) is list
66
+ js = f.json_schema_extra if isinstance(f.json_schema_extra, dict) else {}
67
+ max_length = next((getattr(m, "max_length", None) for m in (f.metadata or [])
68
+ if getattr(m, "max_length", None) is not None), None)
69
+
70
+ if model_cls is not None:
71
+ kind, default_token = "model_ref", "model_none"
72
+ elif is_list and enum_cls is not None:
73
+ kind, default_token = "list_enum", "list"
74
+ elif is_list: # issue 0037: List[str] -- an OPEN descriptive list dim (verbatim capture, canonicalized at KG)
75
+ kind, default_token = "list_str", "list"
76
+ elif enum_cls is not None:
77
+ kind = "scalar_enum"
78
+ default_token = f"enum:{f.default.value}" if isinstance(f.default, enum.Enum) else "required"
79
+ else: # str / Optional[str]
80
+ kind = "optional_str" if typing.get_origin(ann) is typing.Union else "str"
81
+ default_token = "required" if f.default is PydanticUndefined else "none"
82
+
83
+ return TemplateFieldSpec(
84
+ model=model_name, name=name, kind=kind, default_token=default_token,
85
+ definition=f.description or "",
86
+ enum_class=enum_cls.__name__ if enum_cls is not None else None,
87
+ model_ref=model_cls.__name__ if model_cls is not None else None,
88
+ edge_label=js.get("edge_label"),
89
+ max_length=max_length,
90
+ examples=tuple(f.examples or ()),
91
+ )
92
+
93
+
94
+ def introspect_template_fields() -> list[TemplateFieldSpec]:
95
+ """The template's fields as structured specs (stable order: model order, then field-declaration order)."""
96
+ specs: list[TemplateFieldSpec] = []
97
+ for model_name, model in _MODELS:
98
+ for name, f in model.model_fields.items():
99
+ specs.append(_spec(model_name, name, f))
100
+ return specs
rag_wright/py.typed ADDED
File without changes
@@ -0,0 +1,2 @@
1
+ """Reference-pack facades (EP-REF): thin, worked-example wrappers over the engine API for the engine's reference
2
+ CONTRACT/COMPLIANCE domain. They show a product the exact call shape; a product owns its own seam + guardrails."""
@@ -0,0 +1,41 @@
1
+ """EP-REF-1c: thin REFERENCE invoker wrappers for the GENERIC compliance leg -- worked examples showing a product
2
+ the exact call shape: open a workspace, then invoke the capability BY NAME via `ainvoke_subgraph`. Each is a
3
+ one-liner over the engine API -- no store/embedder/model-id/id-parsing. Reference-ONLY: a product owns the
4
+ FTC-routing / ad-compliance variants (`run_ad_compliance_check`) and its guardrails/human-gate policy.
5
+ """
6
+ from __future__ import annotations
7
+
8
+ from typing import Any, Optional
9
+
10
+ from rag_wright.api import ainvoke_subgraph
11
+
12
+
13
+ async def invoke_compliance_check(ws: Any, *, subject_text: str, source_doc: str, k: int = 8,
14
+ sources: Optional[list[str]] = None):
15
+ """Check a subject TEXT against the Requirement KG -> a `ComplianceReport` (cited findings + gap matrix).
16
+ `sources` scopes to named policies (None = store-wide)."""
17
+ return await ainvoke_subgraph(
18
+ "compliance_check",
19
+ {"subject_text": subject_text, "source_doc": source_doc, "k": k, "sources": sources},
20
+ resources=ws)
21
+
22
+
23
+ async def invoke_document_check(ws: Any, *, doc_name: str, data: bytes, k: int = 8,
24
+ sources: Optional[list[str]] = None):
25
+ """Check a subject DOCUMENT (raw bytes: PDF/DOCX/HTML/TXT) against the Requirement KG -> a `ComplianceReport`."""
26
+ return await ainvoke_subgraph(
27
+ "compliance_check",
28
+ {"doc_name": doc_name, "data": data, "k": k, "sources": sources},
29
+ resources=ws)
30
+
31
+
32
+ async def invoke_policy_ingest(ws: Any, *, source: str, sections_path: Any = None,
33
+ doc_name: Optional[str] = None, data: Optional[bytes] = None):
34
+ """Ingest a policy into the Requirement KG -> an `IngestionReport`. Give EITHER `sections_path` (a pre-sectioned
35
+ eCFR-style `sections.json`) OR `doc_name` + `data` (a policy DOCUMENT's raw bytes, split at its headings)."""
36
+ inputs: dict = {"source": source}
37
+ if data is not None:
38
+ inputs |= {"doc_name": doc_name, "data": data}
39
+ else:
40
+ inputs["sections_path"] = sections_path
41
+ return await ainvoke_subgraph("compliance_ingestion", inputs, resources=ws)
@@ -0,0 +1,123 @@
1
+ """EP-REF-1d: a REFERENCE product seam for the engine's CONTRACT/COMPLIANCE reference domain -- a worked example
2
+ of what a product seam looks like AFTER the domain-agnostic separation. It is the shape EP-SEAM-3 refactors
3
+ RuleWright's real `engine/seam.py` toward.
4
+
5
+ Every method is a thin composition over the engine: `open_workspace` (tenancy = one call), the invokers
6
+ (`ainvoke_subgraph`), the generic API reads (`entities_by_name`), and the reference-pack store extensions
7
+ (`ContractKGStore` / `ComplianceStore`) + the compliance invoker wrappers. It holds NO `ArcadeDBStore`, embedder,
8
+ model id, or id-string parsing -- those are engine calls.
9
+
10
+ This is a worked EXAMPLE, deliberately thin. A real product adds, around these calls, the concerns marked
11
+ `# PRODUCT OWNS:` below -- tenancy policy, scoping (`ScopeViolation`), the FTC/ad-compliance variants, the
12
+ unknown-policy guard, presentation/citation types, caching, and routing the engine's usage/progress to its own
13
+ telemetry. Those stay product-side; see `docs/product/seam-adaptation-guide.md`.
14
+ """
15
+ from __future__ import annotations
16
+
17
+ from typing import Any, Optional
18
+
19
+ from rag_wright.api import (
20
+ EngineConfig,
21
+ ainvoke_subgraph,
22
+ aparse_document,
23
+ entities_by_name,
24
+ open_workspace,
25
+ source_document,
26
+ )
27
+ from rag_wright.capabilities.compliance_store import ComplianceStore
28
+ from rag_wright.capabilities.contract_kg_store import ContractKGStore
29
+ from rag_wright.reference.compliance import invoke_compliance_check, invoke_policy_ingest
30
+
31
+
32
+ class ContractComplianceSeam:
33
+ """A thin reference seam over the engine for the contract/compliance reference domain."""
34
+
35
+ def __init__(self, config: EngineConfig) -> None:
36
+ self._config = config
37
+
38
+ # --- tenancy -------------------------------------------------------------------------------------
39
+ def open(self, corpus: str, *, reset: bool = False):
40
+ """Open a workspace for a corpus (one engine call). PRODUCT OWNS: per-tenant corpus selection + auth policy."""
41
+ return open_workspace(self._config, corpus=corpus, reset=reset)
42
+
43
+ # --- ingestion + query (compose the invokers) ----------------------------------------------------
44
+ async def ingest_contract(self, ws, doc_id: str, *, cache_dir: str, text: Optional[str] = None,
45
+ path: Any = None, metadata: Optional[dict] = None):
46
+ """Ingest one contract: build a `SourceDocument` (docling-parse a file PATH, or wrap TEXT) then run the
47
+ ingestion capability. Returns the engine's `IngestionReport`. PRODUCT OWNS: the document source + progress UI
48
+ (the engine emits progress/usage; route it to your telemetry)."""
49
+ sd = (await aparse_document(doc_id, path, cache_dir=cache_dir, metadata=metadata) if path is not None
50
+ else source_document(doc_id, text=text or ""))
51
+ return await ainvoke_subgraph("contract_ingestion_pipeline",
52
+ {"document": sd, "cache_dir": cache_dir}, resources=ws)
53
+
54
+ async def ask_contract(self, ws, contract_id: str, question: str):
55
+ """Single-document Q&A over one contract. Returns the engine's cited answer (raw -- PRODUCT OWNS: what
56
+ counts as 'answered' / how to render the citations)."""
57
+ return await ainvoke_subgraph("intra_document_qa",
58
+ {"contract_id": contract_id, "question": question}, resources=ws)
59
+
60
+ async def search_corpus(self, ws, query: str, *, k: int = 8, documents: Optional[list[str]] = None):
61
+ """Corpus-wide typed/similarity retrieval (Leg B). The leg extracts the query's typed constraints itself;
62
+ pass `documents` to scope to a workspace's docs. PRODUCT OWNS: ranking/selection presentation."""
63
+ return await ainvoke_subgraph("typed_property_retrieval",
64
+ {"query": query, "k": k, "documents": documents}, resources=ws)
65
+
66
+ # --- relational / terms / citations (reference-pack store extensions over the workspace store) ----
67
+ def find_party(self, ws, name: str) -> list[tuple[str, str]]:
68
+ """A party NAME -> every `(entity_id, stored_name)` it resolves to (engine-normalized; one name can match
69
+ several nodes -- all are returned, never the first only)."""
70
+ return sorted((str(e["entity_id"]), str(e.get("name") or "")) for e in entities_by_name(ws, name))
71
+
72
+ def counterparties(self, ws, entity_id: str, *, max_hops: int = 1, documents: Optional[list[str]] = None):
73
+ """The parties this one has a CONTRACTS_WITH edge to (one hop by default)."""
74
+ return ContractKGStore(ws._store).party_counterparties(entity_id, max_hops=max_hops, documents=documents)
75
+
76
+ def affiliates(self, ws, entity_id: str, *, documents: Optional[list[str]] = None):
77
+ """The parties this one has an AFFILIATE_OF edge to (same corporate group -- a separate traversal)."""
78
+ return ContractKGStore(ws._store).party_affiliates(entity_id, documents=documents)
79
+
80
+ def contract_terms(self, ws, contract_id: str) -> list:
81
+ """The typed clauses of one contract (the full view -- keeps AMBIGUOUS out-of-vocab values)."""
82
+ return ContractKGStore(ws._store).contract_terms(contract_id)
83
+
84
+ def span_locations(self, ws, contract_id: str) -> list:
85
+ """Every span's position (pages/bbox/offsets) + the clause ids on it -- for a citation preview."""
86
+ return ContractKGStore(ws._store).span_locations(contract_id)
87
+
88
+ @staticmethod
89
+ def canonical_clause_type(label: str) -> Optional[str]:
90
+ """Map a user's clause label onto the taxonomy (alias-resolving), or None -- to validate a correction."""
91
+ return ContractKGStore.canonical_clause_type(label)
92
+
93
+ @staticmethod
94
+ def clause_type_vocabulary() -> tuple[str, ...]:
95
+ """Every clause type a sweep/correction UI can be scoped to."""
96
+ return ContractKGStore.clause_type_vocabulary()
97
+
98
+ # --- compliance (reference invoker wrappers + the read facade) -----------------------------------
99
+ async def ingest_policy(self, ws, *, source: str, sections_path: Any = None,
100
+ doc_name: Optional[str] = None, data: Optional[bytes] = None):
101
+ """Ingest a policy into the Requirement KG (sections.json OR a document's bytes). Returns an
102
+ `IngestionReport`. PRODUCT OWNS: the unknown-policy guard / curation workflow."""
103
+ return await invoke_policy_ingest(ws, source=source, sections_path=sections_path,
104
+ doc_name=doc_name, data=data)
105
+
106
+ async def check(self, ws, *, subject_text: str, source_doc: str, k: int = 8,
107
+ sources: Optional[list[str]] = None):
108
+ """Check a subject against the Requirement KG -> a `ComplianceReport` (the GENERIC verdict). PRODUCT OWNS:
109
+ the FTC/ad-compliance tuned variant (`run_ad_compliance_check`), the unknown-policy guard, the human gate."""
110
+ return await invoke_compliance_check(ws, subject_text=subject_text, source_doc=source_doc,
111
+ k=k, sources=sources)
112
+
113
+ def requirements_for(self, ws, source: str) -> list[dict]:
114
+ """The curated requirement rows under one policy (the proof an ingest landed)."""
115
+ return ComplianceStore(ws._store).requirements_for(source)
116
+
117
+ def requirement_locations(self, ws, source: str) -> list:
118
+ """Where each requirement of one policy sits in its document (pages + bbox) -- for a citation preview."""
119
+ return ComplianceStore(ws._store).requirement_locations(source)
120
+
121
+ def curated_requirement_count(self, ws, sources: Optional[list[str]] = None) -> int:
122
+ """How many requirements are in scope -- the denominator of an honest coverage statement."""
123
+ return ComplianceStore(ws._store).curated_requirement_count(sources)
@@ -0,0 +1,7 @@
1
+ """Authored skill content built as ordinary software.
2
+
3
+ Holds the RLM skill (FR-C.10): an authored SKILL.md teaching the divide-and-conquer method
4
+ (load a working set into an interpreter, slice and dispatch the work in code, synthesize the
5
+ results). Used by the RLM chunking capability (ingestion) and the RLM synthesis capability
6
+ (query). It is not provided by any build tool and is not a compiler feature in this repo.
7
+ """
@@ -0,0 +1,47 @@
1
+ ---
2
+ name: claim_extraction
3
+ description: >
4
+ The subject-document claim-extraction method: read an advertisement and pull out its distinct CHECKABLE
5
+ claims (the assertions a regulator could test), each with its kind, the disclosures present near it, and
6
+ whether the ad references evidence. Applied by the compliance_check subgraph (subject side). The schema is
7
+ the co-located asset `template.py` (ExtractedAd / ExtractedClaim); the deterministic mapping to the closed
8
+ Claim vocab is the claim_adaptation FUNCTION's job, not this skill's.
9
+ ---
10
+
11
+ # Claim extraction: what checkable claims does this ad make?
12
+
13
+ This skill teaches a **method**, not a behavior. It turns a subject advertisement into the checkable claims the
14
+ compliance check will judge. It is authored software (an Agent Skill), with its extraction **schema** as the
15
+ co-located asset **`template.py`** (`ExtractedAd` → `ExtractedClaim[]`), referenced here and filled by the model.
16
+
17
+ ## What to extract (the schema — `template.py`)
18
+
19
+ Per ad, produce an `ExtractedAd` whose `claims` are the distinct **checkable** assertions. For each claim:
20
+
21
+ - **assertion_text** — the claim itself, quoted or closely paraphrased; one claim per entry.
22
+ - **claim_type** — the kind, from the closed vocab: `efficacy, comparative, pricing, health, environmental,
23
+ endorsement, performance, guarantee`.
24
+ - **disclosures_present** — any disclaimers/qualifiers near the claim: `#ad`, `paid partnership`, `results vary`.
25
+ - **evidence_referenced** — whether the ad points to a study/data for the claim.
26
+ - **actor / subject_product / quantitative_value / medium** — when present.
27
+
28
+ Extract the **checkable** assertions (a regulator could test them), not pure subjective flourish — but when in
29
+ doubt, extract it; the judgment step decides puffery vs objective claim.
30
+
31
+ ## The reliability method (docling-graph, from GP-1B)
32
+
33
+ The extractor runs through docling-graph in API mode. Three settings are load-bearing and must not drift:
34
+
35
+ 1. **source must be a file path**, not a raw string — docling-graph `stat()`s it (write the text to a temp file).
36
+ 2. **`structured_output=False`** (json_object) — the strict nested json_schema returns nothing on some models;
37
+ json_object is reliable across DeepSeek / Gemma / Granite.
38
+ 3. **a `max_tokens` cap** — an unknown provider else gets a generic 8192 context window and SKIPS the LLM.
39
+
40
+ Ads are short, so `extraction_contract="direct"` (one call) is correct here.
41
+
42
+ ## What this skill does NOT own (the applying capability's job)
43
+
44
+ - mapping `claim_type` to the closed `ClaimType` vocab and coercing an off-vocab value (kept-but-AMBIGUOUS, a
45
+ checkable assertion is never dropped) — the `claim_adaptation` FUNCTION;
46
+ - the `claim_id` content-hash identity and the span provenance — the FUNCTION;
47
+ - concurrency, timeouts, and the compliance judgment that follows — the subgraph.
@@ -0,0 +1 @@
1
+ """claim_extraction agent-skill folder: SKILL.md + the template.py schema asset."""
@@ -0,0 +1,50 @@
1
+ """Schema ASSET for the `claim_extraction` agent skill (referenced by this folder's SKILL.md).
2
+
3
+ The docling-graph extraction template the skill fills: `ExtractedAd` (the subject ad) with its checkable
4
+ `ExtractedClaim`s. Loose strings by design (robust to model output); the deterministic `claim_adaptation`
5
+ FUNCTION maps them to the closed CC-1 `Claim` vocab. Co-located with the skill because the schema IS part of
6
+ the authored extraction method (an Agent-Skill asset), not a hidden implementation detail.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ from pydantic import BaseModel, ConfigDict, Field
12
+
13
+ from rag_wright.capabilities.dg_extraction import edge
14
+
15
+
16
+ class ExtractedClaim(BaseModel):
17
+ """One checkable assertion the LLM reads out of a subject ad (a docling-graph child entity)."""
18
+
19
+ model_config = ConfigDict(graph_id_fields=["assertion_text"], extra="ignore", populate_by_name=True)
20
+
21
+ assertion_text: str = Field(
22
+ description="One checkable factual claim the ad makes, quoted or closely paraphrased (one claim per entry)")
23
+ claim_type: str = Field(
24
+ default="",
25
+ description=("The kind of claim, chosen from: efficacy, comparative, pricing, health, environmental, "
26
+ "endorsement, performance, guarantee"))
27
+ actor: str = Field(default="", description=(
28
+ "DEON-8: the ROLE of the party this claim involves -- a role word, NOT a person's or company's name. "
29
+ "Choose the general role: advertiser, endorser, expert, manufacturer, seller. (E.g. 'Dr. Miller "
30
+ "recommends ...' -> endorser, not 'Dr. Miller'.) Empty if no clear actor."))
31
+ subject_product: str = Field(default="", description="The product or brand the claim is about")
32
+ quantitative_value: str = Field(
33
+ default="", description="Any specific number/quantity claimed, e.g. '30 pounds in one month', '2x faster'")
34
+ disclosures_present: list[str] = Field(
35
+ default_factory=list,
36
+ description="Disclaimers/qualifiers present near the claim, e.g. '#ad', 'paid partnership', 'results vary'")
37
+ evidence_referenced: bool = Field(
38
+ default=False, description="Whether the ad references evidence/substantiation for the claim (a study, data)")
39
+ medium: str = Field(default="", description="The medium, e.g. social, tv, print, podcast, web")
40
+
41
+
42
+ class ExtractedAd(BaseModel):
43
+ """The subject document and the distinct checkable claims it makes (the docling-graph root entity)."""
44
+
45
+ model_config = ConfigDict(graph_id_fields=["subject"], extra="ignore", populate_by_name=True)
46
+
47
+ subject: str = Field(description="A short label for the subject ad (the brand/product or a headline phrase)")
48
+ claims: list[ExtractedClaim] = edge(
49
+ "MAKES_CLAIM", default_factory=list,
50
+ description="The distinct checkable claims the ad makes (one entry per claim)")
@@ -0,0 +1,59 @@
1
+ ---
2
+ name: compliance_judgment
3
+ description: >
4
+ The advertising-compliance judgment method: given ONE advertising claim and ONE applicable regulatory
5
+ requirement, and seeing only the ad text (never the advertiser's evidence files), decide whether the claim
6
+ clearly violates the requirement, clearly satisfies it, or cannot be judged from the text and must be
7
+ escalated for human review. Applied by the compliance_check subgraph (query side). The applying capability
8
+ owns the deterministic guarantees (verdict vocab, conservative default, both-sided citation) -- this skill
9
+ teaches only the reading.
10
+ ---
11
+
12
+ # Compliance judgment: does this claim satisfy or violate this requirement?
13
+
14
+ This skill teaches a **method**, not a behavior. It extends the grounding-judge idea (ADR-0028) from
15
+ "is X supported by cue Y?" to "does claim X satisfy or violate requirement Y?". A capability applies it with
16
+ its own contract (`ComplianceFinding`) and its own guarantees; those guarantees are the **applying
17
+ capability's** job, not the method's (see "What this skill does NOT own").
18
+
19
+ ## The one hard constraint: you see only the ad text
20
+
21
+ You are given the requirement and the claim. You **cannot** see the advertiser's studies, substantiation
22
+ files, or evidence. So you can only judge what the *text itself* shows. This constraint is the whole reason the
23
+ verdict is three-way, not two-way.
24
+
25
+ ## The three verdicts
26
+
27
+ - **violation** — reserve this for what is **clearly wrong from the text itself**:
28
+ 1. the claim **overclaims proof** — it asserts it is "clinically proven", "scientifically proven", "science
29
+ backed", "doctor proven", or "guaranteed" **without pointing to an actual study or data** (the
30
+ proof-language *is* the unsubstantiated claim, not evidence for it);
31
+ 2. an endorsement is **missing a required disclosure** — no "#ad" / "paid partnership" is present when a
32
+ material connection would need disclosing;
33
+ 3. a review or testimonial is **fake or deceptive**.
34
+
35
+ - **needs_review** — an **objective** efficacy / health / performance / factual claim that may well be true, but
36
+ the ad shows **no evidence** and makes **no overclaim**. You cannot verify its substantiation from the text
37
+ alone, so **escalate** it: a human will check the advertiser's substantiation file. Do **not** call this a
38
+ violation (you don't have the evidence), and do **not** clear it as compliant (you can't confirm it either).
39
+
40
+ - **compliant** — the requirement does not bite, because one of:
41
+ - the claim is mere **subjective opinion or taste/experience puffery** ("smooth flavor", "relaxing", "I like
42
+ it") — there is nothing objective to substantiate;
43
+ - the required **disclosure is present** (see the claim's `disclosures_present`, e.g. "#ad");
44
+ - the ad **actually points to real evidence** (a specific study / data / citation) for the objective claim;
45
+ - there is **no objective claim** to substantiate.
46
+
47
+ ## The discipline
48
+
49
+ Only say **violation** when the text clearly shows the breach; only say **compliant** when the text clearly
50
+ clears it; **otherwise `needs_review`**. Uncertainty is escalation, never a silent pass. Give a one-sentence
51
+ rationale and a confidence in [0,1].
52
+
53
+ ## What this skill does NOT own (the applying capability's job)
54
+
55
+ - the verdict **vocabulary** and the **conservative default** — an unreadable or missing verdict maps to
56
+ `needs_review` deterministically, in the capability, not here;
57
+ - the **both-sided citation** — the exact claim span and the exact requirement clause are attached from the
58
+ INPUTS by the capability, never authored by this skill (the model rules; it never fabricates a citation);
59
+ - the **ad-level rollup** (how many findings make an ad a violation) and the **human gate**.
@@ -0,0 +1,106 @@
1
+ ---
2
+ name: corpus_ingest
3
+ description: >
4
+ The repeatable method for ingesting ANY new corpus of contracts into the one contract KG and
5
+ auto-connecting it (parties <-> clauses). Point the GENERIC ingestion pipeline
6
+ (contract_ingestion_pipeline) at the corpus through a single thin CorpusAdapter -- never a
7
+ re-implemented ingest_xyz(). Write the adapter (parse + canonical source_doc_id + optional metadata),
8
+ point the entity registry at the corpus's parties, and run run_corpus_ingestion; the pipeline chunks,
9
+ segments, function-classifies, extracts clauses + the party graph, resolves entities, and writes both. Applied over src/rag_wright/subgraphs/contract_ingestion_pipeline.py.
10
+ ---
11
+
12
+ # Ingesting a new corpus into the contract KG
13
+
14
+ This skill teaches a **method**, not a behavior. The design that makes it cheap: ONE generic, corpus-agnostic
15
+ ingestion pipeline (LG-3d, `contract_ingestion_pipeline`) plus a thin per-corpus **`CorpusAdapter`**. Adding a
16
+ corpus is one adapter — **never** a re-implemented `ingest_xyz()` that duplicates the flow. See ADR-0033
17
+ (unified KG), HYG-1 (canonical identity), ADR-0037 (the clause template is authoritative code), and
18
+ `docs/corpus_ingest_recipe.md` (the prose recipe this skill formalizes).
19
+
20
+ ## The method: one generic pipeline + one thin adapter
21
+
22
+ The pipeline is fixed and shared. Everything corpus-specific lives behind one seam,
23
+ `CorpusAdapter.documents() -> Iterable[SourceDocument]`. `SourceDocument` is `{source_doc_id, text, metadata}`.
24
+ The pipeline, per document, runs: **chunk (semantic_chunking) → segment → LegalBERT function-classify →
25
+ clause-extract ∥ graph-extract → entity_resolution → write (clause KG + entity graph)**. (The KG-7
26
+ `party_clause_linking`/PartyTo post-step was retired — issue 0028 / ADR-0091 — since party→clause is reached
27
+ via CONTRACTS_WITH provenance + the contract-scoped clause KG.)
28
+
29
+ ## A new *contract* corpus — 3 steps
30
+
31
+ ### 1. Write one `CorpusAdapter` — the ONLY new code
32
+ `documents()` owns everything corpus-specific:
33
+ - enumerate the corpus's files/records;
34
+ - **parse each to text** (PDF → docling; JSON → read; …) — parsing lives here, so the pipeline is parse-agnostic;
35
+ - assign the id via `canonical_source_doc_id(...)` (HYG-1) — this is **load-bearing**: it is what lets the new
36
+ corpus's clauses, spans, entities, and contracts share ONE id scheme and auto-connect. A slug that disagrees
37
+ with the rest (e.g. `-` for spaces instead of `_`) silently breaks the cross-graph join;
38
+ - optionally attach corpus quirks on `SourceDocument.metadata` (annotated parties, pre-segmented spans, …).
39
+
40
+ `CuadAdapter` (in `contract_ingestion_pipeline.py`) is the reference implementation.
41
+
42
+ ### 2. Point the entity registry at the corpus's parties
43
+ Extend the EDGAR verified registry (`build_verified_registry`) for the corpus's public companies, or accept
44
+ `UNLINKED` / `PRIVATE` for parties not in the registry (the honest closed-world gap).
45
+
46
+ ### 3. Run it (monitored)
47
+
48
+ ```python
49
+ from rag_wright.subgraphs.contract_ingestion_pipeline import (
50
+ arun_corpus_ingestion, aproduction_document_ingest,
51
+ )
52
+
53
+ report = await arun_corpus_ingestion(
54
+ YourAdapter(path, limit=N), # test on a FEW docs first; never a full re-ingest without intent
55
+ aproduction_document_ingest(store, cache_dir=..., registry=...),
56
+ )
57
+ # report: documents_ingested, dead_lettered (per-doc), per_document
58
+ ```
59
+
60
+ The ingest **extraction models are caller-configurable** (like the query/compliance entrypoints; env vars stay
61
+ the fallback): `aproduction_document_ingest(store, cache_dir=..., registry=..., extract_model=..., list_model=...,
62
+ samples=...)`. `extract_model` is the primary clause-property model (an `ExtractionModel` or a bare model-id
63
+ string; default = the `RAG_SERVING` backend model, granite). `list_model` is the SECOND model for the cross-model
64
+ UNION on the LIST-bearing dims only (carve_out / covered_subject / damage_type) — granite and gemma
65
+ under-enumerate different list items, so their union is more complete; `"off"` disables it, default = gemma. If
66
+ you override `extract_model` (e.g. to qwen), set `list_model` deliberately — the union only helps if the two
67
+ models are complementary. `graph_extract_model` sets the party+affiliation extraction model (they share one),
68
+ `judge_model` the ingest semantic-judge model, and `chunk_model` the chunker's boundary-refinement model (used
69
+ only for over-cap sections). All accept a bare id or an `ExtractionModel` and default to their backend/env value,
70
+ so every ingest LLM surface (clause extract + list union, party/affiliation, judge, chunk boundary, and the
71
+ pre-existing `classify_fn`) is now a call-site argument.
72
+
73
+ `run_corpus_ingestion` streams `X/N` progress; a bad document dead-letters and is skipped (one bad doc never
74
+ kills the corpus). Write to a SCRATCH database first (`from_env(database=..., reset=True)`) to keep it
75
+ non-destructive while proving it out.
76
+
77
+ ## Rules (not optional)
78
+
79
+ 1. **Never write an `ingest_xyz()` that re-implements the flow.** A new corpus = one `CorpusAdapter`, then
80
+ `run_corpus_ingestion(adapter, ...)`. If you find yourself copying the pipeline, stop.
81
+ 2. **The canonical `source_doc_id` is load-bearing.** Always derive it via `canonical_source_doc_id`; a
82
+ divergent slug breaks the cross-graph join (HYG-1). Every graph must share one id scheme.
83
+ 3. **Extraction is concurrent, per-item tolerant, and cached.** Clause/graph extraction runs under
84
+ `map_concurrent` (granite is ~10s/single call); a truncated/failed span is SKIPPED, not fatal to the
85
+ document; and each successful extraction is cached by clause-id + template-schema-version, so a re-run or a
86
+ template change re-extracts only what it must.
87
+ 4. **Long-running runs stream `X/N` and are actively monitored** (CLAUDE.md) — never launch-and-forget.
88
+ 5. **The clause template is authoritative code, not regenerated** (ADR-0037); tune extraction by editing
89
+ `clause_template.py`, never by chasing a regeneration from the `.ttl`/spec.
90
+
91
+ ## A new *domain* (non-contract)
92
+ The pipeline *structure* stays; the capabilities it binds change: a **new extraction template** (bootstrap a
93
+ fresh `.py` from a new ontology, then hand-maintain it — ADR-0037), a **retrained/replaced function classifier**
94
+ (new taxonomy), and possibly a different entity registry.
95
+
96
+ ## What this skill does NOT own (deferred to the pipeline / capabilities)
97
+ - **The pipeline internals** (`build_document_ingest` graph, dead-letter, the extraction seams) — LG-3d.
98
+ - **The extraction capabilities** — semantic_chunking, the function classifier, clause extraction (docling-graph
99
+ + granite, ADR-0037 template), GP-1B graph_extraction (ADR-0035), entity_resolution.
100
+ - **The span/embedding retrieval index** — wired into the pipeline as a parallel `index_spans` branch off the
101
+ shared `segment` node (INGEST-REFACTOR phase 2a); a corpus now gets the typed KG + entity graph + the
102
+ dense/sparse retrieval index in one pass. Indexing is best-effort (a failed index degrades to 0 spans, never
103
+ dead-letters the document's KG).
104
+
105
+ The method is: parse behind the adapter, one canonical id, run the generic pipeline, connect once. Keep this
106
+ file about that shape; the pipeline supplies the flow, the capabilities, and the tests.
@@ -0,0 +1,51 @@
1
+ ---
2
+ name: extraction_semantic_judge
3
+ description: >
4
+ The clause-property faithfulness method: given ONE extracted property (a dimension = a value) and the clause
5
+ text it was extracted from, decide whether a careful reading of THIS clause genuinely SUPPORTS that property.
6
+ For the closed SEMANTIC dimensions (mutuality, favorability, party_asymmetry, cap_basis, the consent regimes)
7
+ whose value is a reading with no surface token, this is what the lexical and symbolic gates cannot reach.
8
+ Layer 3 of the neuro-symbolic extraction-fidelity cascade (ADR-0040). The applying capability owns the
9
+ deterministic guarantees (which dimensions are semantic, the AMBIGUOUS downgrade, concurrency) -- this skill
10
+ teaches only the reading.
11
+ ---
12
+
13
+ # Extraction semantic judge: does this clause support this property?
14
+
15
+ This skill teaches a **method**, not a behavior. It extends the grounding-judge idea (ADR-0028) from
16
+ "is X supported by cue Y in the text?" to the harder **semantic** case: "does a faithful reading of THIS clause
17
+ support the property (dimension = value) that was extracted from it?" A capability applies it with its own
18
+ contract (`ClausePropertyRecord`) and its own guarantees; those guarantees are the **applying capability's**
19
+ job, not the method's (see "What this skill does NOT own").
20
+
21
+ ## What you are judging
22
+
23
+ You are given one extracted **property** as `dimension = value` (with a short plain-language meaning of what
24
+ that property claims about the clause) and the **clause text**. These are the closed SEMANTIC dimensions:
25
+ `mutuality` (obligation runs both ways vs one), `favorability` (which side a term favors), `party_asymmetry`,
26
+ `cap_basis` (fixed_fee vs multiple_of_fees), the governing-law multiplicity, IP ownership, the non-solicit
27
+ target, the renewal mechanism, the change-of-control / assignment consent regimes, the MFN scope, the
28
+ termination right. Their value is a **reading** of the clause, not a token you can grep for -- which is exactly
29
+ why a model is spent here and nowhere else in the cascade.
30
+
31
+ ## The one hard rule: strictness
32
+
33
+ Decide **supported=true** only if the clause **genuinely** supports the property. Decide **supported=false**
34
+ if the clause does not support it or contradicts it. Be strict:
35
+
36
+ - **mere plausibility is not support** -- that a mutual reading is *possible* is not enough; the clause must
37
+ actually bear it;
38
+ - **absence of support in this clause means supported=false** -- if the text is silent on what the property
39
+ asserts, it is not supported. Do not import world knowledge or the "usual" drafting; judge only THIS clause.
40
+
41
+ A `false` verdict means the reading is unsupported by the text; the applying capability will downgrade that
42
+ assertion to AMBIGUOUS (kept but flagged), never delete it. Give a one-sentence reason.
43
+
44
+ ## What this skill does NOT own (the applying capability's job)
45
+
46
+ - **which dimensions are semantic** -- the set of dimensions this judge runs on (closed vocabulary, no lexical
47
+ cue) is selected deterministically by the capability, not decided here;
48
+ - the **AMBIGUOUS downgrade** and the "leave untouched on a judge error / no ruling" conservative rule --
49
+ applied by the capability, not by this method;
50
+ - the **per-assertion dispatch and concurrency** -- the capability runs this reading over each surviving
51
+ (non-AMBIGUOUS) semantic assertion; the method judges exactly one at a time.
@@ -0,0 +1 @@
1
+ """extraction_semantic_judge agent-skill folder: SKILL.md (the clause-property faithfulness method)."""