rag-wright 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (184) hide show
  1. rag_wright/__init__.py +13 -0
  2. rag_wright/api/__init__.py +33 -0
  3. rag_wright/api/config.py +59 -0
  4. rag_wright/api/discover.py +70 -0
  5. rag_wright/api/documents.py +39 -0
  6. rag_wright/api/ids.py +31 -0
  7. rag_wright/api/invoke.py +99 -0
  8. rag_wright/api/kg.py +61 -0
  9. rag_wright/api/mcp.py +94 -0
  10. rag_wright/api/usage.py +30 -0
  11. rag_wright/api/workspace.py +85 -0
  12. rag_wright/capabilities/__init__.py +8 -0
  13. rag_wright/capabilities/answer_generator.py +427 -0
  14. rag_wright/capabilities/ard.py +286 -0
  15. rag_wright/capabilities/assertion_extraction.py +79 -0
  16. rag_wright/capabilities/chunk_read.py +58 -0
  17. rag_wright/capabilities/chunk_write.py +163 -0
  18. rag_wright/capabilities/claim_extraction.py +153 -0
  19. rag_wright/capabilities/clause_exception_linking.py +117 -0
  20. rag_wright/capabilities/compliance_judgment.py +322 -0
  21. rag_wright/capabilities/compliance_store.py +87 -0
  22. rag_wright/capabilities/contract_kg_serve.py +156 -0
  23. rag_wright/capabilities/contract_kg_store.py +251 -0
  24. rag_wright/capabilities/dg_extraction.py +585 -0
  25. rag_wright/capabilities/disambiguation.py +163 -0
  26. rag_wright/capabilities/document_parse.py +87 -0
  27. rag_wright/capabilities/document_scope.py +49 -0
  28. rag_wright/capabilities/embedding.py +164 -0
  29. rag_wright/capabilities/embedding_profiles.py +43 -0
  30. rag_wright/capabilities/entity_resolution.py +154 -0
  31. rag_wright/capabilities/fusion.py +64 -0
  32. rag_wright/capabilities/graph_extraction.py +243 -0
  33. rag_wright/capabilities/graph_query.py +73 -0
  34. rag_wright/capabilities/graph_storage.py +111 -0
  35. rag_wright/capabilities/highlight_serve.py +142 -0
  36. rag_wright/capabilities/hybrid_search.py +65 -0
  37. rag_wright/capabilities/invoke.py +31 -0
  38. rag_wright/capabilities/jev_decision.py +38 -0
  39. rag_wright/capabilities/manifests.py +872 -0
  40. rag_wright/capabilities/okf_navigate.py +456 -0
  41. rag_wright/capabilities/parsing.py +286 -0
  42. rag_wright/capabilities/property_boosted_retrieval.py +125 -0
  43. rag_wright/capabilities/query_function_classifier.py +94 -0
  44. rag_wright/capabilities/query_understanding.py +109 -0
  45. rag_wright/capabilities/registry.py +262 -0
  46. rag_wright/capabilities/remote_encoders.py +94 -0
  47. rag_wright/capabilities/requirement_extraction.py +247 -0
  48. rag_wright/capabilities/reranking.py +123 -0
  49. rag_wright/capabilities/retrieval_core.py +126 -0
  50. rag_wright/capabilities/rlm_chunking.py +808 -0
  51. rag_wright/capabilities/rlm_synthesis.py +316 -0
  52. rag_wright/capabilities/scan_quality.py +136 -0
  53. rag_wright/capabilities/span_relevance_judgment.py +191 -0
  54. rag_wright/capabilities/vision_to_text.py +85 -0
  55. rag_wright/capabilities/vlm_ocr.py +85 -0
  56. rag_wright/contracts/__init__.py +6 -0
  57. rag_wright/contracts/chunk.py +79 -0
  58. rag_wright/contracts/compliance.py +303 -0
  59. rag_wright/contracts/contract_meta.py +27 -0
  60. rag_wright/contracts/extraction.py +130 -0
  61. rag_wright/contracts/function.py +167 -0
  62. rag_wright/contracts/function_routing.py +91 -0
  63. rag_wright/contracts/highlight.py +74 -0
  64. rag_wright/contracts/identifiers.py +153 -0
  65. rag_wright/contracts/jurisdiction.py +96 -0
  66. rag_wright/contracts/ontology.py +142 -0
  67. rag_wright/contracts/property.py +201 -0
  68. rag_wright/contracts/provenance.py +78 -0
  69. rag_wright/contracts/query_intent.py +53 -0
  70. rag_wright/contracts/span.py +76 -0
  71. rag_wright/contracts/value_match.py +84 -0
  72. rag_wright/corpus/__init__.py +0 -0
  73. rag_wright/corpus/canonicalize.py +116 -0
  74. rag_wright/corpus/cuad.py +153 -0
  75. rag_wright/corpus/cuad_ingestion.py +72 -0
  76. rag_wright/corpus/document_parser.py +299 -0
  77. rag_wright/corpus/edgar.py +231 -0
  78. rag_wright/corpus/gcs_ingestion.py +120 -0
  79. rag_wright/corpus/http.py +110 -0
  80. rag_wright/corpus/selection.py +152 -0
  81. rag_wright/mcp/__init__.py +11 -0
  82. rag_wright/mcp/compliance_server.py +299 -0
  83. rag_wright/mcp/intra_document_qa_server.py +170 -0
  84. rag_wright/mcp/relational_qa_server.py +171 -0
  85. rag_wright/mcp/session_store.py +64 -0
  86. rag_wright/mcp/typed_property_retrieval_server.py +191 -0
  87. rag_wright/models/__init__.py +8 -0
  88. rag_wright/models/profiles.py +331 -0
  89. rag_wright/models/seam.py +497 -0
  90. rag_wright/models/tag_structured.py +285 -0
  91. rag_wright/models/tracing.py +179 -0
  92. rag_wright/models/usage.py +102 -0
  93. rag_wright/okf/__init__.py +11 -0
  94. rag_wright/okf/compile.py +292 -0
  95. rag_wright/okf/document.py +47 -0
  96. rag_wright/okf/enrich.py +176 -0
  97. rag_wright/okf/links.py +190 -0
  98. rag_wright/okf/lint.py +105 -0
  99. rag_wright/ontology/__init__.py +6 -0
  100. rag_wright/ontology/_generated_template_meta.py +60 -0
  101. rag_wright/ontology/_generated_vocab.py +52 -0
  102. rag_wright/ontology/clause_template.py +964 -0
  103. rag_wright/ontology/codegen.py +84 -0
  104. rag_wright/ontology/compliance_bridge.ttl +186 -0
  105. rag_wright/ontology/contract_bridge.ttl +2685 -0
  106. rag_wright/ontology/contract_taxonomy.py +24 -0
  107. rag_wright/ontology/derive.py +58 -0
  108. rag_wright/ontology/loader.py +435 -0
  109. rag_wright/ontology/packs/ftc_16cfr255.ttl +29 -0
  110. rag_wright/ontology/registry.py +87 -0
  111. rag_wright/ontology/template_introspect.py +100 -0
  112. rag_wright/py.typed +0 -0
  113. rag_wright/reference/__init__.py +2 -0
  114. rag_wright/reference/compliance.py +41 -0
  115. rag_wright/reference/contract_seam.py +123 -0
  116. rag_wright/skills/__init__.py +7 -0
  117. rag_wright/skills/claim_extraction/SKILL.md +47 -0
  118. rag_wright/skills/claim_extraction/__init__.py +1 -0
  119. rag_wright/skills/claim_extraction/template.py +50 -0
  120. rag_wright/skills/compliance_judgment/SKILL.md +59 -0
  121. rag_wright/skills/corpus_ingest/SKILL.md +106 -0
  122. rag_wright/skills/extraction_semantic_judge/SKILL.md +51 -0
  123. rag_wright/skills/extraction_semantic_judge/__init__.py +1 -0
  124. rag_wright/skills/generation/SKILL.md +64 -0
  125. rag_wright/skills/generation/__init__.py +1 -0
  126. rag_wright/skills/generic_compliance_judgment/SKILL.md +58 -0
  127. rag_wright/skills/okf_navigate/SKILL.md +137 -0
  128. rag_wright/skills/requirement_extraction/SKILL.md +47 -0
  129. rag_wright/skills/requirement_extraction/__init__.py +1 -0
  130. rag_wright/skills/requirement_extraction/template.py +50 -0
  131. rag_wright/skills/rlm/SKILL.md +186 -0
  132. rag_wright/skills/rlm/__init__.py +31 -0
  133. rag_wright/skills/rlm/agent.py +292 -0
  134. rag_wright/skills/span_relevance_judgment/SKILL.md +67 -0
  135. rag_wright/skills/vision_to_text/SKILL.md +36 -0
  136. rag_wright/skills/vision_to_text/__init__.py +1 -0
  137. rag_wright/spans/__init__.py +1 -0
  138. rag_wright/spans/boundary.py +78 -0
  139. rag_wright/spans/clause_function_classifier.py +490 -0
  140. rag_wright/spans/clause_kg_extractor.py +337 -0
  141. rag_wright/spans/cuad_labels.py +81 -0
  142. rag_wright/spans/dim_classifier.py +158 -0
  143. rag_wright/spans/dim_fleet.json +411 -0
  144. rag_wright/spans/function_classifier.py +77 -0
  145. rag_wright/spans/function_families.py +62 -0
  146. rag_wright/spans/hybrid_classifier.py +103 -0
  147. rag_wright/spans/legalbert_classifier.py +83 -0
  148. rag_wright/spans/model_capabilities.py +107 -0
  149. rag_wright/spans/new_function_labels.py +111 -0
  150. rag_wright/spans/page_map.py +68 -0
  151. rag_wright/spans/property_extractor.py +365 -0
  152. rag_wright/spans/property_grounding.py +182 -0
  153. rag_wright/spans/reclassify.py +77 -0
  154. rag_wright/spans/scarce_function_labels.py +105 -0
  155. rag_wright/spans/segment.py +341 -0
  156. rag_wright/spans/semantic_judge.py +197 -0
  157. rag_wright/spans/symbolic_validation.py +131 -0
  158. rag_wright/spans/tag_clause_extractor.py +182 -0
  159. rag_wright/store/__init__.py +6 -0
  160. rag_wright/store/arcadedb.py +1135 -0
  161. rag_wright/store/chunk_text.py +66 -0
  162. rag_wright/store/seam.py +213 -0
  163. rag_wright/subgraphs/__init__.py +0 -0
  164. rag_wright/subgraphs/async_ingestion.py +204 -0
  165. rag_wright/subgraphs/compliance_check.py +1042 -0
  166. rag_wright/subgraphs/compliance_ingestion.py +306 -0
  167. rag_wright/subgraphs/contract_ingestion_pipeline.py +999 -0
  168. rag_wright/subgraphs/graph_extraction.py +102 -0
  169. rag_wright/subgraphs/intra_document_qa.py +328 -0
  170. rag_wright/subgraphs/observability.py +140 -0
  171. rag_wright/subgraphs/query_constraint_extraction.py +73 -0
  172. rag_wright/subgraphs/relational_qa.py +165 -0
  173. rag_wright/subgraphs/requirement_extraction.py +137 -0
  174. rag_wright/subgraphs/scaffold.py +65 -0
  175. rag_wright/subgraphs/semantic_chunking.py +183 -0
  176. rag_wright/subgraphs/typed_clause_extraction.py +172 -0
  177. rag_wright/subgraphs/typed_property_retrieval.py +278 -0
  178. rag_wright/util/__init__.py +1 -0
  179. rag_wright/util/concurrent.py +153 -0
  180. rag_wright/util/spacy_model.py +45 -0
  181. rag_wright-0.1.0.dist-info/METADATA +168 -0
  182. rag_wright-0.1.0.dist-info/RECORD +184 -0
  183. rag_wright-0.1.0.dist-info/WHEEL +4 -0
  184. rag_wright-0.1.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,85 @@
1
+ """Vision-to-text (FR-C.9, split from `generation` by ADR-0014, T29), SKILL-SPLIT: image/scan -> text.
2
+
3
+ The Gemma 4 class model (GENERAL role) transcribes a scanned filing's images to text, exercising the
4
+ image-only PDF subset (ADR-0002). This is a SINGLE grounded vision-language act -- the ingestion-side twin of
5
+ answer `generation` -- so it is an `agent_skill`, not a function (per the capability-architecture rubric: a
6
+ function is deterministic and takes no model; a single LLM act is an authored skill). The transcription METHOD
7
+ is authored as `skills/vision_to_text/SKILL.md`; `SeamVisionModel` is its runtime, sending the image as an
8
+ OpenAI-compatible multimodal message (a base64 data URI) on the GENERAL role via the model-profile seam -- no
9
+ provider or model flag lives here.
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ import base64
15
+ from pathlib import Path
16
+ from typing import Protocol, runtime_checkable
17
+
18
+ from langchain_core.messages import HumanMessage
19
+ from pydantic import BaseModel
20
+
21
+ from rag_wright.capabilities.registry import CapabilityRegistry
22
+ from rag_wright.models.profiles import ModelRole, model_for
23
+ from rag_wright.models.seam import build_model
24
+
25
+ _SKILL_PATH = Path(__file__).parents[1] / "skills" / "vision_to_text" / "SKILL.md"
26
+
27
+
28
+ class VisionTranscription(BaseModel):
29
+ """The vision-to-text capability's output contract: the transcribed text of an image."""
30
+
31
+ text: str
32
+
33
+
34
+ def transcription_method() -> str:
35
+ """The transcription method (the `vision_to_text` SKILL body, YAML frontmatter stripped) used as the
36
+ vision prompt. Authored knowledge (skills/vision_to_text/SKILL.md), not a hardcoded string."""
37
+ text = _SKILL_PATH.read_text(encoding="utf-8")
38
+ if text.startswith("---"):
39
+ marker = text.find("\n---", 3)
40
+ if marker != -1:
41
+ text = text[marker + 4 :]
42
+ return text.strip()
43
+
44
+
45
+ @runtime_checkable
46
+ class VisionModel(Protocol):
47
+ """The vision seam: transcribe an image's text. `SeamVisionModel` binds it; tests stub it."""
48
+
49
+ def image_to_text(self, image: bytes, *, media_type: str) -> str: ...
50
+
51
+
52
+ class SeamVisionModel:
53
+ """The `vision_to_text` SKILL's runtime: a multimodal call on the GENERAL (Gemma 4 class) model via the
54
+ seam, with the SKILL.md method as the instruction."""
55
+
56
+ def __init__(self, model_id: str | None = None) -> None:
57
+ self._model_id = model_id or model_for(ModelRole.GENERAL)
58
+ self._method = transcription_method()
59
+
60
+ def image_to_text(self, image: bytes, *, media_type: str = "image/png") -> str:
61
+ data_uri = f"data:{media_type};base64,{base64.b64encode(image).decode('ascii')}"
62
+ message = HumanMessage(content=[
63
+ {"type": "text", "text": self._method},
64
+ {"type": "image_url", "image_url": {"url": data_uri}},
65
+ ])
66
+ result = build_model(self._model_id).invoke([message])
67
+ return result.content if hasattr(result, "content") else str(result)
68
+
69
+
70
+ def vision_to_text(image: bytes, *, model: VisionModel, media_type: str = "image/png") -> str:
71
+ """Apply the `vision_to_text` SKILL: transcribe a scanned image to text (ingestion-side, image-only
72
+ filings). `model` is the skill runtime (SeamVisionModel in production; a stub in tests)."""
73
+ return model.image_to_text(image, media_type=media_type)
74
+
75
+
76
+ def register_vision_to_text(registry: CapabilityRegistry) -> None:
77
+ """Register `vision_to_text` as an AGENT_SKILL (FR-C.9, split from `generation` by ADR-0014; SKILL-SPLIT): a
78
+ single grounded vision-language act, authored as `skills/vision_to_text/SKILL.md` and applied via the seam.
79
+ Typed output = `VisionTranscription`."""
80
+ registry.register(
81
+ "vision_to_text",
82
+ contract=VisionTranscription,
83
+ kind="agent_skill",
84
+ display_name="Vision-to-text (scanned-image transcription; authored skill)",
85
+ )
@@ -0,0 +1,85 @@
1
+ """Issue 0009-VLM: VLM-based OCR via a remote vision model -- the escalation for DEGRADED scans that
2
+ character-based OCR cannot read (severe blur / faded ink), where a strong VLM reads the text by language
3
+ context like a human (benchmark: Gemma-4 char_sim 0.991 on the heavy scan vs ~0.01-0.10 for OCR; see
4
+ docs/eval/ocr_benchmark.md).
5
+
6
+ docling's `ApiVlmOptions` points at any OpenAI-compatible endpoint; OpenRouter by default (the same provider we
7
+ already use for Gemma-4 -- no Modal, no local model). The model is the `VISION_OCR` role (default Gemma-4,
8
+ swappable via `RAG_MODEL_VISION_OCR`), so the escalation model is chosen through the model-profile seam, never
9
+ hardcoded (the standing model rule).
10
+ """
11
+ from __future__ import annotations
12
+
13
+ import os
14
+ import tempfile
15
+ from pathlib import Path
16
+ from typing import Any, Optional
17
+
18
+ from rag_wright.models.profiles import ModelRole, model_for
19
+
20
+ _OPENROUTER_URL = "https://openrouter.ai/api/v1/chat/completions"
21
+ _OCR_PROMPT = ("Transcribe ALL text from this document page exactly as it appears, preserving the reading order. "
22
+ "Output only the transcribed text as markdown, with no commentary.")
23
+
24
+
25
+ def openrouter_vlm_options(model_id: Optional[str] = None, *, api_key: Optional[str] = None,
26
+ base_url: str = _OPENROUTER_URL, scale: float = 3.0,
27
+ prompt: Optional[str] = None, timeout: float = 180.0):
28
+ """A docling `ApiVlmOptions` for a remote VLM OCR call. Model defaults to the VISION_OCR role (Gemma-4);
29
+ endpoint is OpenRouter; `scale` renders the page at higher resolution for the model."""
30
+ from docling.datamodel.pipeline_options import ApiVlmOptions, ResponseFormat
31
+
32
+ model_id = model_id or model_for(ModelRole.VISION_OCR)
33
+ key = api_key if api_key is not None else os.environ.get("OPENROUTER_API_KEY", "")
34
+ return ApiVlmOptions(
35
+ url=base_url, headers={"Authorization": f"Bearer {key}"},
36
+ params={"model": model_id, "max_tokens": 8192}, prompt=prompt or _OCR_PROMPT,
37
+ response_format=ResponseFormat.MARKDOWN, scale=scale, timeout=timeout)
38
+
39
+
40
+ def build_vlm_ocr_converter(vlm_options: Any = None, *, model_id: Optional[str] = None, **kw):
41
+ """A docling `DocumentConverter` whose PDF pipeline is the VLM pipeline over `vlm_options` (default: the
42
+ OpenRouter VLM). `enable_remote_services` is set so docling may call the API."""
43
+ from docling.datamodel.base_models import InputFormat
44
+ from docling.datamodel.pipeline_options import VlmPipelineOptions
45
+ from docling.document_converter import DocumentConverter, PdfFormatOption
46
+ from docling.pipeline.vlm_pipeline import VlmPipeline
47
+
48
+ opts = vlm_options or openrouter_vlm_options(model_id=model_id, **kw)
49
+ vopts = VlmPipelineOptions(vlm_options=opts)
50
+ vopts.enable_remote_services = True # required for an API-based VLM
51
+ return DocumentConverter(
52
+ format_options={InputFormat.PDF: PdfFormatOption(pipeline_cls=VlmPipeline, pipeline_options=vopts)})
53
+
54
+
55
+ def vlm_ocr(data: bytes, name: str = "scan.pdf", *, converter: Any = None, model_id: Optional[str] = None) -> str:
56
+ """OCR a document's raw bytes via the VLM -> markdown text. `converter` is injectable (hermetic tests);
57
+ production builds the OpenRouter converter for the VISION_OCR model."""
58
+ conv = converter or build_vlm_ocr_converter(model_id=model_id)
59
+ tmp = Path(tempfile.mkdtemp(prefix="vlmocr_")) / name
60
+ tmp.write_bytes(data)
61
+ return conv.convert(str(tmp)).document.export_to_markdown()
62
+
63
+
64
+ class VlmOCRParser:
65
+ """A `Parser` (convert(source) -> DoclingDocument) backed by the remote VLM converter -- the 0009 escalation
66
+ parser the tiered path routes degraded scans to. The converter is injectable for tests."""
67
+
68
+ def __init__(self, converter: Any = None) -> None:
69
+ self._converter = converter
70
+
71
+ def convert(self, source):
72
+ conv = self._converter or build_vlm_ocr_converter()
73
+ return conv.convert(str(source)).document
74
+
75
+ def parse_range(self, source, page_range):
76
+ """PARSE-3: VLM-parse only pages `page_range` (1-based, inclusive), so the tiered path escalates just the
77
+ degraded pages instead of the whole document."""
78
+ conv = self._converter or build_vlm_ocr_converter()
79
+ return conv.convert(str(source), page_range=page_range).document
80
+
81
+
82
+ def register_vlm_ocr(registry) -> None:
83
+ """0009-VLM: register `vlm_ocr` (function; document bytes -> transcribed text via a remote VLM)."""
84
+ registry.register("vlm_ocr", contract=str, kind="function",
85
+ display_name="VLM OCR (degraded-scan escalation via OpenRouter, default Gemma-4)")
@@ -0,0 +1,6 @@
1
+ """Pydantic contracts: the ontology, the extraction contracts, and the shared
2
+ identifiers (chunk_id per FR-S.2, entity_id per FR-S.3).
3
+
4
+ These contracts double as the registered capability contracts each FR-C capability is
5
+ registered under. Frozen in Phase 3 before implementation so Phase 4 tests assert against them.
6
+ """
@@ -0,0 +1,79 @@
1
+ """The chunk record contract (FR-I.3, FR-S.1).
2
+
3
+ One record per chunk in the hybrid retrieval index. It holds the `chunk_id`, the summary, the
4
+ dense-over-summary vector, the sparse-over-full-text vector, extracted keywords and entities, and
5
+ source metadata. The record does not hold the raw full chunk text: FR-I.3 enumerates the summary
6
+ plus vectors. The text is used once at chunking to compute the `chunk_id` content hash, then persisted
7
+ to the chunk-text sidecar (`store/chunk_text.py`, T40) keyed by `chunk_id` and rehydrated at query time
8
+ by `chunk_read` (T38); it is deliberately kept out of the index to keep the index dense-over-summary.
9
+
10
+ The load-bearing part of this contract is the vector shapes. They must match what BGE-M3 produces
11
+ (the embedding capability, T19) and what the ArcadeDB dense `LSM_VECTOR` and sparse
12
+ `LSM_SPARSE_VECTOR` indexes bind (the store seam, T13), so the dense dimension and the sparse
13
+ key/value ranges are validated here.
14
+ """
15
+
16
+ from __future__ import annotations
17
+
18
+ import math
19
+
20
+ from pydantic import BaseModel, field_validator
21
+
22
+ from rag_wright.contracts.identifiers import ChunkId
23
+
24
+ # BAAI/bge-m3 dense embedding dimension. The dense vector is computed over the summary (FR-I.3);
25
+ # the ArcadeDB dense LSM_VECTOR index (T13) binds this dimension.
26
+ BGE_M3_DENSE_DIM = 1024
27
+
28
+ # Metadata values are kept to filterable JSON scalars so the store can index and filter on them
29
+ # (metadata filters, FR-Q.1).
30
+ MetadataValue = str | int | float | bool
31
+
32
+
33
+ class ChunkRecord(BaseModel):
34
+ """A chunk's record in the retrieval index (FR-I.3, FR-S.1)."""
35
+
36
+ chunk_id: ChunkId
37
+ summary: str
38
+ dense_vector: list[float] # dense over the summary; length == BGE_M3_DENSE_DIM
39
+ sparse_vector: dict[int, float] # sparse over full text: token-id index -> non-negative weight
40
+ keywords: list[str] = []
41
+ # Unresolved surface-form entity mentions for retrieval metadata only. These are NOT resolved
42
+ # and must NOT be joined to the graph's canonical entity_ids; resolution to entity_id happens
43
+ # later (FR-C.7) and canonical entities live on graph nodes (T24). Naming this entity_mentions
44
+ # (not entities) keeps the pre-resolution side of the fragmentation seam unambiguous.
45
+ entity_mentions: list[str] = []
46
+ source_metadata: dict[str, MetadataValue] = {}
47
+
48
+ @field_validator("summary")
49
+ @classmethod
50
+ def _summary_non_empty(cls, v: str) -> str:
51
+ if not v.strip():
52
+ raise ValueError("summary must be non-empty")
53
+ return v
54
+
55
+ @field_validator("dense_vector")
56
+ @classmethod
57
+ def _dense_shape(cls, v: list[float]) -> list[float]:
58
+ if len(v) != BGE_M3_DENSE_DIM:
59
+ raise ValueError(f"dense_vector must have length {BGE_M3_DENSE_DIM}, got {len(v)}")
60
+ if not all(math.isfinite(x) for x in v):
61
+ raise ValueError("dense_vector must contain only finite values")
62
+ return v
63
+
64
+ @field_validator("sparse_vector")
65
+ @classmethod
66
+ def _sparse_ranges(cls, v: dict[int, float]) -> dict[int, float]:
67
+ for token_id, weight in v.items():
68
+ if token_id < 0:
69
+ raise ValueError(f"sparse_vector token ids must be >= 0, got {token_id}")
70
+ if not math.isfinite(weight) or weight < 0:
71
+ raise ValueError(f"sparse_vector weights must be finite and >= 0, got {weight}")
72
+ return v
73
+
74
+ @field_validator("keywords", "entity_mentions")
75
+ @classmethod
76
+ def _no_blank_items(cls, v: list[str]) -> list[str]:
77
+ if any(not item.strip() for item in v):
78
+ raise ValueError("keywords and entity_mentions must not contain blank strings")
79
+ return v
@@ -0,0 +1,303 @@
1
+ """Compliance-module contracts (CC-1, roadmap §13.1): `Requirement` and `Claim`.
2
+
3
+ The compliance check is two-sided retrieval + entailment. The REGULATORY side ingests into `Requirement`
4
+ nodes (a single deontic rule); the SUBJECT side extracts `Claim` nodes (a checkable ad assertion). CC-6's
5
+ applicability match (claim -> applicable requirements) reuses the Leg-B retrieval router, so a `Constraint`
6
+ here is exactly the `(dimension, value)` pair the router matches on (`applicability_scope` <-> a claim's
7
+ scope). Provenance + confidence on everything (FR-S.4); no claim without a citation (FR-Q.6).
8
+
9
+ The vocab is the thin AUTHORED advertising layer (closed `DeonticType`/`ClaimType`/`Severity`), grounded on
10
+ the public deontic backbone (ODRL/LKIF) in the sibling `compliance_bridge.ttl`. These contracts double as the
11
+ registered capability contracts (`requirement_extraction`, `claim_extraction`, ...).
12
+ """
13
+
14
+ from __future__ import annotations
15
+
16
+ import hashlib
17
+ from enum import Enum
18
+ from pathlib import Path
19
+ from typing import Literal, Optional
20
+
21
+ from pydantic import BaseModel, ConfigDict, field_validator, model_validator
22
+
23
+ from rag_wright.contracts.identifiers import canonical_source_doc_id
24
+ from rag_wright.contracts.provenance import ConfidenceTag
25
+
26
+ # the sibling ontology (CC-1): public deontic backbone (ODRL/LKIF) + PROV + the thin authored ad vocab
27
+ BRIDGE_TTL_PATH = Path(__file__).resolve().parent.parent / "ontology" / "compliance_bridge.ttl"
28
+
29
+
30
+ # ADR-0066 P3b: the closed vocabularies below (DeonticType / ClaimType / Severity / RuleScope / Verdict) are
31
+ # AUTHORITATIVE in compliance_bridge.ttl (owl:oneOf). To change one, edit the ttl -- these enums are drift-locked
32
+ # to it by tests/ontology/test_compliance_ontology_authoritative.py.
33
+ class DeonticType(str, Enum):
34
+ """The rule's deontic force (LKIF/ODRL closed vocab): what it obliges, forbids, or permits."""
35
+
36
+ OBLIGATION = "obligation"
37
+ PROHIBITION = "prohibition"
38
+ PERMISSION = "permission"
39
+
40
+
41
+ class ClaimType(str, Enum):
42
+ """The kind of checkable assertion an ad makes (closed authored vocab, roadmap §13.1)."""
43
+
44
+ EFFICACY = "efficacy"
45
+ COMPARATIVE = "comparative"
46
+ PRICING = "pricing"
47
+ HEALTH = "health"
48
+ ENVIRONMENTAL = "environmental"
49
+ ENDORSEMENT = "endorsement"
50
+ PERFORMANCE = "performance"
51
+ GUARANTEE = "guarantee"
52
+
53
+
54
+ class Severity(str, Enum):
55
+ """Optional severity of a requirement (drives triage, not the verdict)."""
56
+
57
+ LOW = "low"
58
+ MEDIUM = "med"
59
+ HIGH = "high"
60
+
61
+
62
+ class Constraint(BaseModel):
63
+ """One applicability condition, a `(dimension, value)` pair. The shape is deliberately identical to the
64
+ retrieval router's constraint tuple (`property_boosted_retrieval` / `query_function_classifier`), so
65
+ CC-6's claim -> applicable-requirement match reuses Leg-B rather than a new matcher."""
66
+
67
+ model_config = ConfigDict(frozen=True)
68
+
69
+ dimension: str
70
+ value: str
71
+
72
+ def as_tuple(self) -> tuple[str, str]:
73
+ return (self.dimension, self.value)
74
+
75
+
76
+ def _nonblank(v: str, what: str) -> str:
77
+ if not isinstance(v, str) or not v.strip():
78
+ raise ValueError(f"{what} must be a non-empty string")
79
+ return v.strip()
80
+
81
+
82
+ def _content_id(source: str, section: str, text: str) -> str:
83
+ """`<canonical-source>:<section>:<hash16>` -- deterministic content-hash id (RAC-1), the clause_id
84
+ scheme generalized (source-unit + locator + content hash), so an unchanged rule/claim keeps its id."""
85
+ digest = hashlib.sha256(text.encode("utf-8")).hexdigest()[:16]
86
+ return f"{canonical_source_doc_id(source)}:{section}:{digest}"
87
+
88
+
89
+ class Requirement(BaseModel):
90
+ """A single regulatory rule extracted from the requirements corpus (roadmap §13.1).
91
+
92
+ `requirement_id = <source_reg>:<section>:<hash>`. Carries its `citation` (always cited, FR-Q.6) and a
93
+ `confidence` tag (graph-derived fact, FR-S.4). `applicability_scope` is matched against a claim's scope.
94
+ """
95
+
96
+ requirement_id: str
97
+ source: str
98
+ citation: str # section / paragraph -- the human-readable provenance, always present
99
+ deontic_type: DeonticType
100
+ actor: str # who it binds: advertiser, endorser, ... (open, matched against a claim's actor)
101
+ applicability_scope: list[Constraint] = []
102
+ requirement_text: str
103
+ evidence_standard: str | None = None
104
+ trigger_condition: str | None = None
105
+ severity: Severity | None = None
106
+ # issue 0043: page provenance for the POLICY side, in the same shape span provenance uses (issue 0032), so one
107
+ # product code path serves both sides of a finding. `citation` stays the always-present human-readable
108
+ # provenance; these are additive. A requirement is bound to a policy SECTION, so `pages` are that section's
109
+ # source page(s) (from the parse's per-item provenance -- present on scans, where a text search would fail
110
+ # silently). `bbox` is best-effort (`(l, t, r, b)`) and usually None for a multi-item section; never fabricated.
111
+ pages: list[int] = []
112
+ bbox: tuple[float, float, float, float] | None = None
113
+ confidence: ConfidenceTag = ConfidenceTag.EXTRACTED
114
+ defenses: list[str] = [] # DEON-9 (query-time only, never persisted): same-source PERMISSIONS linked as
115
+ # carve-outs/exceptions that may EXCUSE this O/F rule -- passed to the judge as structured context so a
116
+ # legitimate exception is not a false violation (ADR-0044 pattern, requirement side).
117
+
118
+ @field_validator("requirement_id", "source", "citation", "requirement_text")
119
+ @classmethod
120
+ def _required_nonblank(cls, v: str, info) -> str:
121
+ return _nonblank(v, info.field_name)
122
+
123
+ @staticmethod
124
+ def make_id(source: str, section: str, requirement_text: str) -> str:
125
+ return _content_id(source, section, requirement_text)
126
+
127
+
128
+ class CheckableFact(BaseModel):
129
+ """COMP-VERDICT-GENERIC: the DOMAIN-AGNOSTIC subject-fact the compliance verdict core consumes -- a checkable
130
+ assertion (any domain) with its provenance. The verdict machinery (semantic retrieval + LLM judge + finding /
131
+ report) needs only THIS (an id + text + provenance); advertising `Claim` is a SPECIALIZATION that adds typed
132
+ claim fields for structured routing. A NEW compliance domain either uses a bare `CheckableFact` (generic
133
+ verdict) or subclasses this with its own enrichment -- WITHOUT touching the base or other domains.
134
+
135
+ `fact_id = <source_doc>:<index>:<hash>`; `(source_doc, doc_start, doc_end, assertion_text)` is the span
136
+ provenance (cited, FR-Q.6); `confidence` is the graph-derived tag (FR-S.4). NB: never persisted -- a
137
+ query-time object only (the KG stores `Requirement`), so this shape is a pure query-side/contract concern."""
138
+
139
+ fact_id: str
140
+ source_doc: str
141
+ assertion_text: str # the checkable statement text (domain-neutral: an "assertion" is any checkable claim/fact)
142
+ # issue 0044: JUDGE-ONLY structured signals (e.g. DEON-8's ad-level disclosure/evidence union) rendered as
143
+ # document context FOR THE JUDGE but NOT part of the citation -- so engine scaffolding never surfaces as a
144
+ # quote from the user's document. The judge appends this; `assemble_finding`'s citation uses `assertion_text`
145
+ # only. Empty for a plain fact (domain-neutral: the generic path carries none).
146
+ document_signals: str = ""
147
+ # issue 0044: whether `assertion_text` is one VERBATIM span from the document, or ASSEMBLED evidence (the
148
+ # obligation path joins the top-N relevant spans with "\n\n"). Copied onto the finding so a consumer can tell
149
+ # a verbatim quote from a synthesised excerpt WITHOUT parsing prose, and render/attribute it accordingly.
150
+ citation_kind: Literal["verbatim", "assembled"] = "verbatim"
151
+ section: str | None = None # UNIFY-A: the section/heading locator this fact came from (e.g. "4.2"); the
152
+ # finding cites "doc § {section}: {assertion}" when set. Additive/optional: None -> the old "doc: assertion".
153
+ element_kind: str | None = None # SEG-1: docling structural label of the source element (paragraph /
154
+ # list_item / section_header / ...); selects the within-section marker in `locator()`.
155
+ element_ordinal: int | None = None # SEG-1: the element's within-section ordinal (the ¶ / bullet number).
156
+ scope: list["Constraint"] = [] # DEON-5: the assertion's inferred applicability constraints (e.g.
157
+ # Constraint("actor", "endorser")); dimension-agnostic, matched against a requirement's applicability_scope by
158
+ # the generic `constraint_applies` router (DEON-6), and aggregated to the document's SubjectScope (DEON-7).
159
+ doc_start: int | None = None # span provenance: char offsets in source_doc (optional)
160
+ doc_end: int | None = None
161
+ confidence: ConfidenceTag = ConfidenceTag.EXTRACTED
162
+
163
+ def locator(self) -> str:
164
+ """SEG-1: the human structural locator -- `§ {section}` plus a within-section element marker when present
165
+ (`¶N` for a paragraph, `· bullet N` for a list item). Empty string when the fact has no section (e.g. a
166
+ structureless paste), so the finding cites just `doc: {assertion}`. The citation is built from this."""
167
+ if not (self.section and self.section.strip()):
168
+ return ""
169
+ loc = f"§ {self.section}"
170
+ if self.element_ordinal is not None:
171
+ if self.element_kind == "list_item":
172
+ loc += f" · bullet {self.element_ordinal}"
173
+ else: # paragraph / text / default
174
+ loc += f" ¶{self.element_ordinal}"
175
+ return loc
176
+
177
+ @field_validator("fact_id", "source_doc", "assertion_text")
178
+ @classmethod
179
+ def _required_nonblank(cls, v: str, info) -> str:
180
+ return _nonblank(v, info.field_name)
181
+
182
+ @model_validator(mode="after")
183
+ def _ordered_offsets(self) -> CheckableFact:
184
+ if self.doc_start is not None and self.doc_end is not None and self.doc_start >= self.doc_end:
185
+ raise ValueError(f"doc_start ({self.doc_start}) must be < doc_end ({self.doc_end})")
186
+ return self
187
+
188
+ @staticmethod
189
+ def make_id(source_doc: str, index: int, text: str) -> str:
190
+ return _content_id(source_doc, str(index), text)
191
+
192
+
193
+ class Claim(CheckableFact):
194
+ """An ADVERTISING checkable element (roadmap §13.1) -- a `CheckableFact` SPECIALIZED with the typed claim
195
+ fields the advertising compliance path uses for structured routing (claim_type) + disclosure/substantiation
196
+ judging. Inherits id/text/provenance + validators + `make_id` from `CheckableFact`."""
197
+
198
+ claim_type: ClaimType
199
+ actor: str | None = None
200
+ subject_product: str | None = None
201
+ quantitative_value: str | None = None
202
+ disclosures_present: list[str] = []
203
+ evidence_referenced: bool = False
204
+ medium: str | None = None
205
+
206
+
207
+ class RuleScope(str, Enum):
208
+ """How a requirement is narrowed at check time (CC-8a, the ontology routing tag; `compliance_bridge.ttl`
209
+ cmp:RuleScope). CONTENT rules (substantiation / claim-specific) narrow by semantic similarity to the claim;
210
+ CONTEXT rules (disclosure / material connection -- apply to ANY claim in an endorsement regardless of its
211
+ content) are ALWAYS included, never left to similarity. See [[ontology-lever-vs-extraction-lever]]."""
212
+
213
+ CONTENT = "content"
214
+ CONTEXT = "context"
215
+
216
+
217
+ class Verdict(str, Enum):
218
+ """The compliance judgment output (CC-4, §13.2). Closed vocab; matches `compliance_bridge.ttl` cmp:Verdict.
219
+ `needs_review` is the conservative default under uncertainty (never a silent compliant/violation)."""
220
+
221
+ COMPLIANT = "compliant"
222
+ VIOLATION = "violation"
223
+ NEEDS_REVIEW = "needs_review"
224
+
225
+
226
+ class ComplianceFinding(BaseModel):
227
+ """One `(claim, requirement)` judgment (CC-4, §13.2): a verdict + rationale + BOTH-SIDED citation +
228
+ confidence. The citations are the trust product -- an auditor sees the exact ad span AND the exact reg
229
+ clause. Every `violation`/`needs_review` is human-gated (`needs_human_review`); false-negatives are
230
+ liability, so uncertainty never silently clears."""
231
+
232
+ claim_id: str
233
+ requirement_id: str
234
+ verdict: Verdict
235
+ rationale: str = ""
236
+ citation_claim: str # the subject span text (provenance, cited -- FR-Q.6)
237
+ citation_requirement: str # the reg clause / section (provenance, cited)
238
+ # issue 0044: is `citation_claim` one VERBATIM span from the document, or ASSEMBLED evidence (the obligation
239
+ # path cites the top-N relevant spans joined by "\n\n")? A consumer renders assembled evidence differently
240
+ # instead of quoting it as the user's exact words. Default "verbatim" -> unchanged for every existing path.
241
+ citation_claim_kind: Literal["verbatim", "assembled"] = "verbatim"
242
+ confidence: float = 0.0
243
+
244
+ @field_validator("confidence")
245
+ @classmethod
246
+ def _confidence_in_unit(cls, v: float) -> float:
247
+ if not 0.0 <= v <= 1.0:
248
+ raise ValueError(f"confidence must be in [0, 1], got {v}")
249
+ return v
250
+
251
+ @property
252
+ def needs_human_review(self) -> bool:
253
+ """Every violation requires human confirmation before it leaves the tool; needs_review always does.
254
+ A `compliant` finding does not gate (§13.2)."""
255
+ return self.verdict in (Verdict.VIOLATION, Verdict.NEEDS_REVIEW)
256
+
257
+
258
+ _AD_VIOLATION_THRESHOLD = 2 # a lone violation finding among many rules escalates, not hard-flags (RG-5 aggregation)
259
+
260
+
261
+ class RequirementLocation(BaseModel):
262
+ """Where a curated requirement sits in its policy document, for a citation preview (EP-REF-1b): the
263
+ requirement id, its citation label, the source page number(s), an optional `[l, t, r, b]` rectangle, and
264
+ the rule text. `pages`/`bbox` are best-effort -- a requirement curated before provenance landed (ADR-0107)
265
+ reads back with `pages=[]` / `bbox=None`, which a preview renders honestly rather than erroring."""
266
+
267
+ requirement_id: str
268
+ citation: str
269
+ pages: list[int] = []
270
+ bbox: Optional[tuple[float, float, float, float]] = None
271
+ text: str
272
+
273
+
274
+ class ComplianceReport(BaseModel):
275
+ """The `compliance_check` output (CC-6, §13.3): a subject document's cited findings + a per-requirement gap
276
+ matrix + a verdict summary. Both-sided cited; every violation/needs_review is human-gated per finding."""
277
+
278
+ source_doc: str
279
+ findings: list[ComplianceFinding] = []
280
+ summary: dict[str, int] = {} # verdict -> count (compliant/violation/needs_review)
281
+ gap_matrix: list[dict] = [] # per-requirement rollup: {requirement_id, citation, verdict, claims_checked}
282
+ # SEG-6 (0009-WIRE2 / ENG-1 principle): pages the tiered OCR could not read even after VLM escalation, so a
283
+ # verdict on a scanned subject is never SILENTLY based on half-read text. Empty = fully readable. The product
284
+ # surfaces this as "pages X-Y unreadable; results for those pages are incomplete".
285
+ ocr_unreadable_pages: list[int] = []
286
+ # ADR-0068 (engine issue 0013): (assertion, rule) pairs the symbolic ACTOR gate SKIPPED before any judge call
287
+ # -- so the gate is never a SILENT recall loss. Each: {requirement_id, citation, actor, subject_actors, scope
288
+ # ("assertion"|"document"), claim_id}. Empty is the norm (the recall-first gate skips only ontology-disjoint
289
+ # roles); a non-empty list lets a consumer state honest coverage ("checked N rules; K pairs gated by role").
290
+ gated_pairs: list[dict] = []
291
+
292
+ @property
293
+ def verdict(self) -> Verdict:
294
+ """The AD-LEVEL verdict rolled up from the findings (RG-5): VIOLATION only when violation findings are a
295
+ real signal (>= threshold, so one spurious finding among many rules does not hard-flag); else
296
+ NEEDS_REVIEW if anything fired (a lone violation OR any needs_review -> escalate for a human); else
297
+ COMPLIANT. It never CLEARS an ad that had a violation finding -- a real violation is never a silent pass."""
298
+ v = self.summary.get("violation", 0)
299
+ if v >= _AD_VIOLATION_THRESHOLD:
300
+ return Verdict.VIOLATION
301
+ if v or self.summary.get("needs_review", 0):
302
+ return Verdict.NEEDS_REVIEW
303
+ return Verdict.COMPLIANT
@@ -0,0 +1,27 @@
1
+ """Contract-level metadata record (CU-A1, ADR-0029).
2
+
3
+ The CUAD pipeline is document-scoped: a contract is LOOKED UP by id (not searched), then all retrieval happens
4
+ within it. `ContractRecord` is that lookup/filter unit -- the contract node the `Span` records point back to
5
+ via `contract_id`. Metadata fields are best-effort (many come from a value-type extraction pass, CU-C2) and
6
+ default empty; only `contract_id` is required.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ from pydantic import BaseModel
12
+
13
+
14
+ class ContractRecord(BaseModel):
15
+ """One contract's metadata for lookup + filtering (the parent of its clauses/spans)."""
16
+
17
+ model_config = {"frozen": True}
18
+
19
+ contract_id: str # the canonical document id (== spans' contract_id / the parse source_doc_id)
20
+ name: str = "" # the contract's name/title (CUAD "Document Name")
21
+ agreement_type: str = "" # e.g. "Distributor Agreement"
22
+ parties: list[str] = [] # signing parties (CUAD "Parties")
23
+ agreement_date: str = "" # as-written date string (not normalized here)
24
+ effective_date: str = ""
25
+ source_doc_id: str = "" # the parse source id (usually == contract_id)
26
+ content_hash: str = "" # source content hash (provenance / idempotence)
27
+ page_count: int | None = None # for PDF-overlay citation, when known