rag-wright 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- rag_wright/__init__.py +13 -0
- rag_wright/api/__init__.py +33 -0
- rag_wright/api/config.py +59 -0
- rag_wright/api/discover.py +70 -0
- rag_wright/api/documents.py +39 -0
- rag_wright/api/ids.py +31 -0
- rag_wright/api/invoke.py +99 -0
- rag_wright/api/kg.py +61 -0
- rag_wright/api/mcp.py +94 -0
- rag_wright/api/usage.py +30 -0
- rag_wright/api/workspace.py +85 -0
- rag_wright/capabilities/__init__.py +8 -0
- rag_wright/capabilities/answer_generator.py +427 -0
- rag_wright/capabilities/ard.py +286 -0
- rag_wright/capabilities/assertion_extraction.py +79 -0
- rag_wright/capabilities/chunk_read.py +58 -0
- rag_wright/capabilities/chunk_write.py +163 -0
- rag_wright/capabilities/claim_extraction.py +153 -0
- rag_wright/capabilities/clause_exception_linking.py +117 -0
- rag_wright/capabilities/compliance_judgment.py +322 -0
- rag_wright/capabilities/compliance_store.py +87 -0
- rag_wright/capabilities/contract_kg_serve.py +156 -0
- rag_wright/capabilities/contract_kg_store.py +251 -0
- rag_wright/capabilities/dg_extraction.py +585 -0
- rag_wright/capabilities/disambiguation.py +163 -0
- rag_wright/capabilities/document_parse.py +87 -0
- rag_wright/capabilities/document_scope.py +49 -0
- rag_wright/capabilities/embedding.py +164 -0
- rag_wright/capabilities/embedding_profiles.py +43 -0
- rag_wright/capabilities/entity_resolution.py +154 -0
- rag_wright/capabilities/fusion.py +64 -0
- rag_wright/capabilities/graph_extraction.py +243 -0
- rag_wright/capabilities/graph_query.py +73 -0
- rag_wright/capabilities/graph_storage.py +111 -0
- rag_wright/capabilities/highlight_serve.py +142 -0
- rag_wright/capabilities/hybrid_search.py +65 -0
- rag_wright/capabilities/invoke.py +31 -0
- rag_wright/capabilities/jev_decision.py +38 -0
- rag_wright/capabilities/manifests.py +872 -0
- rag_wright/capabilities/okf_navigate.py +456 -0
- rag_wright/capabilities/parsing.py +286 -0
- rag_wright/capabilities/property_boosted_retrieval.py +125 -0
- rag_wright/capabilities/query_function_classifier.py +94 -0
- rag_wright/capabilities/query_understanding.py +109 -0
- rag_wright/capabilities/registry.py +262 -0
- rag_wright/capabilities/remote_encoders.py +94 -0
- rag_wright/capabilities/requirement_extraction.py +247 -0
- rag_wright/capabilities/reranking.py +123 -0
- rag_wright/capabilities/retrieval_core.py +126 -0
- rag_wright/capabilities/rlm_chunking.py +808 -0
- rag_wright/capabilities/rlm_synthesis.py +316 -0
- rag_wright/capabilities/scan_quality.py +136 -0
- rag_wright/capabilities/span_relevance_judgment.py +191 -0
- rag_wright/capabilities/vision_to_text.py +85 -0
- rag_wright/capabilities/vlm_ocr.py +85 -0
- rag_wright/contracts/__init__.py +6 -0
- rag_wright/contracts/chunk.py +79 -0
- rag_wright/contracts/compliance.py +303 -0
- rag_wright/contracts/contract_meta.py +27 -0
- rag_wright/contracts/extraction.py +130 -0
- rag_wright/contracts/function.py +167 -0
- rag_wright/contracts/function_routing.py +91 -0
- rag_wright/contracts/highlight.py +74 -0
- rag_wright/contracts/identifiers.py +153 -0
- rag_wright/contracts/jurisdiction.py +96 -0
- rag_wright/contracts/ontology.py +142 -0
- rag_wright/contracts/property.py +201 -0
- rag_wright/contracts/provenance.py +78 -0
- rag_wright/contracts/query_intent.py +53 -0
- rag_wright/contracts/span.py +76 -0
- rag_wright/contracts/value_match.py +84 -0
- rag_wright/corpus/__init__.py +0 -0
- rag_wright/corpus/canonicalize.py +116 -0
- rag_wright/corpus/cuad.py +153 -0
- rag_wright/corpus/cuad_ingestion.py +72 -0
- rag_wright/corpus/document_parser.py +299 -0
- rag_wright/corpus/edgar.py +231 -0
- rag_wright/corpus/gcs_ingestion.py +120 -0
- rag_wright/corpus/http.py +110 -0
- rag_wright/corpus/selection.py +152 -0
- rag_wright/mcp/__init__.py +11 -0
- rag_wright/mcp/compliance_server.py +299 -0
- rag_wright/mcp/intra_document_qa_server.py +170 -0
- rag_wright/mcp/relational_qa_server.py +171 -0
- rag_wright/mcp/session_store.py +64 -0
- rag_wright/mcp/typed_property_retrieval_server.py +191 -0
- rag_wright/models/__init__.py +8 -0
- rag_wright/models/profiles.py +331 -0
- rag_wright/models/seam.py +497 -0
- rag_wright/models/tag_structured.py +285 -0
- rag_wright/models/tracing.py +179 -0
- rag_wright/models/usage.py +102 -0
- rag_wright/okf/__init__.py +11 -0
- rag_wright/okf/compile.py +292 -0
- rag_wright/okf/document.py +47 -0
- rag_wright/okf/enrich.py +176 -0
- rag_wright/okf/links.py +190 -0
- rag_wright/okf/lint.py +105 -0
- rag_wright/ontology/__init__.py +6 -0
- rag_wright/ontology/_generated_template_meta.py +60 -0
- rag_wright/ontology/_generated_vocab.py +52 -0
- rag_wright/ontology/clause_template.py +964 -0
- rag_wright/ontology/codegen.py +84 -0
- rag_wright/ontology/compliance_bridge.ttl +186 -0
- rag_wright/ontology/contract_bridge.ttl +2685 -0
- rag_wright/ontology/contract_taxonomy.py +24 -0
- rag_wright/ontology/derive.py +58 -0
- rag_wright/ontology/loader.py +435 -0
- rag_wright/ontology/packs/ftc_16cfr255.ttl +29 -0
- rag_wright/ontology/registry.py +87 -0
- rag_wright/ontology/template_introspect.py +100 -0
- rag_wright/py.typed +0 -0
- rag_wright/reference/__init__.py +2 -0
- rag_wright/reference/compliance.py +41 -0
- rag_wright/reference/contract_seam.py +123 -0
- rag_wright/skills/__init__.py +7 -0
- rag_wright/skills/claim_extraction/SKILL.md +47 -0
- rag_wright/skills/claim_extraction/__init__.py +1 -0
- rag_wright/skills/claim_extraction/template.py +50 -0
- rag_wright/skills/compliance_judgment/SKILL.md +59 -0
- rag_wright/skills/corpus_ingest/SKILL.md +106 -0
- rag_wright/skills/extraction_semantic_judge/SKILL.md +51 -0
- rag_wright/skills/extraction_semantic_judge/__init__.py +1 -0
- rag_wright/skills/generation/SKILL.md +64 -0
- rag_wright/skills/generation/__init__.py +1 -0
- rag_wright/skills/generic_compliance_judgment/SKILL.md +58 -0
- rag_wright/skills/okf_navigate/SKILL.md +137 -0
- rag_wright/skills/requirement_extraction/SKILL.md +47 -0
- rag_wright/skills/requirement_extraction/__init__.py +1 -0
- rag_wright/skills/requirement_extraction/template.py +50 -0
- rag_wright/skills/rlm/SKILL.md +186 -0
- rag_wright/skills/rlm/__init__.py +31 -0
- rag_wright/skills/rlm/agent.py +292 -0
- rag_wright/skills/span_relevance_judgment/SKILL.md +67 -0
- rag_wright/skills/vision_to_text/SKILL.md +36 -0
- rag_wright/skills/vision_to_text/__init__.py +1 -0
- rag_wright/spans/__init__.py +1 -0
- rag_wright/spans/boundary.py +78 -0
- rag_wright/spans/clause_function_classifier.py +490 -0
- rag_wright/spans/clause_kg_extractor.py +337 -0
- rag_wright/spans/cuad_labels.py +81 -0
- rag_wright/spans/dim_classifier.py +158 -0
- rag_wright/spans/dim_fleet.json +411 -0
- rag_wright/spans/function_classifier.py +77 -0
- rag_wright/spans/function_families.py +62 -0
- rag_wright/spans/hybrid_classifier.py +103 -0
- rag_wright/spans/legalbert_classifier.py +83 -0
- rag_wright/spans/model_capabilities.py +107 -0
- rag_wright/spans/new_function_labels.py +111 -0
- rag_wright/spans/page_map.py +68 -0
- rag_wright/spans/property_extractor.py +365 -0
- rag_wright/spans/property_grounding.py +182 -0
- rag_wright/spans/reclassify.py +77 -0
- rag_wright/spans/scarce_function_labels.py +105 -0
- rag_wright/spans/segment.py +341 -0
- rag_wright/spans/semantic_judge.py +197 -0
- rag_wright/spans/symbolic_validation.py +131 -0
- rag_wright/spans/tag_clause_extractor.py +182 -0
- rag_wright/store/__init__.py +6 -0
- rag_wright/store/arcadedb.py +1135 -0
- rag_wright/store/chunk_text.py +66 -0
- rag_wright/store/seam.py +213 -0
- rag_wright/subgraphs/__init__.py +0 -0
- rag_wright/subgraphs/async_ingestion.py +204 -0
- rag_wright/subgraphs/compliance_check.py +1042 -0
- rag_wright/subgraphs/compliance_ingestion.py +306 -0
- rag_wright/subgraphs/contract_ingestion_pipeline.py +999 -0
- rag_wright/subgraphs/graph_extraction.py +102 -0
- rag_wright/subgraphs/intra_document_qa.py +328 -0
- rag_wright/subgraphs/observability.py +140 -0
- rag_wright/subgraphs/query_constraint_extraction.py +73 -0
- rag_wright/subgraphs/relational_qa.py +165 -0
- rag_wright/subgraphs/requirement_extraction.py +137 -0
- rag_wright/subgraphs/scaffold.py +65 -0
- rag_wright/subgraphs/semantic_chunking.py +183 -0
- rag_wright/subgraphs/typed_clause_extraction.py +172 -0
- rag_wright/subgraphs/typed_property_retrieval.py +278 -0
- rag_wright/util/__init__.py +1 -0
- rag_wright/util/concurrent.py +153 -0
- rag_wright/util/spacy_model.py +45 -0
- rag_wright-0.1.0.dist-info/METADATA +168 -0
- rag_wright-0.1.0.dist-info/RECORD +184 -0
- rag_wright-0.1.0.dist-info/WHEEL +4 -0
- rag_wright-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
"""Vision-to-text (FR-C.9, split from `generation` by ADR-0014, T29), SKILL-SPLIT: image/scan -> text.
|
|
2
|
+
|
|
3
|
+
The Gemma 4 class model (GENERAL role) transcribes a scanned filing's images to text, exercising the
|
|
4
|
+
image-only PDF subset (ADR-0002). This is a SINGLE grounded vision-language act -- the ingestion-side twin of
|
|
5
|
+
answer `generation` -- so it is an `agent_skill`, not a function (per the capability-architecture rubric: a
|
|
6
|
+
function is deterministic and takes no model; a single LLM act is an authored skill). The transcription METHOD
|
|
7
|
+
is authored as `skills/vision_to_text/SKILL.md`; `SeamVisionModel` is its runtime, sending the image as an
|
|
8
|
+
OpenAI-compatible multimodal message (a base64 data URI) on the GENERAL role via the model-profile seam -- no
|
|
9
|
+
provider or model flag lives here.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import base64
|
|
15
|
+
from pathlib import Path
|
|
16
|
+
from typing import Protocol, runtime_checkable
|
|
17
|
+
|
|
18
|
+
from langchain_core.messages import HumanMessage
|
|
19
|
+
from pydantic import BaseModel
|
|
20
|
+
|
|
21
|
+
from rag_wright.capabilities.registry import CapabilityRegistry
|
|
22
|
+
from rag_wright.models.profiles import ModelRole, model_for
|
|
23
|
+
from rag_wright.models.seam import build_model
|
|
24
|
+
|
|
25
|
+
_SKILL_PATH = Path(__file__).parents[1] / "skills" / "vision_to_text" / "SKILL.md"
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
class VisionTranscription(BaseModel):
|
|
29
|
+
"""The vision-to-text capability's output contract: the transcribed text of an image."""
|
|
30
|
+
|
|
31
|
+
text: str
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def transcription_method() -> str:
|
|
35
|
+
"""The transcription method (the `vision_to_text` SKILL body, YAML frontmatter stripped) used as the
|
|
36
|
+
vision prompt. Authored knowledge (skills/vision_to_text/SKILL.md), not a hardcoded string."""
|
|
37
|
+
text = _SKILL_PATH.read_text(encoding="utf-8")
|
|
38
|
+
if text.startswith("---"):
|
|
39
|
+
marker = text.find("\n---", 3)
|
|
40
|
+
if marker != -1:
|
|
41
|
+
text = text[marker + 4 :]
|
|
42
|
+
return text.strip()
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
@runtime_checkable
|
|
46
|
+
class VisionModel(Protocol):
|
|
47
|
+
"""The vision seam: transcribe an image's text. `SeamVisionModel` binds it; tests stub it."""
|
|
48
|
+
|
|
49
|
+
def image_to_text(self, image: bytes, *, media_type: str) -> str: ...
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
class SeamVisionModel:
|
|
53
|
+
"""The `vision_to_text` SKILL's runtime: a multimodal call on the GENERAL (Gemma 4 class) model via the
|
|
54
|
+
seam, with the SKILL.md method as the instruction."""
|
|
55
|
+
|
|
56
|
+
def __init__(self, model_id: str | None = None) -> None:
|
|
57
|
+
self._model_id = model_id or model_for(ModelRole.GENERAL)
|
|
58
|
+
self._method = transcription_method()
|
|
59
|
+
|
|
60
|
+
def image_to_text(self, image: bytes, *, media_type: str = "image/png") -> str:
|
|
61
|
+
data_uri = f"data:{media_type};base64,{base64.b64encode(image).decode('ascii')}"
|
|
62
|
+
message = HumanMessage(content=[
|
|
63
|
+
{"type": "text", "text": self._method},
|
|
64
|
+
{"type": "image_url", "image_url": {"url": data_uri}},
|
|
65
|
+
])
|
|
66
|
+
result = build_model(self._model_id).invoke([message])
|
|
67
|
+
return result.content if hasattr(result, "content") else str(result)
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def vision_to_text(image: bytes, *, model: VisionModel, media_type: str = "image/png") -> str:
|
|
71
|
+
"""Apply the `vision_to_text` SKILL: transcribe a scanned image to text (ingestion-side, image-only
|
|
72
|
+
filings). `model` is the skill runtime (SeamVisionModel in production; a stub in tests)."""
|
|
73
|
+
return model.image_to_text(image, media_type=media_type)
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def register_vision_to_text(registry: CapabilityRegistry) -> None:
|
|
77
|
+
"""Register `vision_to_text` as an AGENT_SKILL (FR-C.9, split from `generation` by ADR-0014; SKILL-SPLIT): a
|
|
78
|
+
single grounded vision-language act, authored as `skills/vision_to_text/SKILL.md` and applied via the seam.
|
|
79
|
+
Typed output = `VisionTranscription`."""
|
|
80
|
+
registry.register(
|
|
81
|
+
"vision_to_text",
|
|
82
|
+
contract=VisionTranscription,
|
|
83
|
+
kind="agent_skill",
|
|
84
|
+
display_name="Vision-to-text (scanned-image transcription; authored skill)",
|
|
85
|
+
)
|
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
"""Issue 0009-VLM: VLM-based OCR via a remote vision model -- the escalation for DEGRADED scans that
|
|
2
|
+
character-based OCR cannot read (severe blur / faded ink), where a strong VLM reads the text by language
|
|
3
|
+
context like a human (benchmark: Gemma-4 char_sim 0.991 on the heavy scan vs ~0.01-0.10 for OCR; see
|
|
4
|
+
docs/eval/ocr_benchmark.md).
|
|
5
|
+
|
|
6
|
+
docling's `ApiVlmOptions` points at any OpenAI-compatible endpoint; OpenRouter by default (the same provider we
|
|
7
|
+
already use for Gemma-4 -- no Modal, no local model). The model is the `VISION_OCR` role (default Gemma-4,
|
|
8
|
+
swappable via `RAG_MODEL_VISION_OCR`), so the escalation model is chosen through the model-profile seam, never
|
|
9
|
+
hardcoded (the standing model rule).
|
|
10
|
+
"""
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import os
|
|
14
|
+
import tempfile
|
|
15
|
+
from pathlib import Path
|
|
16
|
+
from typing import Any, Optional
|
|
17
|
+
|
|
18
|
+
from rag_wright.models.profiles import ModelRole, model_for
|
|
19
|
+
|
|
20
|
+
_OPENROUTER_URL = "https://openrouter.ai/api/v1/chat/completions"
|
|
21
|
+
_OCR_PROMPT = ("Transcribe ALL text from this document page exactly as it appears, preserving the reading order. "
|
|
22
|
+
"Output only the transcribed text as markdown, with no commentary.")
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def openrouter_vlm_options(model_id: Optional[str] = None, *, api_key: Optional[str] = None,
|
|
26
|
+
base_url: str = _OPENROUTER_URL, scale: float = 3.0,
|
|
27
|
+
prompt: Optional[str] = None, timeout: float = 180.0):
|
|
28
|
+
"""A docling `ApiVlmOptions` for a remote VLM OCR call. Model defaults to the VISION_OCR role (Gemma-4);
|
|
29
|
+
endpoint is OpenRouter; `scale` renders the page at higher resolution for the model."""
|
|
30
|
+
from docling.datamodel.pipeline_options import ApiVlmOptions, ResponseFormat
|
|
31
|
+
|
|
32
|
+
model_id = model_id or model_for(ModelRole.VISION_OCR)
|
|
33
|
+
key = api_key if api_key is not None else os.environ.get("OPENROUTER_API_KEY", "")
|
|
34
|
+
return ApiVlmOptions(
|
|
35
|
+
url=base_url, headers={"Authorization": f"Bearer {key}"},
|
|
36
|
+
params={"model": model_id, "max_tokens": 8192}, prompt=prompt or _OCR_PROMPT,
|
|
37
|
+
response_format=ResponseFormat.MARKDOWN, scale=scale, timeout=timeout)
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def build_vlm_ocr_converter(vlm_options: Any = None, *, model_id: Optional[str] = None, **kw):
|
|
41
|
+
"""A docling `DocumentConverter` whose PDF pipeline is the VLM pipeline over `vlm_options` (default: the
|
|
42
|
+
OpenRouter VLM). `enable_remote_services` is set so docling may call the API."""
|
|
43
|
+
from docling.datamodel.base_models import InputFormat
|
|
44
|
+
from docling.datamodel.pipeline_options import VlmPipelineOptions
|
|
45
|
+
from docling.document_converter import DocumentConverter, PdfFormatOption
|
|
46
|
+
from docling.pipeline.vlm_pipeline import VlmPipeline
|
|
47
|
+
|
|
48
|
+
opts = vlm_options or openrouter_vlm_options(model_id=model_id, **kw)
|
|
49
|
+
vopts = VlmPipelineOptions(vlm_options=opts)
|
|
50
|
+
vopts.enable_remote_services = True # required for an API-based VLM
|
|
51
|
+
return DocumentConverter(
|
|
52
|
+
format_options={InputFormat.PDF: PdfFormatOption(pipeline_cls=VlmPipeline, pipeline_options=vopts)})
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def vlm_ocr(data: bytes, name: str = "scan.pdf", *, converter: Any = None, model_id: Optional[str] = None) -> str:
|
|
56
|
+
"""OCR a document's raw bytes via the VLM -> markdown text. `converter` is injectable (hermetic tests);
|
|
57
|
+
production builds the OpenRouter converter for the VISION_OCR model."""
|
|
58
|
+
conv = converter or build_vlm_ocr_converter(model_id=model_id)
|
|
59
|
+
tmp = Path(tempfile.mkdtemp(prefix="vlmocr_")) / name
|
|
60
|
+
tmp.write_bytes(data)
|
|
61
|
+
return conv.convert(str(tmp)).document.export_to_markdown()
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
class VlmOCRParser:
|
|
65
|
+
"""A `Parser` (convert(source) -> DoclingDocument) backed by the remote VLM converter -- the 0009 escalation
|
|
66
|
+
parser the tiered path routes degraded scans to. The converter is injectable for tests."""
|
|
67
|
+
|
|
68
|
+
def __init__(self, converter: Any = None) -> None:
|
|
69
|
+
self._converter = converter
|
|
70
|
+
|
|
71
|
+
def convert(self, source):
|
|
72
|
+
conv = self._converter or build_vlm_ocr_converter()
|
|
73
|
+
return conv.convert(str(source)).document
|
|
74
|
+
|
|
75
|
+
def parse_range(self, source, page_range):
|
|
76
|
+
"""PARSE-3: VLM-parse only pages `page_range` (1-based, inclusive), so the tiered path escalates just the
|
|
77
|
+
degraded pages instead of the whole document."""
|
|
78
|
+
conv = self._converter or build_vlm_ocr_converter()
|
|
79
|
+
return conv.convert(str(source), page_range=page_range).document
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def register_vlm_ocr(registry) -> None:
|
|
83
|
+
"""0009-VLM: register `vlm_ocr` (function; document bytes -> transcribed text via a remote VLM)."""
|
|
84
|
+
registry.register("vlm_ocr", contract=str, kind="function",
|
|
85
|
+
display_name="VLM OCR (degraded-scan escalation via OpenRouter, default Gemma-4)")
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
"""Pydantic contracts: the ontology, the extraction contracts, and the shared
|
|
2
|
+
identifiers (chunk_id per FR-S.2, entity_id per FR-S.3).
|
|
3
|
+
|
|
4
|
+
These contracts double as the registered capability contracts each FR-C capability is
|
|
5
|
+
registered under. Frozen in Phase 3 before implementation so Phase 4 tests assert against them.
|
|
6
|
+
"""
|
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
"""The chunk record contract (FR-I.3, FR-S.1).
|
|
2
|
+
|
|
3
|
+
One record per chunk in the hybrid retrieval index. It holds the `chunk_id`, the summary, the
|
|
4
|
+
dense-over-summary vector, the sparse-over-full-text vector, extracted keywords and entities, and
|
|
5
|
+
source metadata. The record does not hold the raw full chunk text: FR-I.3 enumerates the summary
|
|
6
|
+
plus vectors. The text is used once at chunking to compute the `chunk_id` content hash, then persisted
|
|
7
|
+
to the chunk-text sidecar (`store/chunk_text.py`, T40) keyed by `chunk_id` and rehydrated at query time
|
|
8
|
+
by `chunk_read` (T38); it is deliberately kept out of the index to keep the index dense-over-summary.
|
|
9
|
+
|
|
10
|
+
The load-bearing part of this contract is the vector shapes. They must match what BGE-M3 produces
|
|
11
|
+
(the embedding capability, T19) and what the ArcadeDB dense `LSM_VECTOR` and sparse
|
|
12
|
+
`LSM_SPARSE_VECTOR` indexes bind (the store seam, T13), so the dense dimension and the sparse
|
|
13
|
+
key/value ranges are validated here.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import math
|
|
19
|
+
|
|
20
|
+
from pydantic import BaseModel, field_validator
|
|
21
|
+
|
|
22
|
+
from rag_wright.contracts.identifiers import ChunkId
|
|
23
|
+
|
|
24
|
+
# BAAI/bge-m3 dense embedding dimension. The dense vector is computed over the summary (FR-I.3);
|
|
25
|
+
# the ArcadeDB dense LSM_VECTOR index (T13) binds this dimension.
|
|
26
|
+
BGE_M3_DENSE_DIM = 1024
|
|
27
|
+
|
|
28
|
+
# Metadata values are kept to filterable JSON scalars so the store can index and filter on them
|
|
29
|
+
# (metadata filters, FR-Q.1).
|
|
30
|
+
MetadataValue = str | int | float | bool
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
class ChunkRecord(BaseModel):
|
|
34
|
+
"""A chunk's record in the retrieval index (FR-I.3, FR-S.1)."""
|
|
35
|
+
|
|
36
|
+
chunk_id: ChunkId
|
|
37
|
+
summary: str
|
|
38
|
+
dense_vector: list[float] # dense over the summary; length == BGE_M3_DENSE_DIM
|
|
39
|
+
sparse_vector: dict[int, float] # sparse over full text: token-id index -> non-negative weight
|
|
40
|
+
keywords: list[str] = []
|
|
41
|
+
# Unresolved surface-form entity mentions for retrieval metadata only. These are NOT resolved
|
|
42
|
+
# and must NOT be joined to the graph's canonical entity_ids; resolution to entity_id happens
|
|
43
|
+
# later (FR-C.7) and canonical entities live on graph nodes (T24). Naming this entity_mentions
|
|
44
|
+
# (not entities) keeps the pre-resolution side of the fragmentation seam unambiguous.
|
|
45
|
+
entity_mentions: list[str] = []
|
|
46
|
+
source_metadata: dict[str, MetadataValue] = {}
|
|
47
|
+
|
|
48
|
+
@field_validator("summary")
|
|
49
|
+
@classmethod
|
|
50
|
+
def _summary_non_empty(cls, v: str) -> str:
|
|
51
|
+
if not v.strip():
|
|
52
|
+
raise ValueError("summary must be non-empty")
|
|
53
|
+
return v
|
|
54
|
+
|
|
55
|
+
@field_validator("dense_vector")
|
|
56
|
+
@classmethod
|
|
57
|
+
def _dense_shape(cls, v: list[float]) -> list[float]:
|
|
58
|
+
if len(v) != BGE_M3_DENSE_DIM:
|
|
59
|
+
raise ValueError(f"dense_vector must have length {BGE_M3_DENSE_DIM}, got {len(v)}")
|
|
60
|
+
if not all(math.isfinite(x) for x in v):
|
|
61
|
+
raise ValueError("dense_vector must contain only finite values")
|
|
62
|
+
return v
|
|
63
|
+
|
|
64
|
+
@field_validator("sparse_vector")
|
|
65
|
+
@classmethod
|
|
66
|
+
def _sparse_ranges(cls, v: dict[int, float]) -> dict[int, float]:
|
|
67
|
+
for token_id, weight in v.items():
|
|
68
|
+
if token_id < 0:
|
|
69
|
+
raise ValueError(f"sparse_vector token ids must be >= 0, got {token_id}")
|
|
70
|
+
if not math.isfinite(weight) or weight < 0:
|
|
71
|
+
raise ValueError(f"sparse_vector weights must be finite and >= 0, got {weight}")
|
|
72
|
+
return v
|
|
73
|
+
|
|
74
|
+
@field_validator("keywords", "entity_mentions")
|
|
75
|
+
@classmethod
|
|
76
|
+
def _no_blank_items(cls, v: list[str]) -> list[str]:
|
|
77
|
+
if any(not item.strip() for item in v):
|
|
78
|
+
raise ValueError("keywords and entity_mentions must not contain blank strings")
|
|
79
|
+
return v
|
|
@@ -0,0 +1,303 @@
|
|
|
1
|
+
"""Compliance-module contracts (CC-1, roadmap §13.1): `Requirement` and `Claim`.
|
|
2
|
+
|
|
3
|
+
The compliance check is two-sided retrieval + entailment. The REGULATORY side ingests into `Requirement`
|
|
4
|
+
nodes (a single deontic rule); the SUBJECT side extracts `Claim` nodes (a checkable ad assertion). CC-6's
|
|
5
|
+
applicability match (claim -> applicable requirements) reuses the Leg-B retrieval router, so a `Constraint`
|
|
6
|
+
here is exactly the `(dimension, value)` pair the router matches on (`applicability_scope` <-> a claim's
|
|
7
|
+
scope). Provenance + confidence on everything (FR-S.4); no claim without a citation (FR-Q.6).
|
|
8
|
+
|
|
9
|
+
The vocab is the thin AUTHORED advertising layer (closed `DeonticType`/`ClaimType`/`Severity`), grounded on
|
|
10
|
+
the public deontic backbone (ODRL/LKIF) in the sibling `compliance_bridge.ttl`. These contracts double as the
|
|
11
|
+
registered capability contracts (`requirement_extraction`, `claim_extraction`, ...).
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
import hashlib
|
|
17
|
+
from enum import Enum
|
|
18
|
+
from pathlib import Path
|
|
19
|
+
from typing import Literal, Optional
|
|
20
|
+
|
|
21
|
+
from pydantic import BaseModel, ConfigDict, field_validator, model_validator
|
|
22
|
+
|
|
23
|
+
from rag_wright.contracts.identifiers import canonical_source_doc_id
|
|
24
|
+
from rag_wright.contracts.provenance import ConfidenceTag
|
|
25
|
+
|
|
26
|
+
# the sibling ontology (CC-1): public deontic backbone (ODRL/LKIF) + PROV + the thin authored ad vocab
|
|
27
|
+
BRIDGE_TTL_PATH = Path(__file__).resolve().parent.parent / "ontology" / "compliance_bridge.ttl"
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
# ADR-0066 P3b: the closed vocabularies below (DeonticType / ClaimType / Severity / RuleScope / Verdict) are
|
|
31
|
+
# AUTHORITATIVE in compliance_bridge.ttl (owl:oneOf). To change one, edit the ttl -- these enums are drift-locked
|
|
32
|
+
# to it by tests/ontology/test_compliance_ontology_authoritative.py.
|
|
33
|
+
class DeonticType(str, Enum):
|
|
34
|
+
"""The rule's deontic force (LKIF/ODRL closed vocab): what it obliges, forbids, or permits."""
|
|
35
|
+
|
|
36
|
+
OBLIGATION = "obligation"
|
|
37
|
+
PROHIBITION = "prohibition"
|
|
38
|
+
PERMISSION = "permission"
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
class ClaimType(str, Enum):
|
|
42
|
+
"""The kind of checkable assertion an ad makes (closed authored vocab, roadmap §13.1)."""
|
|
43
|
+
|
|
44
|
+
EFFICACY = "efficacy"
|
|
45
|
+
COMPARATIVE = "comparative"
|
|
46
|
+
PRICING = "pricing"
|
|
47
|
+
HEALTH = "health"
|
|
48
|
+
ENVIRONMENTAL = "environmental"
|
|
49
|
+
ENDORSEMENT = "endorsement"
|
|
50
|
+
PERFORMANCE = "performance"
|
|
51
|
+
GUARANTEE = "guarantee"
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
class Severity(str, Enum):
|
|
55
|
+
"""Optional severity of a requirement (drives triage, not the verdict)."""
|
|
56
|
+
|
|
57
|
+
LOW = "low"
|
|
58
|
+
MEDIUM = "med"
|
|
59
|
+
HIGH = "high"
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
class Constraint(BaseModel):
|
|
63
|
+
"""One applicability condition, a `(dimension, value)` pair. The shape is deliberately identical to the
|
|
64
|
+
retrieval router's constraint tuple (`property_boosted_retrieval` / `query_function_classifier`), so
|
|
65
|
+
CC-6's claim -> applicable-requirement match reuses Leg-B rather than a new matcher."""
|
|
66
|
+
|
|
67
|
+
model_config = ConfigDict(frozen=True)
|
|
68
|
+
|
|
69
|
+
dimension: str
|
|
70
|
+
value: str
|
|
71
|
+
|
|
72
|
+
def as_tuple(self) -> tuple[str, str]:
|
|
73
|
+
return (self.dimension, self.value)
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def _nonblank(v: str, what: str) -> str:
|
|
77
|
+
if not isinstance(v, str) or not v.strip():
|
|
78
|
+
raise ValueError(f"{what} must be a non-empty string")
|
|
79
|
+
return v.strip()
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def _content_id(source: str, section: str, text: str) -> str:
|
|
83
|
+
"""`<canonical-source>:<section>:<hash16>` -- deterministic content-hash id (RAC-1), the clause_id
|
|
84
|
+
scheme generalized (source-unit + locator + content hash), so an unchanged rule/claim keeps its id."""
|
|
85
|
+
digest = hashlib.sha256(text.encode("utf-8")).hexdigest()[:16]
|
|
86
|
+
return f"{canonical_source_doc_id(source)}:{section}:{digest}"
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
class Requirement(BaseModel):
|
|
90
|
+
"""A single regulatory rule extracted from the requirements corpus (roadmap §13.1).
|
|
91
|
+
|
|
92
|
+
`requirement_id = <source_reg>:<section>:<hash>`. Carries its `citation` (always cited, FR-Q.6) and a
|
|
93
|
+
`confidence` tag (graph-derived fact, FR-S.4). `applicability_scope` is matched against a claim's scope.
|
|
94
|
+
"""
|
|
95
|
+
|
|
96
|
+
requirement_id: str
|
|
97
|
+
source: str
|
|
98
|
+
citation: str # section / paragraph -- the human-readable provenance, always present
|
|
99
|
+
deontic_type: DeonticType
|
|
100
|
+
actor: str # who it binds: advertiser, endorser, ... (open, matched against a claim's actor)
|
|
101
|
+
applicability_scope: list[Constraint] = []
|
|
102
|
+
requirement_text: str
|
|
103
|
+
evidence_standard: str | None = None
|
|
104
|
+
trigger_condition: str | None = None
|
|
105
|
+
severity: Severity | None = None
|
|
106
|
+
# issue 0043: page provenance for the POLICY side, in the same shape span provenance uses (issue 0032), so one
|
|
107
|
+
# product code path serves both sides of a finding. `citation` stays the always-present human-readable
|
|
108
|
+
# provenance; these are additive. A requirement is bound to a policy SECTION, so `pages` are that section's
|
|
109
|
+
# source page(s) (from the parse's per-item provenance -- present on scans, where a text search would fail
|
|
110
|
+
# silently). `bbox` is best-effort (`(l, t, r, b)`) and usually None for a multi-item section; never fabricated.
|
|
111
|
+
pages: list[int] = []
|
|
112
|
+
bbox: tuple[float, float, float, float] | None = None
|
|
113
|
+
confidence: ConfidenceTag = ConfidenceTag.EXTRACTED
|
|
114
|
+
defenses: list[str] = [] # DEON-9 (query-time only, never persisted): same-source PERMISSIONS linked as
|
|
115
|
+
# carve-outs/exceptions that may EXCUSE this O/F rule -- passed to the judge as structured context so a
|
|
116
|
+
# legitimate exception is not a false violation (ADR-0044 pattern, requirement side).
|
|
117
|
+
|
|
118
|
+
@field_validator("requirement_id", "source", "citation", "requirement_text")
|
|
119
|
+
@classmethod
|
|
120
|
+
def _required_nonblank(cls, v: str, info) -> str:
|
|
121
|
+
return _nonblank(v, info.field_name)
|
|
122
|
+
|
|
123
|
+
@staticmethod
|
|
124
|
+
def make_id(source: str, section: str, requirement_text: str) -> str:
|
|
125
|
+
return _content_id(source, section, requirement_text)
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
class CheckableFact(BaseModel):
|
|
129
|
+
"""COMP-VERDICT-GENERIC: the DOMAIN-AGNOSTIC subject-fact the compliance verdict core consumes -- a checkable
|
|
130
|
+
assertion (any domain) with its provenance. The verdict machinery (semantic retrieval + LLM judge + finding /
|
|
131
|
+
report) needs only THIS (an id + text + provenance); advertising `Claim` is a SPECIALIZATION that adds typed
|
|
132
|
+
claim fields for structured routing. A NEW compliance domain either uses a bare `CheckableFact` (generic
|
|
133
|
+
verdict) or subclasses this with its own enrichment -- WITHOUT touching the base or other domains.
|
|
134
|
+
|
|
135
|
+
`fact_id = <source_doc>:<index>:<hash>`; `(source_doc, doc_start, doc_end, assertion_text)` is the span
|
|
136
|
+
provenance (cited, FR-Q.6); `confidence` is the graph-derived tag (FR-S.4). NB: never persisted -- a
|
|
137
|
+
query-time object only (the KG stores `Requirement`), so this shape is a pure query-side/contract concern."""
|
|
138
|
+
|
|
139
|
+
fact_id: str
|
|
140
|
+
source_doc: str
|
|
141
|
+
assertion_text: str # the checkable statement text (domain-neutral: an "assertion" is any checkable claim/fact)
|
|
142
|
+
# issue 0044: JUDGE-ONLY structured signals (e.g. DEON-8's ad-level disclosure/evidence union) rendered as
|
|
143
|
+
# document context FOR THE JUDGE but NOT part of the citation -- so engine scaffolding never surfaces as a
|
|
144
|
+
# quote from the user's document. The judge appends this; `assemble_finding`'s citation uses `assertion_text`
|
|
145
|
+
# only. Empty for a plain fact (domain-neutral: the generic path carries none).
|
|
146
|
+
document_signals: str = ""
|
|
147
|
+
# issue 0044: whether `assertion_text` is one VERBATIM span from the document, or ASSEMBLED evidence (the
|
|
148
|
+
# obligation path joins the top-N relevant spans with "\n\n"). Copied onto the finding so a consumer can tell
|
|
149
|
+
# a verbatim quote from a synthesised excerpt WITHOUT parsing prose, and render/attribute it accordingly.
|
|
150
|
+
citation_kind: Literal["verbatim", "assembled"] = "verbatim"
|
|
151
|
+
section: str | None = None # UNIFY-A: the section/heading locator this fact came from (e.g. "4.2"); the
|
|
152
|
+
# finding cites "doc § {section}: {assertion}" when set. Additive/optional: None -> the old "doc: assertion".
|
|
153
|
+
element_kind: str | None = None # SEG-1: docling structural label of the source element (paragraph /
|
|
154
|
+
# list_item / section_header / ...); selects the within-section marker in `locator()`.
|
|
155
|
+
element_ordinal: int | None = None # SEG-1: the element's within-section ordinal (the ¶ / bullet number).
|
|
156
|
+
scope: list["Constraint"] = [] # DEON-5: the assertion's inferred applicability constraints (e.g.
|
|
157
|
+
# Constraint("actor", "endorser")); dimension-agnostic, matched against a requirement's applicability_scope by
|
|
158
|
+
# the generic `constraint_applies` router (DEON-6), and aggregated to the document's SubjectScope (DEON-7).
|
|
159
|
+
doc_start: int | None = None # span provenance: char offsets in source_doc (optional)
|
|
160
|
+
doc_end: int | None = None
|
|
161
|
+
confidence: ConfidenceTag = ConfidenceTag.EXTRACTED
|
|
162
|
+
|
|
163
|
+
def locator(self) -> str:
|
|
164
|
+
"""SEG-1: the human structural locator -- `§ {section}` plus a within-section element marker when present
|
|
165
|
+
(`¶N` for a paragraph, `· bullet N` for a list item). Empty string when the fact has no section (e.g. a
|
|
166
|
+
structureless paste), so the finding cites just `doc: {assertion}`. The citation is built from this."""
|
|
167
|
+
if not (self.section and self.section.strip()):
|
|
168
|
+
return ""
|
|
169
|
+
loc = f"§ {self.section}"
|
|
170
|
+
if self.element_ordinal is not None:
|
|
171
|
+
if self.element_kind == "list_item":
|
|
172
|
+
loc += f" · bullet {self.element_ordinal}"
|
|
173
|
+
else: # paragraph / text / default
|
|
174
|
+
loc += f" ¶{self.element_ordinal}"
|
|
175
|
+
return loc
|
|
176
|
+
|
|
177
|
+
@field_validator("fact_id", "source_doc", "assertion_text")
|
|
178
|
+
@classmethod
|
|
179
|
+
def _required_nonblank(cls, v: str, info) -> str:
|
|
180
|
+
return _nonblank(v, info.field_name)
|
|
181
|
+
|
|
182
|
+
@model_validator(mode="after")
|
|
183
|
+
def _ordered_offsets(self) -> CheckableFact:
|
|
184
|
+
if self.doc_start is not None and self.doc_end is not None and self.doc_start >= self.doc_end:
|
|
185
|
+
raise ValueError(f"doc_start ({self.doc_start}) must be < doc_end ({self.doc_end})")
|
|
186
|
+
return self
|
|
187
|
+
|
|
188
|
+
@staticmethod
|
|
189
|
+
def make_id(source_doc: str, index: int, text: str) -> str:
|
|
190
|
+
return _content_id(source_doc, str(index), text)
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
class Claim(CheckableFact):
|
|
194
|
+
"""An ADVERTISING checkable element (roadmap §13.1) -- a `CheckableFact` SPECIALIZED with the typed claim
|
|
195
|
+
fields the advertising compliance path uses for structured routing (claim_type) + disclosure/substantiation
|
|
196
|
+
judging. Inherits id/text/provenance + validators + `make_id` from `CheckableFact`."""
|
|
197
|
+
|
|
198
|
+
claim_type: ClaimType
|
|
199
|
+
actor: str | None = None
|
|
200
|
+
subject_product: str | None = None
|
|
201
|
+
quantitative_value: str | None = None
|
|
202
|
+
disclosures_present: list[str] = []
|
|
203
|
+
evidence_referenced: bool = False
|
|
204
|
+
medium: str | None = None
|
|
205
|
+
|
|
206
|
+
|
|
207
|
+
class RuleScope(str, Enum):
|
|
208
|
+
"""How a requirement is narrowed at check time (CC-8a, the ontology routing tag; `compliance_bridge.ttl`
|
|
209
|
+
cmp:RuleScope). CONTENT rules (substantiation / claim-specific) narrow by semantic similarity to the claim;
|
|
210
|
+
CONTEXT rules (disclosure / material connection -- apply to ANY claim in an endorsement regardless of its
|
|
211
|
+
content) are ALWAYS included, never left to similarity. See [[ontology-lever-vs-extraction-lever]]."""
|
|
212
|
+
|
|
213
|
+
CONTENT = "content"
|
|
214
|
+
CONTEXT = "context"
|
|
215
|
+
|
|
216
|
+
|
|
217
|
+
class Verdict(str, Enum):
|
|
218
|
+
"""The compliance judgment output (CC-4, §13.2). Closed vocab; matches `compliance_bridge.ttl` cmp:Verdict.
|
|
219
|
+
`needs_review` is the conservative default under uncertainty (never a silent compliant/violation)."""
|
|
220
|
+
|
|
221
|
+
COMPLIANT = "compliant"
|
|
222
|
+
VIOLATION = "violation"
|
|
223
|
+
NEEDS_REVIEW = "needs_review"
|
|
224
|
+
|
|
225
|
+
|
|
226
|
+
class ComplianceFinding(BaseModel):
|
|
227
|
+
"""One `(claim, requirement)` judgment (CC-4, §13.2): a verdict + rationale + BOTH-SIDED citation +
|
|
228
|
+
confidence. The citations are the trust product -- an auditor sees the exact ad span AND the exact reg
|
|
229
|
+
clause. Every `violation`/`needs_review` is human-gated (`needs_human_review`); false-negatives are
|
|
230
|
+
liability, so uncertainty never silently clears."""
|
|
231
|
+
|
|
232
|
+
claim_id: str
|
|
233
|
+
requirement_id: str
|
|
234
|
+
verdict: Verdict
|
|
235
|
+
rationale: str = ""
|
|
236
|
+
citation_claim: str # the subject span text (provenance, cited -- FR-Q.6)
|
|
237
|
+
citation_requirement: str # the reg clause / section (provenance, cited)
|
|
238
|
+
# issue 0044: is `citation_claim` one VERBATIM span from the document, or ASSEMBLED evidence (the obligation
|
|
239
|
+
# path cites the top-N relevant spans joined by "\n\n")? A consumer renders assembled evidence differently
|
|
240
|
+
# instead of quoting it as the user's exact words. Default "verbatim" -> unchanged for every existing path.
|
|
241
|
+
citation_claim_kind: Literal["verbatim", "assembled"] = "verbatim"
|
|
242
|
+
confidence: float = 0.0
|
|
243
|
+
|
|
244
|
+
@field_validator("confidence")
|
|
245
|
+
@classmethod
|
|
246
|
+
def _confidence_in_unit(cls, v: float) -> float:
|
|
247
|
+
if not 0.0 <= v <= 1.0:
|
|
248
|
+
raise ValueError(f"confidence must be in [0, 1], got {v}")
|
|
249
|
+
return v
|
|
250
|
+
|
|
251
|
+
@property
|
|
252
|
+
def needs_human_review(self) -> bool:
|
|
253
|
+
"""Every violation requires human confirmation before it leaves the tool; needs_review always does.
|
|
254
|
+
A `compliant` finding does not gate (§13.2)."""
|
|
255
|
+
return self.verdict in (Verdict.VIOLATION, Verdict.NEEDS_REVIEW)
|
|
256
|
+
|
|
257
|
+
|
|
258
|
+
_AD_VIOLATION_THRESHOLD = 2 # a lone violation finding among many rules escalates, not hard-flags (RG-5 aggregation)
|
|
259
|
+
|
|
260
|
+
|
|
261
|
+
class RequirementLocation(BaseModel):
|
|
262
|
+
"""Where a curated requirement sits in its policy document, for a citation preview (EP-REF-1b): the
|
|
263
|
+
requirement id, its citation label, the source page number(s), an optional `[l, t, r, b]` rectangle, and
|
|
264
|
+
the rule text. `pages`/`bbox` are best-effort -- a requirement curated before provenance landed (ADR-0107)
|
|
265
|
+
reads back with `pages=[]` / `bbox=None`, which a preview renders honestly rather than erroring."""
|
|
266
|
+
|
|
267
|
+
requirement_id: str
|
|
268
|
+
citation: str
|
|
269
|
+
pages: list[int] = []
|
|
270
|
+
bbox: Optional[tuple[float, float, float, float]] = None
|
|
271
|
+
text: str
|
|
272
|
+
|
|
273
|
+
|
|
274
|
+
class ComplianceReport(BaseModel):
|
|
275
|
+
"""The `compliance_check` output (CC-6, §13.3): a subject document's cited findings + a per-requirement gap
|
|
276
|
+
matrix + a verdict summary. Both-sided cited; every violation/needs_review is human-gated per finding."""
|
|
277
|
+
|
|
278
|
+
source_doc: str
|
|
279
|
+
findings: list[ComplianceFinding] = []
|
|
280
|
+
summary: dict[str, int] = {} # verdict -> count (compliant/violation/needs_review)
|
|
281
|
+
gap_matrix: list[dict] = [] # per-requirement rollup: {requirement_id, citation, verdict, claims_checked}
|
|
282
|
+
# SEG-6 (0009-WIRE2 / ENG-1 principle): pages the tiered OCR could not read even after VLM escalation, so a
|
|
283
|
+
# verdict on a scanned subject is never SILENTLY based on half-read text. Empty = fully readable. The product
|
|
284
|
+
# surfaces this as "pages X-Y unreadable; results for those pages are incomplete".
|
|
285
|
+
ocr_unreadable_pages: list[int] = []
|
|
286
|
+
# ADR-0068 (engine issue 0013): (assertion, rule) pairs the symbolic ACTOR gate SKIPPED before any judge call
|
|
287
|
+
# -- so the gate is never a SILENT recall loss. Each: {requirement_id, citation, actor, subject_actors, scope
|
|
288
|
+
# ("assertion"|"document"), claim_id}. Empty is the norm (the recall-first gate skips only ontology-disjoint
|
|
289
|
+
# roles); a non-empty list lets a consumer state honest coverage ("checked N rules; K pairs gated by role").
|
|
290
|
+
gated_pairs: list[dict] = []
|
|
291
|
+
|
|
292
|
+
@property
|
|
293
|
+
def verdict(self) -> Verdict:
|
|
294
|
+
"""The AD-LEVEL verdict rolled up from the findings (RG-5): VIOLATION only when violation findings are a
|
|
295
|
+
real signal (>= threshold, so one spurious finding among many rules does not hard-flag); else
|
|
296
|
+
NEEDS_REVIEW if anything fired (a lone violation OR any needs_review -> escalate for a human); else
|
|
297
|
+
COMPLIANT. It never CLEARS an ad that had a violation finding -- a real violation is never a silent pass."""
|
|
298
|
+
v = self.summary.get("violation", 0)
|
|
299
|
+
if v >= _AD_VIOLATION_THRESHOLD:
|
|
300
|
+
return Verdict.VIOLATION
|
|
301
|
+
if v or self.summary.get("needs_review", 0):
|
|
302
|
+
return Verdict.NEEDS_REVIEW
|
|
303
|
+
return Verdict.COMPLIANT
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
"""Contract-level metadata record (CU-A1, ADR-0029).
|
|
2
|
+
|
|
3
|
+
The CUAD pipeline is document-scoped: a contract is LOOKED UP by id (not searched), then all retrieval happens
|
|
4
|
+
within it. `ContractRecord` is that lookup/filter unit -- the contract node the `Span` records point back to
|
|
5
|
+
via `contract_id`. Metadata fields are best-effort (many come from a value-type extraction pass, CU-C2) and
|
|
6
|
+
default empty; only `contract_id` is required.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from pydantic import BaseModel
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class ContractRecord(BaseModel):
|
|
15
|
+
"""One contract's metadata for lookup + filtering (the parent of its clauses/spans)."""
|
|
16
|
+
|
|
17
|
+
model_config = {"frozen": True}
|
|
18
|
+
|
|
19
|
+
contract_id: str # the canonical document id (== spans' contract_id / the parse source_doc_id)
|
|
20
|
+
name: str = "" # the contract's name/title (CUAD "Document Name")
|
|
21
|
+
agreement_type: str = "" # e.g. "Distributor Agreement"
|
|
22
|
+
parties: list[str] = [] # signing parties (CUAD "Parties")
|
|
23
|
+
agreement_date: str = "" # as-written date string (not normalized here)
|
|
24
|
+
effective_date: str = ""
|
|
25
|
+
source_doc_id: str = "" # the parse source id (usually == contract_id)
|
|
26
|
+
content_hash: str = "" # source content hash (provenance / idempotence)
|
|
27
|
+
page_count: int | None = None # for PDF-overlay citation, when known
|