rag-wright 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- rag_wright/__init__.py +13 -0
- rag_wright/api/__init__.py +33 -0
- rag_wright/api/config.py +59 -0
- rag_wright/api/discover.py +70 -0
- rag_wright/api/documents.py +39 -0
- rag_wright/api/ids.py +31 -0
- rag_wright/api/invoke.py +99 -0
- rag_wright/api/kg.py +61 -0
- rag_wright/api/mcp.py +94 -0
- rag_wright/api/usage.py +30 -0
- rag_wright/api/workspace.py +85 -0
- rag_wright/capabilities/__init__.py +8 -0
- rag_wright/capabilities/answer_generator.py +427 -0
- rag_wright/capabilities/ard.py +286 -0
- rag_wright/capabilities/assertion_extraction.py +79 -0
- rag_wright/capabilities/chunk_read.py +58 -0
- rag_wright/capabilities/chunk_write.py +163 -0
- rag_wright/capabilities/claim_extraction.py +153 -0
- rag_wright/capabilities/clause_exception_linking.py +117 -0
- rag_wright/capabilities/compliance_judgment.py +322 -0
- rag_wright/capabilities/compliance_store.py +87 -0
- rag_wright/capabilities/contract_kg_serve.py +156 -0
- rag_wright/capabilities/contract_kg_store.py +251 -0
- rag_wright/capabilities/dg_extraction.py +585 -0
- rag_wright/capabilities/disambiguation.py +163 -0
- rag_wright/capabilities/document_parse.py +87 -0
- rag_wright/capabilities/document_scope.py +49 -0
- rag_wright/capabilities/embedding.py +164 -0
- rag_wright/capabilities/embedding_profiles.py +43 -0
- rag_wright/capabilities/entity_resolution.py +154 -0
- rag_wright/capabilities/fusion.py +64 -0
- rag_wright/capabilities/graph_extraction.py +243 -0
- rag_wright/capabilities/graph_query.py +73 -0
- rag_wright/capabilities/graph_storage.py +111 -0
- rag_wright/capabilities/highlight_serve.py +142 -0
- rag_wright/capabilities/hybrid_search.py +65 -0
- rag_wright/capabilities/invoke.py +31 -0
- rag_wright/capabilities/jev_decision.py +38 -0
- rag_wright/capabilities/manifests.py +872 -0
- rag_wright/capabilities/okf_navigate.py +456 -0
- rag_wright/capabilities/parsing.py +286 -0
- rag_wright/capabilities/property_boosted_retrieval.py +125 -0
- rag_wright/capabilities/query_function_classifier.py +94 -0
- rag_wright/capabilities/query_understanding.py +109 -0
- rag_wright/capabilities/registry.py +262 -0
- rag_wright/capabilities/remote_encoders.py +94 -0
- rag_wright/capabilities/requirement_extraction.py +247 -0
- rag_wright/capabilities/reranking.py +123 -0
- rag_wright/capabilities/retrieval_core.py +126 -0
- rag_wright/capabilities/rlm_chunking.py +808 -0
- rag_wright/capabilities/rlm_synthesis.py +316 -0
- rag_wright/capabilities/scan_quality.py +136 -0
- rag_wright/capabilities/span_relevance_judgment.py +191 -0
- rag_wright/capabilities/vision_to_text.py +85 -0
- rag_wright/capabilities/vlm_ocr.py +85 -0
- rag_wright/contracts/__init__.py +6 -0
- rag_wright/contracts/chunk.py +79 -0
- rag_wright/contracts/compliance.py +303 -0
- rag_wright/contracts/contract_meta.py +27 -0
- rag_wright/contracts/extraction.py +130 -0
- rag_wright/contracts/function.py +167 -0
- rag_wright/contracts/function_routing.py +91 -0
- rag_wright/contracts/highlight.py +74 -0
- rag_wright/contracts/identifiers.py +153 -0
- rag_wright/contracts/jurisdiction.py +96 -0
- rag_wright/contracts/ontology.py +142 -0
- rag_wright/contracts/property.py +201 -0
- rag_wright/contracts/provenance.py +78 -0
- rag_wright/contracts/query_intent.py +53 -0
- rag_wright/contracts/span.py +76 -0
- rag_wright/contracts/value_match.py +84 -0
- rag_wright/corpus/__init__.py +0 -0
- rag_wright/corpus/canonicalize.py +116 -0
- rag_wright/corpus/cuad.py +153 -0
- rag_wright/corpus/cuad_ingestion.py +72 -0
- rag_wright/corpus/document_parser.py +299 -0
- rag_wright/corpus/edgar.py +231 -0
- rag_wright/corpus/gcs_ingestion.py +120 -0
- rag_wright/corpus/http.py +110 -0
- rag_wright/corpus/selection.py +152 -0
- rag_wright/mcp/__init__.py +11 -0
- rag_wright/mcp/compliance_server.py +299 -0
- rag_wright/mcp/intra_document_qa_server.py +170 -0
- rag_wright/mcp/relational_qa_server.py +171 -0
- rag_wright/mcp/session_store.py +64 -0
- rag_wright/mcp/typed_property_retrieval_server.py +191 -0
- rag_wright/models/__init__.py +8 -0
- rag_wright/models/profiles.py +331 -0
- rag_wright/models/seam.py +497 -0
- rag_wright/models/tag_structured.py +285 -0
- rag_wright/models/tracing.py +179 -0
- rag_wright/models/usage.py +102 -0
- rag_wright/okf/__init__.py +11 -0
- rag_wright/okf/compile.py +292 -0
- rag_wright/okf/document.py +47 -0
- rag_wright/okf/enrich.py +176 -0
- rag_wright/okf/links.py +190 -0
- rag_wright/okf/lint.py +105 -0
- rag_wright/ontology/__init__.py +6 -0
- rag_wright/ontology/_generated_template_meta.py +60 -0
- rag_wright/ontology/_generated_vocab.py +52 -0
- rag_wright/ontology/clause_template.py +964 -0
- rag_wright/ontology/codegen.py +84 -0
- rag_wright/ontology/compliance_bridge.ttl +186 -0
- rag_wright/ontology/contract_bridge.ttl +2685 -0
- rag_wright/ontology/contract_taxonomy.py +24 -0
- rag_wright/ontology/derive.py +58 -0
- rag_wright/ontology/loader.py +435 -0
- rag_wright/ontology/packs/ftc_16cfr255.ttl +29 -0
- rag_wright/ontology/registry.py +87 -0
- rag_wright/ontology/template_introspect.py +100 -0
- rag_wright/py.typed +0 -0
- rag_wright/reference/__init__.py +2 -0
- rag_wright/reference/compliance.py +41 -0
- rag_wright/reference/contract_seam.py +123 -0
- rag_wright/skills/__init__.py +7 -0
- rag_wright/skills/claim_extraction/SKILL.md +47 -0
- rag_wright/skills/claim_extraction/__init__.py +1 -0
- rag_wright/skills/claim_extraction/template.py +50 -0
- rag_wright/skills/compliance_judgment/SKILL.md +59 -0
- rag_wright/skills/corpus_ingest/SKILL.md +106 -0
- rag_wright/skills/extraction_semantic_judge/SKILL.md +51 -0
- rag_wright/skills/extraction_semantic_judge/__init__.py +1 -0
- rag_wright/skills/generation/SKILL.md +64 -0
- rag_wright/skills/generation/__init__.py +1 -0
- rag_wright/skills/generic_compliance_judgment/SKILL.md +58 -0
- rag_wright/skills/okf_navigate/SKILL.md +137 -0
- rag_wright/skills/requirement_extraction/SKILL.md +47 -0
- rag_wright/skills/requirement_extraction/__init__.py +1 -0
- rag_wright/skills/requirement_extraction/template.py +50 -0
- rag_wright/skills/rlm/SKILL.md +186 -0
- rag_wright/skills/rlm/__init__.py +31 -0
- rag_wright/skills/rlm/agent.py +292 -0
- rag_wright/skills/span_relevance_judgment/SKILL.md +67 -0
- rag_wright/skills/vision_to_text/SKILL.md +36 -0
- rag_wright/skills/vision_to_text/__init__.py +1 -0
- rag_wright/spans/__init__.py +1 -0
- rag_wright/spans/boundary.py +78 -0
- rag_wright/spans/clause_function_classifier.py +490 -0
- rag_wright/spans/clause_kg_extractor.py +337 -0
- rag_wright/spans/cuad_labels.py +81 -0
- rag_wright/spans/dim_classifier.py +158 -0
- rag_wright/spans/dim_fleet.json +411 -0
- rag_wright/spans/function_classifier.py +77 -0
- rag_wright/spans/function_families.py +62 -0
- rag_wright/spans/hybrid_classifier.py +103 -0
- rag_wright/spans/legalbert_classifier.py +83 -0
- rag_wright/spans/model_capabilities.py +107 -0
- rag_wright/spans/new_function_labels.py +111 -0
- rag_wright/spans/page_map.py +68 -0
- rag_wright/spans/property_extractor.py +365 -0
- rag_wright/spans/property_grounding.py +182 -0
- rag_wright/spans/reclassify.py +77 -0
- rag_wright/spans/scarce_function_labels.py +105 -0
- rag_wright/spans/segment.py +341 -0
- rag_wright/spans/semantic_judge.py +197 -0
- rag_wright/spans/symbolic_validation.py +131 -0
- rag_wright/spans/tag_clause_extractor.py +182 -0
- rag_wright/store/__init__.py +6 -0
- rag_wright/store/arcadedb.py +1135 -0
- rag_wright/store/chunk_text.py +66 -0
- rag_wright/store/seam.py +213 -0
- rag_wright/subgraphs/__init__.py +0 -0
- rag_wright/subgraphs/async_ingestion.py +204 -0
- rag_wright/subgraphs/compliance_check.py +1042 -0
- rag_wright/subgraphs/compliance_ingestion.py +306 -0
- rag_wright/subgraphs/contract_ingestion_pipeline.py +999 -0
- rag_wright/subgraphs/graph_extraction.py +102 -0
- rag_wright/subgraphs/intra_document_qa.py +328 -0
- rag_wright/subgraphs/observability.py +140 -0
- rag_wright/subgraphs/query_constraint_extraction.py +73 -0
- rag_wright/subgraphs/relational_qa.py +165 -0
- rag_wright/subgraphs/requirement_extraction.py +137 -0
- rag_wright/subgraphs/scaffold.py +65 -0
- rag_wright/subgraphs/semantic_chunking.py +183 -0
- rag_wright/subgraphs/typed_clause_extraction.py +172 -0
- rag_wright/subgraphs/typed_property_retrieval.py +278 -0
- rag_wright/util/__init__.py +1 -0
- rag_wright/util/concurrent.py +153 -0
- rag_wright/util/spacy_model.py +45 -0
- rag_wright-0.1.0.dist-info/METADATA +168 -0
- rag_wright-0.1.0.dist-info/RECORD +184 -0
- rag_wright-0.1.0.dist-info/WHEEL +4 -0
- rag_wright-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
"""KG-5a: subsumption-aware, canonicalized matching of a query property constraint to a clause's value.
|
|
2
|
+
|
|
3
|
+
The typed-KG constraint match (Leg B, 4c) was brittle -- exact string set-intersection -- so it lost signal
|
|
4
|
+
to (1) surface variants of the same jurisdiction and (2) values at a finer granularity than the query. This
|
|
5
|
+
module is the fix: `value_satisfies(dimension, query_value, clause_value)` decides whether a clause's stored
|
|
6
|
+
value satisfies a query constraint, using
|
|
7
|
+
|
|
8
|
+
- JURISDICTION: canonicalization (England/England and Wales/English law -> england), else normalized string;
|
|
9
|
+
- SUBSUMPTION rollup: a more specific closed value satisfies a query for its broader value
|
|
10
|
+
(licensor_affiliates -> affiliates; geographic_and_activity -> geographic AND activity;
|
|
11
|
+
price_and_terms -> price AND terms);
|
|
12
|
+
- everything else: exact match.
|
|
13
|
+
|
|
14
|
+
Additive: no re-extraction, no value-node mutation -- just canonicalization + a small ontology hierarchy
|
|
15
|
+
(also recorded as skos:broader in contract_bridge.ttl). FUTURE (deferred): covered_subject
|
|
16
|
+
{trademark, copyright} -> ip_infringement (legally sound, slightly looser -- add if a query needs it).
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
from __future__ import annotations
|
|
20
|
+
|
|
21
|
+
from pydantic import BaseModel
|
|
22
|
+
|
|
23
|
+
from rag_wright.contracts.jurisdiction import canonicalize_jurisdiction
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
class NormalizedValue(BaseModel):
|
|
27
|
+
"""CAP-REG-2: the canonical form of a typed value (KG-5a) — a surface value normalized (jurisdiction
|
|
28
|
+
canonicalization / subsumption) for matching. `changed` is False when the surface was already canonical.
|
|
29
|
+
The output contract of the `typed_value_normalization` capability."""
|
|
30
|
+
|
|
31
|
+
dimension: str
|
|
32
|
+
surface: str
|
|
33
|
+
canonical: str
|
|
34
|
+
changed: bool
|
|
35
|
+
|
|
36
|
+
# a more-specific closed value -> the broader query value(s) it also satisfies (the three clear cases)
|
|
37
|
+
VALUE_ROLLUP: dict[str, dict[str, set[str]]] = {
|
|
38
|
+
"covered_parties": {
|
|
39
|
+
"licensor_affiliates": {"affiliates"},
|
|
40
|
+
"licensee_affiliates": {"affiliates"},
|
|
41
|
+
},
|
|
42
|
+
"restriction_scope": {
|
|
43
|
+
"geographic_and_activity": {"geographic", "activity"},
|
|
44
|
+
},
|
|
45
|
+
"mfn_scope": {
|
|
46
|
+
"price_and_terms": {"price", "terms"},
|
|
47
|
+
},
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def satisfied_values(dimension: str, clause_value: str) -> set[str]:
|
|
52
|
+
"""The query-constraint values a clause value satisfies: itself plus any broader value it rolls up to."""
|
|
53
|
+
return {clause_value} | VALUE_ROLLUP.get(dimension, {}).get(clause_value, set())
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def _norm_jurisdiction(v: str) -> str:
|
|
57
|
+
return canonicalize_jurisdiction(v) or (v or "").strip().lower()
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def value_satisfies(dimension: str, query_value: str, clause_value: str) -> bool:
|
|
61
|
+
"""Whether a clause's stored `clause_value` satisfies the query's `query_value` on this dimension --
|
|
62
|
+
jurisdiction-canonicalized, subsumption-aware, exact otherwise."""
|
|
63
|
+
if dimension == "jurisdiction":
|
|
64
|
+
return _norm_jurisdiction(query_value) == _norm_jurisdiction(clause_value)
|
|
65
|
+
return query_value in satisfied_values(dimension, clause_value)
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def constraint_match_count(query_constraints: set, clause_props: set) -> int:
|
|
69
|
+
"""How many of the query's (dimension, value) constraints the clause's (dimension, value) props satisfy,
|
|
70
|
+
under canonicalization + subsumption. Replaces the old exact set-intersection count."""
|
|
71
|
+
return sum(
|
|
72
|
+
1 for qd, qv in query_constraints
|
|
73
|
+
if any(cd == qd and value_satisfies(qd, qv, cv) for cd, cv in clause_props)
|
|
74
|
+
)
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def register_typed_value_normalization(registry) -> None:
|
|
78
|
+
"""CAP-REG-2: register `typed_value_normalization` (function; KG-5a canonicalization + subsumption)."""
|
|
79
|
+
registry.register(
|
|
80
|
+
"typed_value_normalization",
|
|
81
|
+
contract=NormalizedValue,
|
|
82
|
+
kind="function",
|
|
83
|
+
display_name="Typed value normalization",
|
|
84
|
+
)
|
|
File without changes
|
|
@@ -0,0 +1,116 @@
|
|
|
1
|
+
"""Entity mention canonicalization -- the normalize / reject / cluster method (T23b).
|
|
2
|
+
|
|
3
|
+
This is the stage the fragmentation exposed as missing, between recognition (T23) and closed-world
|
|
4
|
+
linking (T24): normalize the surface-form variants of one real-world entity to a shared key, reject
|
|
5
|
+
non-entities (template placeholders, role artifacts, bare over-broad tokens), and cluster the
|
|
6
|
+
variants so one entity is one graph node. It is what stops the graph fragmenting (the
|
|
7
|
+
`Bank of America` / `Bank of America, N.A.` / `"<<enter Company Name>>"` problem) and keeps human
|
|
8
|
+
name->CIK verification scaling with entity count, not mention count.
|
|
9
|
+
|
|
10
|
+
Used first here, under human verification, to canonicalize the T10 relational eval set. These
|
|
11
|
+
normalize/reject/cluster rules harden into the T23b capability -- keep them.
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
import re
|
|
17
|
+
|
|
18
|
+
from pydantic import BaseModel
|
|
19
|
+
|
|
20
|
+
# Legal-form suffixes stripped for the clustering key, so "Bank of America" and "Bank of America,
|
|
21
|
+
# N.A." collapse to one key. Applied repeatedly (a name may carry several, e.g. "Co., Ltd.").
|
|
22
|
+
_LEGAL_SUFFIX = re.compile(
|
|
23
|
+
r"[,\.\s]+\b("
|
|
24
|
+
r"incorporated|inc|corporation|corp|company|co|llc|l\.?l\.?c|llp|l\.?p|lp|ltd|limited|plc|"
|
|
25
|
+
r"gmbh|ag|n\.?\s*v\.?|nv|s\.?\s*a\.?|sa|s\.?p\.?a|spa|n\.?\s*a\.?|na|pllc"
|
|
26
|
+
r")\b\.?$",
|
|
27
|
+
re.IGNORECASE,
|
|
28
|
+
)
|
|
29
|
+
_NON_ALNUM = re.compile(r"[^a-z0-9]+")
|
|
30
|
+
_APOSTROPHE = re.compile(r"['’‘`]") # so "Stremick's" == "Stremicks"
|
|
31
|
+
_PLACEHOLDER = re.compile(r"<<|>>|_{2,}|\bxxx+\b|\benter\b|company name|\[\s*\]", re.IGNORECASE)
|
|
32
|
+
# Alias clause markers: everything from the marker on is an alias, not part of the legal name, so
|
|
33
|
+
# "Acme Inc. d/b/a Brand" -> "Acme Inc." and a mention that is only an alias clause ("formerly known
|
|
34
|
+
# as Tradeum, Inc. which d/b/a VerticalNet Solutions") strips to empty and is rejected.
|
|
35
|
+
_ALIAS_MARKER = re.compile(
|
|
36
|
+
r"\b(formerly known as|doing business as|also known as|"
|
|
37
|
+
r"f\s*/?\s*k\s*/?\s*a|d\s*/?\s*b\s*/?\s*a|a\s*/?\s*k\s*/?\s*a)\b",
|
|
38
|
+
re.IGNORECASE,
|
|
39
|
+
)
|
|
40
|
+
# Contract party-definition clause fragments ("...and together with Buyer the Buyer Entities"), not
|
|
41
|
+
# entity names. A mention carrying one of these is a role phrase, not a company.
|
|
42
|
+
_ROLE_PHRASE = re.compile(
|
|
43
|
+
r"\b(together with|buyer entit|seller entit|the buyer|the seller|buyer the|seller the|"
|
|
44
|
+
r"the company and|and together|the parties|collectively)\b",
|
|
45
|
+
re.IGNORECASE,
|
|
46
|
+
)
|
|
47
|
+
|
|
48
|
+
# Bare over-broad / role words that are not entities on their own (a key equal to one of these,
|
|
49
|
+
# after suffix stripping, is a role label or an over-broad match, not a company).
|
|
50
|
+
_OVERBROAD = frozenset(
|
|
51
|
+
{
|
|
52
|
+
"bank", "company", "co", "parties", "party", "buyer", "seller", "purchaser", "vendor",
|
|
53
|
+
"supplier", "licensor", "licensee", "customer", "client", "contractor", "agent", "lender",
|
|
54
|
+
"borrower", "guarantor", "corporation", "corp", "trust", "group", "holdings", "holding",
|
|
55
|
+
"affiliate", "affiliates", "subsidiary", "the company", "the parties",
|
|
56
|
+
# bare generic-token fragments that are not entities on their own
|
|
57
|
+
"services", "solutions", "systems", "technologies", "technology", "international",
|
|
58
|
+
"enterprises", "industries", "communications", "networks", "media", "capital",
|
|
59
|
+
"management", "ventures", "partners", "associates", "advisors", "advisory", "consulting",
|
|
60
|
+
"worldwide", "global", "products",
|
|
61
|
+
}
|
|
62
|
+
)
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def _strip_alias_clause(name: str) -> str:
|
|
66
|
+
"""Cut an alias clause ('... d/b/a X', '... formerly known as Y'), keeping the name before it."""
|
|
67
|
+
match = _ALIAS_MARKER.search(name)
|
|
68
|
+
return name[: match.start()].strip(" ,;(") if match else name
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def normalize_entity_name(name: str) -> str:
|
|
72
|
+
"""Clustering key: lowercase, surrounding quotes/parens and legal suffixes stripped, whitespace
|
|
73
|
+
collapsed. Merges `Bank of America`, `Bank of America, N.A.`, `Bank of America, N. A`."""
|
|
74
|
+
text = _strip_alias_clause(name).strip().strip("\"'()[]").lower()
|
|
75
|
+
text = _APOSTROPHE.sub("", text) # possessive: "stremick's" -> "stremicks"
|
|
76
|
+
prev = None
|
|
77
|
+
while prev != text: # strip stacked suffixes ("co., ltd.")
|
|
78
|
+
prev = text
|
|
79
|
+
text = _LEGAL_SUFFIX.sub("", text).strip(" ,.")
|
|
80
|
+
return _NON_ALNUM.sub(" ", text).strip()
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def is_entity(name: str) -> bool:
|
|
84
|
+
"""Reject non-entities: template placeholders, role artifacts, and bare over-broad tokens."""
|
|
85
|
+
stripped = name.strip().strip("\"'()[]")
|
|
86
|
+
if not stripped or _PLACEHOLDER.search(name) or _ROLE_PHRASE.search(name):
|
|
87
|
+
return False
|
|
88
|
+
if stripped.lower().startswith("collectively"):
|
|
89
|
+
return False
|
|
90
|
+
key = normalize_entity_name(name)
|
|
91
|
+
return bool(key) and key not in _OVERBROAD
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
class EntityCluster(BaseModel):
|
|
95
|
+
"""One real-world entity: its clustering key, a representative surface form, and all variants."""
|
|
96
|
+
|
|
97
|
+
key: str
|
|
98
|
+
representative: str # a full surface form (the longest), for display + EDGAR matching
|
|
99
|
+
variants: list[str]
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def cluster_entities(names: list[str]) -> list[EntityCluster]:
|
|
103
|
+
"""Cluster surface forms into entities by normalized key (non-entities rejected first)."""
|
|
104
|
+
groups: dict[str, list[str]] = {}
|
|
105
|
+
for name in names:
|
|
106
|
+
if not is_entity(name):
|
|
107
|
+
continue
|
|
108
|
+
key = normalize_entity_name(name)
|
|
109
|
+
groups.setdefault(key, [])
|
|
110
|
+
if name not in groups[key]:
|
|
111
|
+
groups[key].append(name)
|
|
112
|
+
clusters = [
|
|
113
|
+
EntityCluster(key=key, representative=max(variants, key=len), variants=sorted(variants))
|
|
114
|
+
for key, variants in groups.items()
|
|
115
|
+
]
|
|
116
|
+
return sorted(clusters, key=lambda c: c.key)
|
|
@@ -0,0 +1,153 @@
|
|
|
1
|
+
"""CUAD parsing + scanned-PDF detection (T7, docs/archive/plans/Corpus_Acquisition.md).
|
|
2
|
+
|
|
3
|
+
`party_entities` extracts the real company parties from the CUAD `Parties` annotation, which mixes
|
|
4
|
+
company names with defined-term role labels ("Company", "MA", "Marketing Affiliate"). It keeps only
|
|
5
|
+
names carrying a corporate suffix and drops bare role labels, conservatively: a role label treated
|
|
6
|
+
as an entity would create a *phantom* shared-party link and inflate the multi-hop material with
|
|
7
|
+
false connections. Better to miss a party than to invent a shared one.
|
|
8
|
+
|
|
9
|
+
`is_scanned_pdf` detects an image-only PDF via the system `pdftotext` using a **low character-count
|
|
10
|
+
threshold, not strict-zero**, so a stray character on a scanned page does not undercount the scanned
|
|
11
|
+
subset (which would starve the vision-to-text path).
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
import ast
|
|
17
|
+
import csv
|
|
18
|
+
import re
|
|
19
|
+
import subprocess
|
|
20
|
+
import tempfile
|
|
21
|
+
from pathlib import Path
|
|
22
|
+
|
|
23
|
+
from rag_wright.contracts.identifiers import canonical_source_doc_id
|
|
24
|
+
from rag_wright.corpus.selection import ContractMeta
|
|
25
|
+
|
|
26
|
+
RASTER_DPI = 150 # render resolution for the synthetic image-only PDFs (readable for OCR)
|
|
27
|
+
|
|
28
|
+
# Corporate-form tokens that mark a party string as a real company (not a role label).
|
|
29
|
+
_CORP_SUFFIX = re.compile(
|
|
30
|
+
r"\b("
|
|
31
|
+
r"inc|incorporated|corp|corporation|company|co|llc|l\.?l\.?c|llp|lp|ltd|limited|plc|"
|
|
32
|
+
r"gmbh|ag|nv|n\.?v|sa|s\.?a|spa|holdings|group|international|technologies|technology|"
|
|
33
|
+
r"systems|networks|solutions|services|pharmaceuticals|labs|laboratories|ventures|partners|"
|
|
34
|
+
r"associates|enterprises|industries|bank|trust|capital|media|communications"
|
|
35
|
+
r")\b",
|
|
36
|
+
re.IGNORECASE,
|
|
37
|
+
)
|
|
38
|
+
|
|
39
|
+
# Bare defined-term role labels that must never be treated as entities, even when they coincide
|
|
40
|
+
# with a corporate-form word ("Company", "Co"). A role label as an entity would invent a shared
|
|
41
|
+
# party and inflate the multi-hop material with false links.
|
|
42
|
+
_ROLE_LABELS = {
|
|
43
|
+
"company", "co", "the company", "parties", "party", "buyer", "seller", "purchaser", "vendor",
|
|
44
|
+
"supplier", "licensor", "licensee", "customer", "client", "contractor", "agent", "lender",
|
|
45
|
+
"borrower", "guarantor", "distributor", "reseller", "manufacturer", "consultant", "partner",
|
|
46
|
+
"affiliate", "marketing affiliate", "ma",
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
SCANNED_CHAR_THRESHOLD = 300 # first-2-page text below this -> scanned (image-only)
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def party_entities(raw: str) -> list[str]:
|
|
53
|
+
"""Company parties from a CUAD `Parties` field (a stringified list), role labels dropped."""
|
|
54
|
+
try:
|
|
55
|
+
items = ast.literal_eval(raw) if raw and raw.strip().startswith("[") else []
|
|
56
|
+
except (ValueError, SyntaxError):
|
|
57
|
+
items = []
|
|
58
|
+
seen: set[str] = set()
|
|
59
|
+
out: list[str] = []
|
|
60
|
+
for item in items:
|
|
61
|
+
name = str(item).strip()
|
|
62
|
+
norm = name.lower().rstrip(".").strip()
|
|
63
|
+
key = name.lower()
|
|
64
|
+
if name and norm not in _ROLE_LABELS and _CORP_SUFFIX.search(name) and key not in seen:
|
|
65
|
+
seen.add(key)
|
|
66
|
+
out.append(name)
|
|
67
|
+
return out
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def is_scanned_by_chars(char_count: int, threshold: int = SCANNED_CHAR_THRESHOLD) -> bool:
|
|
71
|
+
"""Threshold decision: fewer than `threshold` extractable characters -> scanned."""
|
|
72
|
+
return char_count < threshold
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def pdf_text_chars(pdf_path: Path, *, pages: int = 2) -> int:
|
|
76
|
+
"""Characters `pdftotext` extracts from the first `pages` pages (0 for a purely scanned PDF)."""
|
|
77
|
+
try:
|
|
78
|
+
result = subprocess.run(
|
|
79
|
+
["pdftotext", "-l", str(pages), "-q", str(pdf_path), "-"],
|
|
80
|
+
capture_output=True,
|
|
81
|
+
timeout=60,
|
|
82
|
+
)
|
|
83
|
+
return len(result.stdout.decode("utf-8", "ignore").strip())
|
|
84
|
+
except (subprocess.SubprocessError, OSError):
|
|
85
|
+
return 0
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def is_scanned_pdf(pdf_path: Path, *, threshold: int = SCANNED_CHAR_THRESHOLD) -> bool:
|
|
89
|
+
"""Whether a PDF is scanned/image-only (low extractable-text character count)."""
|
|
90
|
+
return is_scanned_by_chars(pdf_text_chars(pdf_path), threshold)
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def rasterize_to_image_pdf(src_pdf: Path, dst_pdf: Path, *, dpi: int = RASTER_DPI) -> int:
|
|
94
|
+
"""Render `src_pdf`'s pages to images and write an **image-only** PDF (no text layer) to `dst_pdf`.
|
|
95
|
+
|
|
96
|
+
CUAD ships no image-only PDFs (all carry a text layer), so to exercise Docling's OCR /
|
|
97
|
+
vision-to-text path we deterministically re-render a few contracts as scans: poppler `pdftoppm`
|
|
98
|
+
rasterizes each page to PNG, Pillow assembles them into a PDF with no extractable text. Given the
|
|
99
|
+
same source and `dpi` this is reproducible. Returns the page count.
|
|
100
|
+
"""
|
|
101
|
+
from PIL import Image # lazy: only the rasterization path needs Pillow
|
|
102
|
+
|
|
103
|
+
dst_pdf.parent.mkdir(parents=True, exist_ok=True)
|
|
104
|
+
with tempfile.TemporaryDirectory() as td:
|
|
105
|
+
prefix = Path(td) / "page"
|
|
106
|
+
subprocess.run(
|
|
107
|
+
["pdftoppm", "-png", "-r", str(dpi), str(src_pdf), str(prefix)],
|
|
108
|
+
check=True,
|
|
109
|
+
capture_output=True,
|
|
110
|
+
timeout=300,
|
|
111
|
+
)
|
|
112
|
+
pages = sorted(Path(td).glob("page*.png"))
|
|
113
|
+
if not pages:
|
|
114
|
+
raise RuntimeError(f"pdftoppm produced no pages for {src_pdf}")
|
|
115
|
+
images = [Image.open(p).convert("RGB") for p in pages]
|
|
116
|
+
images[0].save(dst_pdf, save_all=True, append_images=images[1:])
|
|
117
|
+
return len(images)
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def _agreement_type(pdf_path: Path) -> str:
|
|
121
|
+
"""The agreement type is the PDF's immediate folder name, underscores folded to spaces."""
|
|
122
|
+
return pdf_path.parent.name.replace("_", " ").strip()
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def load_contract_metadata(extracted_root: Path) -> list[ContractMeta]:
|
|
126
|
+
"""Build `ContractMeta` for every CUAD contract that has both a CSV row and a PDF.
|
|
127
|
+
|
|
128
|
+
Runs `pdftotext` per PDF (scanned detection), so this is the slow, I/O part. `extracted_root` is
|
|
129
|
+
the extracted `CUAD_v1/` directory (holding `master_clauses.csv` and `full_contract_pdf/`).
|
|
130
|
+
"""
|
|
131
|
+
pdfs = {
|
|
132
|
+
p.stem.lower(): p
|
|
133
|
+
for p in (extracted_root / "full_contract_pdf").rglob("*")
|
|
134
|
+
if p.suffix.lower() == ".pdf"
|
|
135
|
+
}
|
|
136
|
+
metas: list[ContractMeta] = []
|
|
137
|
+
with open(extracted_root / "master_clauses.csv", newline="", encoding="utf-8") as f:
|
|
138
|
+
for row in csv.DictReader(f):
|
|
139
|
+
filename = (row.get("Filename") or "").strip()
|
|
140
|
+
stem = Path(filename).stem.lower()
|
|
141
|
+
pdf = pdfs.get(stem)
|
|
142
|
+
if pdf is None:
|
|
143
|
+
continue
|
|
144
|
+
metas.append(
|
|
145
|
+
ContractMeta(
|
|
146
|
+
contract_id=canonical_source_doc_id(Path(filename).stem),
|
|
147
|
+
agreement_type=_agreement_type(pdf),
|
|
148
|
+
parties=party_entities(row.get("Parties", "")),
|
|
149
|
+
is_scanned=is_scanned_pdf(pdf),
|
|
150
|
+
size_bytes=pdf.stat().st_size,
|
|
151
|
+
)
|
|
152
|
+
)
|
|
153
|
+
return metas
|
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
"""CUAD ingestion: the reference `CorpusAdapter` + the CUAD ingest driver.
|
|
2
|
+
|
|
3
|
+
Relocated out of the GENERIC `subgraphs.contract_ingestion_pipeline` (which stays corpus-agnostic) so the
|
|
4
|
+
dataset-specific adapter and driver live with the other per-corpus code in `corpus/`. The generic machinery
|
|
5
|
+
(the `CorpusAdapter` seam, `run_corpus_ingestion`, `production_document_ingest`, and the cache-seed helpers) is
|
|
6
|
+
imported from the pipeline; only the CUAD parsing + wiring lives here. Adding another corpus = a sibling adapter
|
|
7
|
+
+ driver here, never touching the generic pipeline.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
from typing import Any, Iterable
|
|
13
|
+
|
|
14
|
+
from rag_wright.subgraphs.contract_ingestion_pipeline import (
|
|
15
|
+
IngestionReport,
|
|
16
|
+
SourceDocument,
|
|
17
|
+
aproduction_document_ingest,
|
|
18
|
+
arun_corpus_ingestion,
|
|
19
|
+
seed_chunk_cache,
|
|
20
|
+
)
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class CuadAdapter:
|
|
24
|
+
"""The REFERENCE `CorpusAdapter` (ADR-locked design): CUAD -> `SourceDocument`s. It is the ONLY CUAD-specific
|
|
25
|
+
code in the ingest path -- it parses the corpus (CUAD ships text in JSON, so no docling parse), assigns the
|
|
26
|
+
ONE canonical `source_doc_id` (HYG-1), and passes the raw title as metadata. Adding another corpus means
|
|
27
|
+
writing a sibling adapter (e.g. `AcordAdapter` carrying pre-segmented spans in `metadata`); the pipeline and
|
|
28
|
+
driver do not change. `run_corpus_ingestion(CuadAdapter(path), ingest_graph, link_fn=...)` ingests it."""
|
|
29
|
+
|
|
30
|
+
def __init__(self, cuad_path: Any, *, limit: int = 0) -> None:
|
|
31
|
+
self._path = cuad_path
|
|
32
|
+
self._limit = limit
|
|
33
|
+
|
|
34
|
+
def documents(self) -> Iterable[SourceDocument]:
|
|
35
|
+
from rag_wright.contracts.identifiers import canonical_source_doc_id
|
|
36
|
+
from rag_wright.spans.cuad_labels import parse_cuad
|
|
37
|
+
|
|
38
|
+
contracts = list(parse_cuad(self._path))
|
|
39
|
+
if self._limit:
|
|
40
|
+
contracts = contracts[: self._limit]
|
|
41
|
+
for contract in contracts:
|
|
42
|
+
yield SourceDocument(
|
|
43
|
+
source_doc_id=canonical_source_doc_id(contract.contract_id),
|
|
44
|
+
text=contract.context,
|
|
45
|
+
metadata={"raw_title": contract.contract_id},
|
|
46
|
+
)
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
async def arun_cuad_ingestion(cuad_path: Any, store: Any, *, cache_dir: Any, limit: int = 0) -> IngestionReport:
|
|
50
|
+
"""INGEST-REFACTOR proof: ingest CUAD through the GENERIC pipeline + `CuadAdapter` -- one call, no
|
|
51
|
+
`ingest_cuad()`. Builds the EDGAR registry, ensures the schema, ingests `limit` documents. Point `store` at a SCRATCH database for a non-destructive smoke.
|
|
52
|
+
|
|
53
|
+
INGEST-REFACTOR (a) cache reuse (the ONLY CUAD-specific wiring): the pipeline reuses GP-1B's party
|
|
54
|
+
extractions (`dg_extracted_parties.json`, ~482) via `party_seed_path`, and prior `chunk()` manifests
|
|
55
|
+
(`data/cache/cuad/chunks`, content-hash keyed so only true matches are reused) copied into the run's cache.
|
|
56
|
+
The unavoidable cost that remains is clause property extraction (the template changed since those were cached)."""
|
|
57
|
+
import json
|
|
58
|
+
from pathlib import Path
|
|
59
|
+
|
|
60
|
+
from rag_wright.capabilities.dg_extraction import build_verified_registry
|
|
61
|
+
|
|
62
|
+
store.ensure_schema()
|
|
63
|
+
seed_chunk_cache(Path(cache_dir) / "chunks", Path("data/cache/cuad/chunks"))
|
|
64
|
+
vset = json.loads(Path("data/edgar/verification_set.json").read_text(encoding="utf-8"))
|
|
65
|
+
ingest_graph = aproduction_document_ingest(
|
|
66
|
+
store, cache_dir=cache_dir, registry=build_verified_registry(vset),
|
|
67
|
+
party_seed_path="data/cache/dg_extracted_parties.json")
|
|
68
|
+
return await arun_corpus_ingestion(
|
|
69
|
+
CuadAdapter(cuad_path, limit=limit), ingest_graph,
|
|
70
|
+
# (issue 0028 / ADR-0091: the KG-7 PartyTo link step was retired; no link_fn is wired.)
|
|
71
|
+
# RESUME-skip: a present Contract node means the whole document already landed (Contract is written last).
|
|
72
|
+
is_done=lambda doc: store.contract_by_id(doc.source_doc_id) is not None)
|