rag-wright 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- rag_wright/__init__.py +13 -0
- rag_wright/api/__init__.py +33 -0
- rag_wright/api/config.py +59 -0
- rag_wright/api/discover.py +70 -0
- rag_wright/api/documents.py +39 -0
- rag_wright/api/ids.py +31 -0
- rag_wright/api/invoke.py +99 -0
- rag_wright/api/kg.py +61 -0
- rag_wright/api/mcp.py +94 -0
- rag_wright/api/usage.py +30 -0
- rag_wright/api/workspace.py +85 -0
- rag_wright/capabilities/__init__.py +8 -0
- rag_wright/capabilities/answer_generator.py +427 -0
- rag_wright/capabilities/ard.py +286 -0
- rag_wright/capabilities/assertion_extraction.py +79 -0
- rag_wright/capabilities/chunk_read.py +58 -0
- rag_wright/capabilities/chunk_write.py +163 -0
- rag_wright/capabilities/claim_extraction.py +153 -0
- rag_wright/capabilities/clause_exception_linking.py +117 -0
- rag_wright/capabilities/compliance_judgment.py +322 -0
- rag_wright/capabilities/compliance_store.py +87 -0
- rag_wright/capabilities/contract_kg_serve.py +156 -0
- rag_wright/capabilities/contract_kg_store.py +251 -0
- rag_wright/capabilities/dg_extraction.py +585 -0
- rag_wright/capabilities/disambiguation.py +163 -0
- rag_wright/capabilities/document_parse.py +87 -0
- rag_wright/capabilities/document_scope.py +49 -0
- rag_wright/capabilities/embedding.py +164 -0
- rag_wright/capabilities/embedding_profiles.py +43 -0
- rag_wright/capabilities/entity_resolution.py +154 -0
- rag_wright/capabilities/fusion.py +64 -0
- rag_wright/capabilities/graph_extraction.py +243 -0
- rag_wright/capabilities/graph_query.py +73 -0
- rag_wright/capabilities/graph_storage.py +111 -0
- rag_wright/capabilities/highlight_serve.py +142 -0
- rag_wright/capabilities/hybrid_search.py +65 -0
- rag_wright/capabilities/invoke.py +31 -0
- rag_wright/capabilities/jev_decision.py +38 -0
- rag_wright/capabilities/manifests.py +872 -0
- rag_wright/capabilities/okf_navigate.py +456 -0
- rag_wright/capabilities/parsing.py +286 -0
- rag_wright/capabilities/property_boosted_retrieval.py +125 -0
- rag_wright/capabilities/query_function_classifier.py +94 -0
- rag_wright/capabilities/query_understanding.py +109 -0
- rag_wright/capabilities/registry.py +262 -0
- rag_wright/capabilities/remote_encoders.py +94 -0
- rag_wright/capabilities/requirement_extraction.py +247 -0
- rag_wright/capabilities/reranking.py +123 -0
- rag_wright/capabilities/retrieval_core.py +126 -0
- rag_wright/capabilities/rlm_chunking.py +808 -0
- rag_wright/capabilities/rlm_synthesis.py +316 -0
- rag_wright/capabilities/scan_quality.py +136 -0
- rag_wright/capabilities/span_relevance_judgment.py +191 -0
- rag_wright/capabilities/vision_to_text.py +85 -0
- rag_wright/capabilities/vlm_ocr.py +85 -0
- rag_wright/contracts/__init__.py +6 -0
- rag_wright/contracts/chunk.py +79 -0
- rag_wright/contracts/compliance.py +303 -0
- rag_wright/contracts/contract_meta.py +27 -0
- rag_wright/contracts/extraction.py +130 -0
- rag_wright/contracts/function.py +167 -0
- rag_wright/contracts/function_routing.py +91 -0
- rag_wright/contracts/highlight.py +74 -0
- rag_wright/contracts/identifiers.py +153 -0
- rag_wright/contracts/jurisdiction.py +96 -0
- rag_wright/contracts/ontology.py +142 -0
- rag_wright/contracts/property.py +201 -0
- rag_wright/contracts/provenance.py +78 -0
- rag_wright/contracts/query_intent.py +53 -0
- rag_wright/contracts/span.py +76 -0
- rag_wright/contracts/value_match.py +84 -0
- rag_wright/corpus/__init__.py +0 -0
- rag_wright/corpus/canonicalize.py +116 -0
- rag_wright/corpus/cuad.py +153 -0
- rag_wright/corpus/cuad_ingestion.py +72 -0
- rag_wright/corpus/document_parser.py +299 -0
- rag_wright/corpus/edgar.py +231 -0
- rag_wright/corpus/gcs_ingestion.py +120 -0
- rag_wright/corpus/http.py +110 -0
- rag_wright/corpus/selection.py +152 -0
- rag_wright/mcp/__init__.py +11 -0
- rag_wright/mcp/compliance_server.py +299 -0
- rag_wright/mcp/intra_document_qa_server.py +170 -0
- rag_wright/mcp/relational_qa_server.py +171 -0
- rag_wright/mcp/session_store.py +64 -0
- rag_wright/mcp/typed_property_retrieval_server.py +191 -0
- rag_wright/models/__init__.py +8 -0
- rag_wright/models/profiles.py +331 -0
- rag_wright/models/seam.py +497 -0
- rag_wright/models/tag_structured.py +285 -0
- rag_wright/models/tracing.py +179 -0
- rag_wright/models/usage.py +102 -0
- rag_wright/okf/__init__.py +11 -0
- rag_wright/okf/compile.py +292 -0
- rag_wright/okf/document.py +47 -0
- rag_wright/okf/enrich.py +176 -0
- rag_wright/okf/links.py +190 -0
- rag_wright/okf/lint.py +105 -0
- rag_wright/ontology/__init__.py +6 -0
- rag_wright/ontology/_generated_template_meta.py +60 -0
- rag_wright/ontology/_generated_vocab.py +52 -0
- rag_wright/ontology/clause_template.py +964 -0
- rag_wright/ontology/codegen.py +84 -0
- rag_wright/ontology/compliance_bridge.ttl +186 -0
- rag_wright/ontology/contract_bridge.ttl +2685 -0
- rag_wright/ontology/contract_taxonomy.py +24 -0
- rag_wright/ontology/derive.py +58 -0
- rag_wright/ontology/loader.py +435 -0
- rag_wright/ontology/packs/ftc_16cfr255.ttl +29 -0
- rag_wright/ontology/registry.py +87 -0
- rag_wright/ontology/template_introspect.py +100 -0
- rag_wright/py.typed +0 -0
- rag_wright/reference/__init__.py +2 -0
- rag_wright/reference/compliance.py +41 -0
- rag_wright/reference/contract_seam.py +123 -0
- rag_wright/skills/__init__.py +7 -0
- rag_wright/skills/claim_extraction/SKILL.md +47 -0
- rag_wright/skills/claim_extraction/__init__.py +1 -0
- rag_wright/skills/claim_extraction/template.py +50 -0
- rag_wright/skills/compliance_judgment/SKILL.md +59 -0
- rag_wright/skills/corpus_ingest/SKILL.md +106 -0
- rag_wright/skills/extraction_semantic_judge/SKILL.md +51 -0
- rag_wright/skills/extraction_semantic_judge/__init__.py +1 -0
- rag_wright/skills/generation/SKILL.md +64 -0
- rag_wright/skills/generation/__init__.py +1 -0
- rag_wright/skills/generic_compliance_judgment/SKILL.md +58 -0
- rag_wright/skills/okf_navigate/SKILL.md +137 -0
- rag_wright/skills/requirement_extraction/SKILL.md +47 -0
- rag_wright/skills/requirement_extraction/__init__.py +1 -0
- rag_wright/skills/requirement_extraction/template.py +50 -0
- rag_wright/skills/rlm/SKILL.md +186 -0
- rag_wright/skills/rlm/__init__.py +31 -0
- rag_wright/skills/rlm/agent.py +292 -0
- rag_wright/skills/span_relevance_judgment/SKILL.md +67 -0
- rag_wright/skills/vision_to_text/SKILL.md +36 -0
- rag_wright/skills/vision_to_text/__init__.py +1 -0
- rag_wright/spans/__init__.py +1 -0
- rag_wright/spans/boundary.py +78 -0
- rag_wright/spans/clause_function_classifier.py +490 -0
- rag_wright/spans/clause_kg_extractor.py +337 -0
- rag_wright/spans/cuad_labels.py +81 -0
- rag_wright/spans/dim_classifier.py +158 -0
- rag_wright/spans/dim_fleet.json +411 -0
- rag_wright/spans/function_classifier.py +77 -0
- rag_wright/spans/function_families.py +62 -0
- rag_wright/spans/hybrid_classifier.py +103 -0
- rag_wright/spans/legalbert_classifier.py +83 -0
- rag_wright/spans/model_capabilities.py +107 -0
- rag_wright/spans/new_function_labels.py +111 -0
- rag_wright/spans/page_map.py +68 -0
- rag_wright/spans/property_extractor.py +365 -0
- rag_wright/spans/property_grounding.py +182 -0
- rag_wright/spans/reclassify.py +77 -0
- rag_wright/spans/scarce_function_labels.py +105 -0
- rag_wright/spans/segment.py +341 -0
- rag_wright/spans/semantic_judge.py +197 -0
- rag_wright/spans/symbolic_validation.py +131 -0
- rag_wright/spans/tag_clause_extractor.py +182 -0
- rag_wright/store/__init__.py +6 -0
- rag_wright/store/arcadedb.py +1135 -0
- rag_wright/store/chunk_text.py +66 -0
- rag_wright/store/seam.py +213 -0
- rag_wright/subgraphs/__init__.py +0 -0
- rag_wright/subgraphs/async_ingestion.py +204 -0
- rag_wright/subgraphs/compliance_check.py +1042 -0
- rag_wright/subgraphs/compliance_ingestion.py +306 -0
- rag_wright/subgraphs/contract_ingestion_pipeline.py +999 -0
- rag_wright/subgraphs/graph_extraction.py +102 -0
- rag_wright/subgraphs/intra_document_qa.py +328 -0
- rag_wright/subgraphs/observability.py +140 -0
- rag_wright/subgraphs/query_constraint_extraction.py +73 -0
- rag_wright/subgraphs/relational_qa.py +165 -0
- rag_wright/subgraphs/requirement_extraction.py +137 -0
- rag_wright/subgraphs/scaffold.py +65 -0
- rag_wright/subgraphs/semantic_chunking.py +183 -0
- rag_wright/subgraphs/typed_clause_extraction.py +172 -0
- rag_wright/subgraphs/typed_property_retrieval.py +278 -0
- rag_wright/util/__init__.py +1 -0
- rag_wright/util/concurrent.py +153 -0
- rag_wright/util/spacy_model.py +45 -0
- rag_wright-0.1.0.dist-info/METADATA +168 -0
- rag_wright-0.1.0.dist-info/RECORD +184 -0
- rag_wright-0.1.0.dist-info/WHEEL +4 -0
- rag_wright-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,131 @@
|
|
|
1
|
+
"""JUDGE-ONTOLOGY-1 (ADR-0040): the symbolic ontology-validation gate -- layer 2 of the
|
|
2
|
+
neuro-symbolic extraction-fidelity cascade.
|
|
3
|
+
|
|
4
|
+
Where the lexical grounding judge (`property_grounding.reground`, ADR-0028) checks whether a value's
|
|
5
|
+
surface cue appears in the text, this gate checks whether the ASSERTED DIMENSION is even APPLICABLE to
|
|
6
|
+
the clause's FUNCTION -- a TYPE constraint the lexical judge structurally cannot see. Example: granite
|
|
7
|
+
emits `nonsolicit_target=employees` on an Anti-Assignment clause. The value is a valid nonsolicit_target
|
|
8
|
+
and its cue may appear in the text, so the lexical judge passes it; but `nonsolicit_target` does not
|
|
9
|
+
belong to Anti-Assignment at all -- a type error, caught here.
|
|
10
|
+
|
|
11
|
+
The applicability map (`FUNCTION_APPLICABLE_DIMS`) is the NEW ontology content ADR-0040 calls for: each
|
|
12
|
+
clause FUNCTION permits a set of property DIMENSIONS. It is compiled to SHACL `sh:closed` NodeShapes (one
|
|
13
|
+
per function, listing the applicable dimension paths) and validated with `pyshacl` -- the symbolic half of
|
|
14
|
+
neuro-symbolic. A non-applicable assertion is downgraded to AMBIGUOUS (kept but flagged), exactly like
|
|
15
|
+
`reground`, so the shared property-value nodes stay clean and soft-boost down-weights it.
|
|
16
|
+
|
|
17
|
+
The gate also enforces CARDINALITY (JUDGE-ONTOLOGY-2): a SCALAR dimension asserted with two conflicting
|
|
18
|
+
values (e.g. granite hedging `cap_basis` = both `fixed_fee` and `multiple_of_fees`) violates `sh:maxCount 1`
|
|
19
|
+
and both values are downgraded -- the real intra-clause "contradiction" class, since the dimensions are
|
|
20
|
+
orthogonal facets and same-dimension conflict is where extraction actually contradicts itself. The 3
|
|
21
|
+
multi-valued dimensions (`_LIST_ENUM_DIMS`: carve_out / covered_subject / damage_type) are left unbounded.
|
|
22
|
+
|
|
23
|
+
The gate also enforces DEONTIC consistency (JUDGE-ONTOLOGY-3, ODRL): a consent-regime dimension carries a
|
|
24
|
+
permission↔restriction polarity in its VALUES (`free`/`unrestricted` = "may freely"; `consent_required` =
|
|
25
|
+
restricted). A function whose defining purpose is to RESTRICT (Anti-Assignment / Non-Transferable License /
|
|
26
|
+
Change Of Control) contradicts a permission-polarity value -- the observed `assignment_consent=free` on a
|
|
27
|
+
"shall not assign" clause. The clause's rule type is DERIVED from its function (reliable, non-circular; not
|
|
28
|
+
parsed from the text), and the check is a `sh:in` (allowed = vocab minus the permission-polarity values) on
|
|
29
|
+
the scoped property shape.
|
|
30
|
+
|
|
31
|
+
Deliberately NOT enforced here (JUDGE-ONTOLOGY-2 scoping): value-in-vocabulary (`sh:in` over the full vocab)
|
|
32
|
+
is already enforced at the Pydantic contract boundary (`property.PropertyAssertion._value_in_vocab_or_ambiguous`),
|
|
33
|
+
so a whole-vocab SHACL shape would duplicate a working validator; and cross-DIMENSION `sh:sparql` rules are
|
|
34
|
+
omitted because this schema's dimensions are orthogonal facets with no hard intra-clause cross-dimension
|
|
35
|
+
contradiction (deferred to a post-MVP / beta-customer iteration on real production data).
|
|
36
|
+
|
|
37
|
+
Deterministic, no model, no network. Downgrade is confidence-independent (a type/cardinality/deontic error is
|
|
38
|
+
wrong whether EXTRACTED or INFERRED); an already-AMBIGUOUS assertion is left as is. A function NOT in the map
|
|
39
|
+
is PERMISSIVE (unvalidated) -- coverage is expanded deliberately, never by guessing a closed set we are unsure
|
|
40
|
+
of. All three checks reuse one record->RDF->pyshacl harness.
|
|
41
|
+
"""
|
|
42
|
+
|
|
43
|
+
from __future__ import annotations
|
|
44
|
+
|
|
45
|
+
from rdflib import Graph, Literal, Namespace, URIRef
|
|
46
|
+
from rdflib.namespace import RDF, SH
|
|
47
|
+
|
|
48
|
+
from rag_wright.contracts.property import (
|
|
49
|
+
ClausePropertyRecord,
|
|
50
|
+
PropertyAssertion,
|
|
51
|
+
PropertyDimension,
|
|
52
|
+
)
|
|
53
|
+
from rag_wright.contracts.provenance import ConfidenceTag
|
|
54
|
+
|
|
55
|
+
_CBR = Namespace("https://ragwright.local/ontology/contract-bridge#")
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def _function_class(function: str) -> URIRef:
|
|
59
|
+
"""A stable CBR class IRI for a function label (RDF has no spaces; encode deterministically)."""
|
|
60
|
+
return _CBR[f"Function_{function.replace(' ', '_')}"]
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def _dim_property(dimension: PropertyDimension) -> URIRef:
|
|
64
|
+
"""The CBR predicate IRI carrying a dimension's asserted value on a clause node."""
|
|
65
|
+
return _CBR[f"dim_{dimension.value}"]
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def _shapes_graph() -> Graph:
|
|
69
|
+
"""ADR-0066 P2: the SHACL shapes come FROM contract_bridge.ttl (the source of truth) -- pyshacl reads the
|
|
70
|
+
persisted sh:NodeShapes directly; no Python-built shapes. The ttl non-SHACL triples are ignored by pyshacl.
|
|
71
|
+
"""
|
|
72
|
+
from rag_wright.ontology.loader import load_shapes_graph
|
|
73
|
+
|
|
74
|
+
return load_shapes_graph()
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def _record_to_rdf(record: ClausePropertyRecord) -> Graph:
|
|
78
|
+
"""Serialize a record's function + assertions to a tiny data graph: the clause typed by its function,
|
|
79
|
+
with one triple per assertion (dimension predicate -> value literal)."""
|
|
80
|
+
g = Graph()
|
|
81
|
+
clause = URIRef("urn:clause:" + record.clause_id)
|
|
82
|
+
g.add((clause, RDF.type, _function_class(record.function)))
|
|
83
|
+
for a in record.assertions:
|
|
84
|
+
g.add((clause, _dim_property(a.dimension), Literal(a.value)))
|
|
85
|
+
return g
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def flagged_dimensions(record: ClausePropertyRecord) -> set[PropertyDimension]:
|
|
89
|
+
"""The dimensions to downgrade: ONLY the FUNCTION-INDEPENDENT contradiction check -- a scalar dimension
|
|
90
|
+
asserted with more than one value (`sh:maxCount 1`). The FUNCTION-DEPENDENT checks are DELIBERATELY IGNORED
|
|
91
|
+
(ADR-0082): clause-function classification is not accurate enough to be load-bearing (~0.5 top-1; memory
|
|
92
|
+
`function-classification-not-load-bearing`), so `sh:closed` (dimension-not-applicable-to-function) and `sh:in`
|
|
93
|
+
(deontic polarity on a restrictive function) would downgrade CORRECT cross-cutting extractions based on an
|
|
94
|
+
unreliable (and often narrow) function map. Function is a KG tag / query-time soft signal, never an ingest
|
|
95
|
+
gate. Empty if the record conforms or the function is unmodeled."""
|
|
96
|
+
if not record.assertions:
|
|
97
|
+
return set()
|
|
98
|
+
from pyshacl import validate
|
|
99
|
+
|
|
100
|
+
conforms, results_graph, _ = validate(
|
|
101
|
+
_record_to_rdf(record), shacl_graph=_shapes_graph(), inference="none", advanced=False
|
|
102
|
+
)
|
|
103
|
+
if conforms:
|
|
104
|
+
return set()
|
|
105
|
+
flagged: set[PropertyDimension] = set()
|
|
106
|
+
_prefix = str(_CBR) + "dim_"
|
|
107
|
+
for result in results_graph.subjects(RDF.type, SH.ValidationResult):
|
|
108
|
+
# keep ONLY the contradiction (maxCount) violations; drop function-dependent closed/in violations
|
|
109
|
+
if results_graph.value(result, SH.sourceConstraintComponent) != SH.MaxCountConstraintComponent:
|
|
110
|
+
continue
|
|
111
|
+
path = results_graph.value(result, SH.resultPath)
|
|
112
|
+
if path is not None and str(path).startswith(_prefix):
|
|
113
|
+
flagged.add(PropertyDimension(str(path)[len(_prefix):]))
|
|
114
|
+
return flagged
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def symbolic_validate(record: ClausePropertyRecord) -> ClausePropertyRecord:
|
|
118
|
+
"""Quality gate (ADR-0040 layer 2): downgrade every assertion whose dimension violates a shape -- not
|
|
119
|
+
applicable to the clause's function, or a scalar dimension asserted with conflicting values -- to
|
|
120
|
+
AMBIGUOUS (kept but flagged, model-agnostic, confidence-independent). A no-op when the function is
|
|
121
|
+
unmodeled or every dimension is valid -- mirrors `property_grounding.reground`."""
|
|
122
|
+
bad = flagged_dimensions(record)
|
|
123
|
+
if not bad:
|
|
124
|
+
return record
|
|
125
|
+
new: list[PropertyAssertion] = [
|
|
126
|
+
a.model_copy(update={"confidence": ConfidenceTag.AMBIGUOUS})
|
|
127
|
+
if a.dimension in bad and a.confidence != ConfidenceTag.AMBIGUOUS
|
|
128
|
+
else a
|
|
129
|
+
for a in record.assertions
|
|
130
|
+
]
|
|
131
|
+
return record.model_copy(update={"assertions": new})
|
|
@@ -0,0 +1,182 @@
|
|
|
1
|
+
"""TAGPARSE-INGEST-1b: FUNCTION-INDEPENDENT clause property extraction via client-side tag-parse.
|
|
2
|
+
|
|
3
|
+
The docling-graph path asks a model to fill the whole 35-field `Clause` template in one server-side-JSON call --
|
|
4
|
+
which granite-4.2 flattens/fails (~5/7) on real clauses (ADR-0079). Function-SCOPED extraction was ruled out
|
|
5
|
+
because clause-function classification is not accurate enough to gate on (~0.5 top-1; memory
|
|
6
|
+
`function-classification-not-load-bearing`). So we decompose the schema WITHOUT the function: the 35 fields are
|
|
7
|
+
split into 7 cohesive THEMATIC groups, each a small tag-parse pass over the SAME `Clause` schema (via
|
|
8
|
+
`build_tag_structured(..., fields=group)`), run concurrently and merged into one `Clause`. Small focused schemas
|
|
9
|
+
are where tag-parse already wins (parties 5/5), and `Clause`'s own validators normalize on the final construct.
|
|
10
|
+
|
|
11
|
+
`CLAUSE_GROUPS` is domain knowledge (which properties belong together); it is a candidate for ontology migration
|
|
12
|
+
(ADR-0066) but lives here for now, like `property_extractor.FUNCTION_DIMENSIONS`. Every non-id `Clause` field is
|
|
13
|
+
in exactly one group (a coverage test enforces this).
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import asyncio
|
|
19
|
+
import os
|
|
20
|
+
from typing import Any
|
|
21
|
+
|
|
22
|
+
from rag_wright.models.profiles import DEFAULT_GENERAL
|
|
23
|
+
from rag_wright.models.tag_structured import _classify, build_tag_structured
|
|
24
|
+
from rag_wright.ontology.clause_template import Clause
|
|
25
|
+
|
|
26
|
+
# The 7 thematic groups (5a consents/control + 5b restrictions/duties per the design). document_reference is the
|
|
27
|
+
# clause's root id, filled separately (from context / source stem), never asked of the model here.
|
|
28
|
+
CLAUSE_GROUPS: dict[str, tuple[str, ...]] = {
|
|
29
|
+
"identity_scope": ("clause_type", "covers", "covers_party_scope", "has_mutuality",
|
|
30
|
+
"has_asymmetry", "has_favorability"),
|
|
31
|
+
"liability_damages": ("caps", "has_claim_scope", "prohibits_damage", "ld_trigger"),
|
|
32
|
+
"temporal_termination": ("bounded_by", "has_renewal", "has_termination_right", "condition_type"),
|
|
33
|
+
"ip_licensing": ("has_ip_ownership", "has_exclusivity_type", "has_right_of_first_type",
|
|
34
|
+
"has_mfn_scope", "royalty_basis"),
|
|
35
|
+
"consents_control": ("has_assignment_consent", "has_coc_consent", "has_escrow_release_trigger"),
|
|
36
|
+
"governing_law_dispute": ("governed_by", "dispute_method"), # split out of the grab-bag: a focused pass so
|
|
37
|
+
# jurisdiction isn't buried among 8 other fields
|
|
38
|
+
"restrictions_duties": ("requires_duty", "has_restriction_scope", "prohibits_solicit", "has_warranty_scope",
|
|
39
|
+
"audit_frequency", "commitment_quantum", "collateral_type"),
|
|
40
|
+
"exceptions": ("excepts", "confidentiality_exception", "force_majeure_event"),
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
_GROUP_PROMPT = (
|
|
44
|
+
"You are extracting the typed properties of ONE contract clause. Extract ONLY the properties below that are "
|
|
45
|
+
"EXPLICITLY stated in the clause; omit any that are not present -- do NOT guess or invent a value. Read each "
|
|
46
|
+
"value verbatim from the clause.\n\nCLAUSE:\n{text}"
|
|
47
|
+
)
|
|
48
|
+
|
|
49
|
+
# NOTE (issue 0036): TWO cost-cutting ideas were tried here and BOTH measured a ~15-18% property-recall loss on
|
|
50
|
+
# Qwen (the current default), concentrated in `excepts` (carve-outs) -- so neither was adopted. (1) A coarse
|
|
51
|
+
# "aspect gate" that pruned groups per clause (removed entirely). (2) Batching two clauses per group call (recall
|
|
52
|
+
# 0.818 at samples=1, 0.844 at samples=4 -- still a real regression, not variance). Splitting the model's attention
|
|
53
|
+
# across clauses, or pruning groups, both under-extract the hard cross-cutting fields. So every group always runs,
|
|
54
|
+
# one clause per call; the only recall-preserving cost lever is `is_extractable_span` (fewer spans reach here).
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
import re as _re
|
|
58
|
+
|
|
59
|
+
_MAX_VALUE_CHARS = 240 # a clause PROPERTY value is short ("12_months", "Delaware"); longer = leaked prose
|
|
60
|
+
_TAG_LIKE = _re.compile(r"<[A-Za-z_][\w-]*>") # a value must never contain XML tags (model reasoning/prompt leak)
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def _sane(value: Any) -> Any:
|
|
64
|
+
"""Drop a garbage string value (a leaked chain-of-thought / prompt echo): too long, or containing XML tags.
|
|
65
|
+
Non-string values pass through (enums/sub-models/lists can't carry this leak). A dropped value -> None so the
|
|
66
|
+
field falls back to its default (never store reasoning text as a clause property)."""
|
|
67
|
+
if isinstance(value, str) and (len(value) > _MAX_VALUE_CHARS or _TAG_LIKE.search(value)):
|
|
68
|
+
return None
|
|
69
|
+
return value
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def _uninformative(v: Any) -> bool:
|
|
73
|
+
"""A scalar value carrying no information: None/empty, or an enum OTHER escape (dropped downstream anyway)."""
|
|
74
|
+
if v is None:
|
|
75
|
+
return True
|
|
76
|
+
s = str(getattr(v, "value", v)).strip()
|
|
77
|
+
return s == "" or s in ("Other", "OTHER") or "Unknown" in s
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def _item_key(item: Any) -> Any:
|
|
81
|
+
"""A hashable identity for a list item so the union can dedup: an enum by its value, a sub-model by its dump."""
|
|
82
|
+
if hasattr(item, "model_dump"):
|
|
83
|
+
return tuple(sorted((k, str(v)) for k, v in item.model_dump().items()))
|
|
84
|
+
return getattr(item, "value", item)
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def _union_lists(lists: list[Any]) -> list[Any]:
|
|
88
|
+
"""Order-preserving UNION of a list-valued field across samples -- the fix for granite's list under-enumeration
|
|
89
|
+
(each sample may emit a different subset; the union recovers the full set)."""
|
|
90
|
+
out: list[Any] = []
|
|
91
|
+
seen: set[Any] = set()
|
|
92
|
+
for lst in lists:
|
|
93
|
+
for item in (lst or []):
|
|
94
|
+
k = _item_key(item)
|
|
95
|
+
if k not in seen:
|
|
96
|
+
seen.add(k)
|
|
97
|
+
out.append(item)
|
|
98
|
+
return out
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def _combine_group(samples: list[Clause | None], fields: tuple[str, ...]) -> dict[str, Any]:
|
|
102
|
+
"""Combine a group's N sampled passes into one field dict: LIST-valued fields are UNIONed across samples
|
|
103
|
+
(list-completeness), scalar/nested fields take the first informative (non-OTHER) sane value. A single sample
|
|
104
|
+
(N=1) reduces to the prior behavior."""
|
|
105
|
+
valid = [r for r in samples if r is not None]
|
|
106
|
+
out: dict[str, Any] = {}
|
|
107
|
+
for f in fields:
|
|
108
|
+
kind, _sub, _hint = _classify(Clause.model_fields[f].annotation)
|
|
109
|
+
vals = [getattr(r, f) for r in valid]
|
|
110
|
+
if kind in ("list_scalar", "nested_list"):
|
|
111
|
+
merged = _union_lists(vals)
|
|
112
|
+
if merged:
|
|
113
|
+
out[f] = merged
|
|
114
|
+
else:
|
|
115
|
+
sane = [_sane(v) for v in vals]
|
|
116
|
+
pick = next((v for v in sane if v is not None and not _uninformative(v)),
|
|
117
|
+
next((v for v in sane if v is not None), None))
|
|
118
|
+
if pick is not None:
|
|
119
|
+
out[f] = pick
|
|
120
|
+
return out
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def _group_has_list(fields: tuple[str, ...]) -> bool:
|
|
124
|
+
"""True if the group has any LIST-valued field (list_scalar / nested_list) -- the fields where under-
|
|
125
|
+
enumeration bites, and the only ones a cross-model union is worth paying a second model for."""
|
|
126
|
+
return any(_classify(Clause.model_fields[f].annotation)[0] in ("list_scalar", "nested_list") for f in fields)
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
async def atag_extract_clause(text: str, model_id: str, *, document_reference: str = "",
|
|
130
|
+
temperature: float = 0.0, samples: int | None = None,
|
|
131
|
+
list_model: str | None = None) -> Clause:
|
|
132
|
+
"""Extract a clause's typed properties as ONE `Clause`, function-independently: EVERY thematic tag-parse pass
|
|
133
|
+
over the `Clause` schema, merged. Each pass degrades on its own (build_tag_structured re-asks then
|
|
134
|
+
omits-to-default); a failing pass leaves its group at defaults (never fails the whole clause). (The former
|
|
135
|
+
aspect gate that pruned groups was removed in issue 0036 -- it measured a ~18% property-recall loss on Qwen.)
|
|
136
|
+
|
|
137
|
+
`samples` (env `RAG_INGEST_CLAUSE_SAMPLES`, default 1) runs each group N times and UNIONs the LIST-valued
|
|
138
|
+
fields across samples -- the inference-time fix for granite's list under-enumeration.
|
|
139
|
+
|
|
140
|
+
`list_model` (env `RAG_INGEST_LIST_MODEL`, DEFAULT gemma) enables a CROSS-MODEL union: for LIST-bearing groups
|
|
141
|
+
ONLY, also run a second (stronger, complementary) model and union its list values with the main model's.
|
|
142
|
+
granite and gemma under-enumerate DIFFERENT items, so their union is more complete than either alone (it fixes
|
|
143
|
+
the CONSISTENT misses same-model multi-sample can't); scoping the second model to list-bearing groups keeps its
|
|
144
|
+
cost off the ~half of groups with no list field. Scalars prefer the main model (its results are unioned first).
|
|
145
|
+
ON by default; disable with `RAG_INGEST_LIST_MODEL=off`."""
|
|
146
|
+
n = samples if samples is not None else max(1, int(os.environ.get("RAG_INGEST_CLAUSE_SAMPLES", "1")))
|
|
147
|
+
# cross-model list model: explicit ARGUMENT wins, else env, else the profile's general model (gemma). Any of
|
|
148
|
+
# them may be "off"/"none"/"" to disable -- so the default-on model is configurable, never a buried hardcode.
|
|
149
|
+
_raw = list_model if list_model is not None else os.environ.get("RAG_INGEST_LIST_MODEL", DEFAULT_GENERAL)
|
|
150
|
+
lm = None if not _raw or str(_raw).strip().lower() in ("none", "off") else str(_raw).strip()
|
|
151
|
+
stemp = temperature if n == 1 else max(temperature, 0.5) # diversity across samples for the union to help
|
|
152
|
+
|
|
153
|
+
async def _pass(fields: tuple[str, ...], model: str) -> Clause | None:
|
|
154
|
+
try:
|
|
155
|
+
return await build_tag_structured(
|
|
156
|
+
model, Clause, fields=set(fields), temperature=stemp, label="clause-group",
|
|
157
|
+
).ainvoke(_GROUP_PROMPT.format(text=text))
|
|
158
|
+
except Exception: # noqa: BLE001 - a persistently-failing pass degrades to defaults, not a hard error
|
|
159
|
+
return None
|
|
160
|
+
|
|
161
|
+
async def _group(fields: tuple[str, ...]) -> dict[str, Any]:
|
|
162
|
+
models = [model_id] # main model first so its scalar values win 'first informative'
|
|
163
|
+
if lm and lm != model_id and _group_has_list(fields):
|
|
164
|
+
models.append(lm) # cross-model union, LIST-bearing groups only (cost-scoped)
|
|
165
|
+
runs = [r for m in models for r in await asyncio.gather(*[_pass(fields, m) for _ in range(n)])]
|
|
166
|
+
return _combine_group(runs, fields)
|
|
167
|
+
|
|
168
|
+
dicts = await asyncio.gather(*[_group(f) for f in CLAUSE_GROUPS.values()])
|
|
169
|
+
merged: dict[str, Any] = {}
|
|
170
|
+
for d in dicts:
|
|
171
|
+
merged.update(d)
|
|
172
|
+
merged["document_reference"] = document_reference or None
|
|
173
|
+
return Clause(**merged)
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
def tag_extract_clause(text: str, model_id: str, *, document_reference: str = "",
|
|
177
|
+
temperature: float = 0.0, samples: int | None = None,
|
|
178
|
+
list_model: str | None = None) -> Clause:
|
|
179
|
+
"""Sync wrapper over `atag_extract_clause` (parity with the docling-graph `extract_clause`)."""
|
|
180
|
+
return asyncio.run(atag_extract_clause(
|
|
181
|
+
text, model_id, document_reference=document_reference, temperature=temperature,
|
|
182
|
+
samples=samples, list_model=list_model))
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
"""The single ArcadeDB store behind the query-skill seam (FR-S.1, FR-S.5).
|
|
2
|
+
|
|
3
|
+
Holds both the hybrid retrieval index and the knowledge graph in one multi-model database.
|
|
4
|
+
Reached only through the query-skill interface so the store implementation is swappable
|
|
5
|
+
(Graphify for a prototype graph, and the eval-gated LanceDB fallback for the retrieval leg).
|
|
6
|
+
"""
|