rag-wright 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (184) hide show
  1. rag_wright/__init__.py +13 -0
  2. rag_wright/api/__init__.py +33 -0
  3. rag_wright/api/config.py +59 -0
  4. rag_wright/api/discover.py +70 -0
  5. rag_wright/api/documents.py +39 -0
  6. rag_wright/api/ids.py +31 -0
  7. rag_wright/api/invoke.py +99 -0
  8. rag_wright/api/kg.py +61 -0
  9. rag_wright/api/mcp.py +94 -0
  10. rag_wright/api/usage.py +30 -0
  11. rag_wright/api/workspace.py +85 -0
  12. rag_wright/capabilities/__init__.py +8 -0
  13. rag_wright/capabilities/answer_generator.py +427 -0
  14. rag_wright/capabilities/ard.py +286 -0
  15. rag_wright/capabilities/assertion_extraction.py +79 -0
  16. rag_wright/capabilities/chunk_read.py +58 -0
  17. rag_wright/capabilities/chunk_write.py +163 -0
  18. rag_wright/capabilities/claim_extraction.py +153 -0
  19. rag_wright/capabilities/clause_exception_linking.py +117 -0
  20. rag_wright/capabilities/compliance_judgment.py +322 -0
  21. rag_wright/capabilities/compliance_store.py +87 -0
  22. rag_wright/capabilities/contract_kg_serve.py +156 -0
  23. rag_wright/capabilities/contract_kg_store.py +251 -0
  24. rag_wright/capabilities/dg_extraction.py +585 -0
  25. rag_wright/capabilities/disambiguation.py +163 -0
  26. rag_wright/capabilities/document_parse.py +87 -0
  27. rag_wright/capabilities/document_scope.py +49 -0
  28. rag_wright/capabilities/embedding.py +164 -0
  29. rag_wright/capabilities/embedding_profiles.py +43 -0
  30. rag_wright/capabilities/entity_resolution.py +154 -0
  31. rag_wright/capabilities/fusion.py +64 -0
  32. rag_wright/capabilities/graph_extraction.py +243 -0
  33. rag_wright/capabilities/graph_query.py +73 -0
  34. rag_wright/capabilities/graph_storage.py +111 -0
  35. rag_wright/capabilities/highlight_serve.py +142 -0
  36. rag_wright/capabilities/hybrid_search.py +65 -0
  37. rag_wright/capabilities/invoke.py +31 -0
  38. rag_wright/capabilities/jev_decision.py +38 -0
  39. rag_wright/capabilities/manifests.py +872 -0
  40. rag_wright/capabilities/okf_navigate.py +456 -0
  41. rag_wright/capabilities/parsing.py +286 -0
  42. rag_wright/capabilities/property_boosted_retrieval.py +125 -0
  43. rag_wright/capabilities/query_function_classifier.py +94 -0
  44. rag_wright/capabilities/query_understanding.py +109 -0
  45. rag_wright/capabilities/registry.py +262 -0
  46. rag_wright/capabilities/remote_encoders.py +94 -0
  47. rag_wright/capabilities/requirement_extraction.py +247 -0
  48. rag_wright/capabilities/reranking.py +123 -0
  49. rag_wright/capabilities/retrieval_core.py +126 -0
  50. rag_wright/capabilities/rlm_chunking.py +808 -0
  51. rag_wright/capabilities/rlm_synthesis.py +316 -0
  52. rag_wright/capabilities/scan_quality.py +136 -0
  53. rag_wright/capabilities/span_relevance_judgment.py +191 -0
  54. rag_wright/capabilities/vision_to_text.py +85 -0
  55. rag_wright/capabilities/vlm_ocr.py +85 -0
  56. rag_wright/contracts/__init__.py +6 -0
  57. rag_wright/contracts/chunk.py +79 -0
  58. rag_wright/contracts/compliance.py +303 -0
  59. rag_wright/contracts/contract_meta.py +27 -0
  60. rag_wright/contracts/extraction.py +130 -0
  61. rag_wright/contracts/function.py +167 -0
  62. rag_wright/contracts/function_routing.py +91 -0
  63. rag_wright/contracts/highlight.py +74 -0
  64. rag_wright/contracts/identifiers.py +153 -0
  65. rag_wright/contracts/jurisdiction.py +96 -0
  66. rag_wright/contracts/ontology.py +142 -0
  67. rag_wright/contracts/property.py +201 -0
  68. rag_wright/contracts/provenance.py +78 -0
  69. rag_wright/contracts/query_intent.py +53 -0
  70. rag_wright/contracts/span.py +76 -0
  71. rag_wright/contracts/value_match.py +84 -0
  72. rag_wright/corpus/__init__.py +0 -0
  73. rag_wright/corpus/canonicalize.py +116 -0
  74. rag_wright/corpus/cuad.py +153 -0
  75. rag_wright/corpus/cuad_ingestion.py +72 -0
  76. rag_wright/corpus/document_parser.py +299 -0
  77. rag_wright/corpus/edgar.py +231 -0
  78. rag_wright/corpus/gcs_ingestion.py +120 -0
  79. rag_wright/corpus/http.py +110 -0
  80. rag_wright/corpus/selection.py +152 -0
  81. rag_wright/mcp/__init__.py +11 -0
  82. rag_wright/mcp/compliance_server.py +299 -0
  83. rag_wright/mcp/intra_document_qa_server.py +170 -0
  84. rag_wright/mcp/relational_qa_server.py +171 -0
  85. rag_wright/mcp/session_store.py +64 -0
  86. rag_wright/mcp/typed_property_retrieval_server.py +191 -0
  87. rag_wright/models/__init__.py +8 -0
  88. rag_wright/models/profiles.py +331 -0
  89. rag_wright/models/seam.py +497 -0
  90. rag_wright/models/tag_structured.py +285 -0
  91. rag_wright/models/tracing.py +179 -0
  92. rag_wright/models/usage.py +102 -0
  93. rag_wright/okf/__init__.py +11 -0
  94. rag_wright/okf/compile.py +292 -0
  95. rag_wright/okf/document.py +47 -0
  96. rag_wright/okf/enrich.py +176 -0
  97. rag_wright/okf/links.py +190 -0
  98. rag_wright/okf/lint.py +105 -0
  99. rag_wright/ontology/__init__.py +6 -0
  100. rag_wright/ontology/_generated_template_meta.py +60 -0
  101. rag_wright/ontology/_generated_vocab.py +52 -0
  102. rag_wright/ontology/clause_template.py +964 -0
  103. rag_wright/ontology/codegen.py +84 -0
  104. rag_wright/ontology/compliance_bridge.ttl +186 -0
  105. rag_wright/ontology/contract_bridge.ttl +2685 -0
  106. rag_wright/ontology/contract_taxonomy.py +24 -0
  107. rag_wright/ontology/derive.py +58 -0
  108. rag_wright/ontology/loader.py +435 -0
  109. rag_wright/ontology/packs/ftc_16cfr255.ttl +29 -0
  110. rag_wright/ontology/registry.py +87 -0
  111. rag_wright/ontology/template_introspect.py +100 -0
  112. rag_wright/py.typed +0 -0
  113. rag_wright/reference/__init__.py +2 -0
  114. rag_wright/reference/compliance.py +41 -0
  115. rag_wright/reference/contract_seam.py +123 -0
  116. rag_wright/skills/__init__.py +7 -0
  117. rag_wright/skills/claim_extraction/SKILL.md +47 -0
  118. rag_wright/skills/claim_extraction/__init__.py +1 -0
  119. rag_wright/skills/claim_extraction/template.py +50 -0
  120. rag_wright/skills/compliance_judgment/SKILL.md +59 -0
  121. rag_wright/skills/corpus_ingest/SKILL.md +106 -0
  122. rag_wright/skills/extraction_semantic_judge/SKILL.md +51 -0
  123. rag_wright/skills/extraction_semantic_judge/__init__.py +1 -0
  124. rag_wright/skills/generation/SKILL.md +64 -0
  125. rag_wright/skills/generation/__init__.py +1 -0
  126. rag_wright/skills/generic_compliance_judgment/SKILL.md +58 -0
  127. rag_wright/skills/okf_navigate/SKILL.md +137 -0
  128. rag_wright/skills/requirement_extraction/SKILL.md +47 -0
  129. rag_wright/skills/requirement_extraction/__init__.py +1 -0
  130. rag_wright/skills/requirement_extraction/template.py +50 -0
  131. rag_wright/skills/rlm/SKILL.md +186 -0
  132. rag_wright/skills/rlm/__init__.py +31 -0
  133. rag_wright/skills/rlm/agent.py +292 -0
  134. rag_wright/skills/span_relevance_judgment/SKILL.md +67 -0
  135. rag_wright/skills/vision_to_text/SKILL.md +36 -0
  136. rag_wright/skills/vision_to_text/__init__.py +1 -0
  137. rag_wright/spans/__init__.py +1 -0
  138. rag_wright/spans/boundary.py +78 -0
  139. rag_wright/spans/clause_function_classifier.py +490 -0
  140. rag_wright/spans/clause_kg_extractor.py +337 -0
  141. rag_wright/spans/cuad_labels.py +81 -0
  142. rag_wright/spans/dim_classifier.py +158 -0
  143. rag_wright/spans/dim_fleet.json +411 -0
  144. rag_wright/spans/function_classifier.py +77 -0
  145. rag_wright/spans/function_families.py +62 -0
  146. rag_wright/spans/hybrid_classifier.py +103 -0
  147. rag_wright/spans/legalbert_classifier.py +83 -0
  148. rag_wright/spans/model_capabilities.py +107 -0
  149. rag_wright/spans/new_function_labels.py +111 -0
  150. rag_wright/spans/page_map.py +68 -0
  151. rag_wright/spans/property_extractor.py +365 -0
  152. rag_wright/spans/property_grounding.py +182 -0
  153. rag_wright/spans/reclassify.py +77 -0
  154. rag_wright/spans/scarce_function_labels.py +105 -0
  155. rag_wright/spans/segment.py +341 -0
  156. rag_wright/spans/semantic_judge.py +197 -0
  157. rag_wright/spans/symbolic_validation.py +131 -0
  158. rag_wright/spans/tag_clause_extractor.py +182 -0
  159. rag_wright/store/__init__.py +6 -0
  160. rag_wright/store/arcadedb.py +1135 -0
  161. rag_wright/store/chunk_text.py +66 -0
  162. rag_wright/store/seam.py +213 -0
  163. rag_wright/subgraphs/__init__.py +0 -0
  164. rag_wright/subgraphs/async_ingestion.py +204 -0
  165. rag_wright/subgraphs/compliance_check.py +1042 -0
  166. rag_wright/subgraphs/compliance_ingestion.py +306 -0
  167. rag_wright/subgraphs/contract_ingestion_pipeline.py +999 -0
  168. rag_wright/subgraphs/graph_extraction.py +102 -0
  169. rag_wright/subgraphs/intra_document_qa.py +328 -0
  170. rag_wright/subgraphs/observability.py +140 -0
  171. rag_wright/subgraphs/query_constraint_extraction.py +73 -0
  172. rag_wright/subgraphs/relational_qa.py +165 -0
  173. rag_wright/subgraphs/requirement_extraction.py +137 -0
  174. rag_wright/subgraphs/scaffold.py +65 -0
  175. rag_wright/subgraphs/semantic_chunking.py +183 -0
  176. rag_wright/subgraphs/typed_clause_extraction.py +172 -0
  177. rag_wright/subgraphs/typed_property_retrieval.py +278 -0
  178. rag_wright/util/__init__.py +1 -0
  179. rag_wright/util/concurrent.py +153 -0
  180. rag_wright/util/spacy_model.py +45 -0
  181. rag_wright-0.1.0.dist-info/METADATA +168 -0
  182. rag_wright-0.1.0.dist-info/RECORD +184 -0
  183. rag_wright-0.1.0.dist-info/WHEEL +4 -0
  184. rag_wright-0.1.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,131 @@
1
+ """JUDGE-ONTOLOGY-1 (ADR-0040): the symbolic ontology-validation gate -- layer 2 of the
2
+ neuro-symbolic extraction-fidelity cascade.
3
+
4
+ Where the lexical grounding judge (`property_grounding.reground`, ADR-0028) checks whether a value's
5
+ surface cue appears in the text, this gate checks whether the ASSERTED DIMENSION is even APPLICABLE to
6
+ the clause's FUNCTION -- a TYPE constraint the lexical judge structurally cannot see. Example: granite
7
+ emits `nonsolicit_target=employees` on an Anti-Assignment clause. The value is a valid nonsolicit_target
8
+ and its cue may appear in the text, so the lexical judge passes it; but `nonsolicit_target` does not
9
+ belong to Anti-Assignment at all -- a type error, caught here.
10
+
11
+ The applicability map (`FUNCTION_APPLICABLE_DIMS`) is the NEW ontology content ADR-0040 calls for: each
12
+ clause FUNCTION permits a set of property DIMENSIONS. It is compiled to SHACL `sh:closed` NodeShapes (one
13
+ per function, listing the applicable dimension paths) and validated with `pyshacl` -- the symbolic half of
14
+ neuro-symbolic. A non-applicable assertion is downgraded to AMBIGUOUS (kept but flagged), exactly like
15
+ `reground`, so the shared property-value nodes stay clean and soft-boost down-weights it.
16
+
17
+ The gate also enforces CARDINALITY (JUDGE-ONTOLOGY-2): a SCALAR dimension asserted with two conflicting
18
+ values (e.g. granite hedging `cap_basis` = both `fixed_fee` and `multiple_of_fees`) violates `sh:maxCount 1`
19
+ and both values are downgraded -- the real intra-clause "contradiction" class, since the dimensions are
20
+ orthogonal facets and same-dimension conflict is where extraction actually contradicts itself. The 3
21
+ multi-valued dimensions (`_LIST_ENUM_DIMS`: carve_out / covered_subject / damage_type) are left unbounded.
22
+
23
+ The gate also enforces DEONTIC consistency (JUDGE-ONTOLOGY-3, ODRL): a consent-regime dimension carries a
24
+ permission↔restriction polarity in its VALUES (`free`/`unrestricted` = "may freely"; `consent_required` =
25
+ restricted). A function whose defining purpose is to RESTRICT (Anti-Assignment / Non-Transferable License /
26
+ Change Of Control) contradicts a permission-polarity value -- the observed `assignment_consent=free` on a
27
+ "shall not assign" clause. The clause's rule type is DERIVED from its function (reliable, non-circular; not
28
+ parsed from the text), and the check is a `sh:in` (allowed = vocab minus the permission-polarity values) on
29
+ the scoped property shape.
30
+
31
+ Deliberately NOT enforced here (JUDGE-ONTOLOGY-2 scoping): value-in-vocabulary (`sh:in` over the full vocab)
32
+ is already enforced at the Pydantic contract boundary (`property.PropertyAssertion._value_in_vocab_or_ambiguous`),
33
+ so a whole-vocab SHACL shape would duplicate a working validator; and cross-DIMENSION `sh:sparql` rules are
34
+ omitted because this schema's dimensions are orthogonal facets with no hard intra-clause cross-dimension
35
+ contradiction (deferred to a post-MVP / beta-customer iteration on real production data).
36
+
37
+ Deterministic, no model, no network. Downgrade is confidence-independent (a type/cardinality/deontic error is
38
+ wrong whether EXTRACTED or INFERRED); an already-AMBIGUOUS assertion is left as is. A function NOT in the map
39
+ is PERMISSIVE (unvalidated) -- coverage is expanded deliberately, never by guessing a closed set we are unsure
40
+ of. All three checks reuse one record->RDF->pyshacl harness.
41
+ """
42
+
43
+ from __future__ import annotations
44
+
45
+ from rdflib import Graph, Literal, Namespace, URIRef
46
+ from rdflib.namespace import RDF, SH
47
+
48
+ from rag_wright.contracts.property import (
49
+ ClausePropertyRecord,
50
+ PropertyAssertion,
51
+ PropertyDimension,
52
+ )
53
+ from rag_wright.contracts.provenance import ConfidenceTag
54
+
55
+ _CBR = Namespace("https://ragwright.local/ontology/contract-bridge#")
56
+
57
+
58
+ def _function_class(function: str) -> URIRef:
59
+ """A stable CBR class IRI for a function label (RDF has no spaces; encode deterministically)."""
60
+ return _CBR[f"Function_{function.replace(' ', '_')}"]
61
+
62
+
63
+ def _dim_property(dimension: PropertyDimension) -> URIRef:
64
+ """The CBR predicate IRI carrying a dimension's asserted value on a clause node."""
65
+ return _CBR[f"dim_{dimension.value}"]
66
+
67
+
68
+ def _shapes_graph() -> Graph:
69
+ """ADR-0066 P2: the SHACL shapes come FROM contract_bridge.ttl (the source of truth) -- pyshacl reads the
70
+ persisted sh:NodeShapes directly; no Python-built shapes. The ttl non-SHACL triples are ignored by pyshacl.
71
+ """
72
+ from rag_wright.ontology.loader import load_shapes_graph
73
+
74
+ return load_shapes_graph()
75
+
76
+
77
+ def _record_to_rdf(record: ClausePropertyRecord) -> Graph:
78
+ """Serialize a record's function + assertions to a tiny data graph: the clause typed by its function,
79
+ with one triple per assertion (dimension predicate -> value literal)."""
80
+ g = Graph()
81
+ clause = URIRef("urn:clause:" + record.clause_id)
82
+ g.add((clause, RDF.type, _function_class(record.function)))
83
+ for a in record.assertions:
84
+ g.add((clause, _dim_property(a.dimension), Literal(a.value)))
85
+ return g
86
+
87
+
88
+ def flagged_dimensions(record: ClausePropertyRecord) -> set[PropertyDimension]:
89
+ """The dimensions to downgrade: ONLY the FUNCTION-INDEPENDENT contradiction check -- a scalar dimension
90
+ asserted with more than one value (`sh:maxCount 1`). The FUNCTION-DEPENDENT checks are DELIBERATELY IGNORED
91
+ (ADR-0082): clause-function classification is not accurate enough to be load-bearing (~0.5 top-1; memory
92
+ `function-classification-not-load-bearing`), so `sh:closed` (dimension-not-applicable-to-function) and `sh:in`
93
+ (deontic polarity on a restrictive function) would downgrade CORRECT cross-cutting extractions based on an
94
+ unreliable (and often narrow) function map. Function is a KG tag / query-time soft signal, never an ingest
95
+ gate. Empty if the record conforms or the function is unmodeled."""
96
+ if not record.assertions:
97
+ return set()
98
+ from pyshacl import validate
99
+
100
+ conforms, results_graph, _ = validate(
101
+ _record_to_rdf(record), shacl_graph=_shapes_graph(), inference="none", advanced=False
102
+ )
103
+ if conforms:
104
+ return set()
105
+ flagged: set[PropertyDimension] = set()
106
+ _prefix = str(_CBR) + "dim_"
107
+ for result in results_graph.subjects(RDF.type, SH.ValidationResult):
108
+ # keep ONLY the contradiction (maxCount) violations; drop function-dependent closed/in violations
109
+ if results_graph.value(result, SH.sourceConstraintComponent) != SH.MaxCountConstraintComponent:
110
+ continue
111
+ path = results_graph.value(result, SH.resultPath)
112
+ if path is not None and str(path).startswith(_prefix):
113
+ flagged.add(PropertyDimension(str(path)[len(_prefix):]))
114
+ return flagged
115
+
116
+
117
+ def symbolic_validate(record: ClausePropertyRecord) -> ClausePropertyRecord:
118
+ """Quality gate (ADR-0040 layer 2): downgrade every assertion whose dimension violates a shape -- not
119
+ applicable to the clause's function, or a scalar dimension asserted with conflicting values -- to
120
+ AMBIGUOUS (kept but flagged, model-agnostic, confidence-independent). A no-op when the function is
121
+ unmodeled or every dimension is valid -- mirrors `property_grounding.reground`."""
122
+ bad = flagged_dimensions(record)
123
+ if not bad:
124
+ return record
125
+ new: list[PropertyAssertion] = [
126
+ a.model_copy(update={"confidence": ConfidenceTag.AMBIGUOUS})
127
+ if a.dimension in bad and a.confidence != ConfidenceTag.AMBIGUOUS
128
+ else a
129
+ for a in record.assertions
130
+ ]
131
+ return record.model_copy(update={"assertions": new})
@@ -0,0 +1,182 @@
1
+ """TAGPARSE-INGEST-1b: FUNCTION-INDEPENDENT clause property extraction via client-side tag-parse.
2
+
3
+ The docling-graph path asks a model to fill the whole 35-field `Clause` template in one server-side-JSON call --
4
+ which granite-4.2 flattens/fails (~5/7) on real clauses (ADR-0079). Function-SCOPED extraction was ruled out
5
+ because clause-function classification is not accurate enough to gate on (~0.5 top-1; memory
6
+ `function-classification-not-load-bearing`). So we decompose the schema WITHOUT the function: the 35 fields are
7
+ split into 7 cohesive THEMATIC groups, each a small tag-parse pass over the SAME `Clause` schema (via
8
+ `build_tag_structured(..., fields=group)`), run concurrently and merged into one `Clause`. Small focused schemas
9
+ are where tag-parse already wins (parties 5/5), and `Clause`'s own validators normalize on the final construct.
10
+
11
+ `CLAUSE_GROUPS` is domain knowledge (which properties belong together); it is a candidate for ontology migration
12
+ (ADR-0066) but lives here for now, like `property_extractor.FUNCTION_DIMENSIONS`. Every non-id `Clause` field is
13
+ in exactly one group (a coverage test enforces this).
14
+ """
15
+
16
+ from __future__ import annotations
17
+
18
+ import asyncio
19
+ import os
20
+ from typing import Any
21
+
22
+ from rag_wright.models.profiles import DEFAULT_GENERAL
23
+ from rag_wright.models.tag_structured import _classify, build_tag_structured
24
+ from rag_wright.ontology.clause_template import Clause
25
+
26
+ # The 7 thematic groups (5a consents/control + 5b restrictions/duties per the design). document_reference is the
27
+ # clause's root id, filled separately (from context / source stem), never asked of the model here.
28
+ CLAUSE_GROUPS: dict[str, tuple[str, ...]] = {
29
+ "identity_scope": ("clause_type", "covers", "covers_party_scope", "has_mutuality",
30
+ "has_asymmetry", "has_favorability"),
31
+ "liability_damages": ("caps", "has_claim_scope", "prohibits_damage", "ld_trigger"),
32
+ "temporal_termination": ("bounded_by", "has_renewal", "has_termination_right", "condition_type"),
33
+ "ip_licensing": ("has_ip_ownership", "has_exclusivity_type", "has_right_of_first_type",
34
+ "has_mfn_scope", "royalty_basis"),
35
+ "consents_control": ("has_assignment_consent", "has_coc_consent", "has_escrow_release_trigger"),
36
+ "governing_law_dispute": ("governed_by", "dispute_method"), # split out of the grab-bag: a focused pass so
37
+ # jurisdiction isn't buried among 8 other fields
38
+ "restrictions_duties": ("requires_duty", "has_restriction_scope", "prohibits_solicit", "has_warranty_scope",
39
+ "audit_frequency", "commitment_quantum", "collateral_type"),
40
+ "exceptions": ("excepts", "confidentiality_exception", "force_majeure_event"),
41
+ }
42
+
43
+ _GROUP_PROMPT = (
44
+ "You are extracting the typed properties of ONE contract clause. Extract ONLY the properties below that are "
45
+ "EXPLICITLY stated in the clause; omit any that are not present -- do NOT guess or invent a value. Read each "
46
+ "value verbatim from the clause.\n\nCLAUSE:\n{text}"
47
+ )
48
+
49
+ # NOTE (issue 0036): TWO cost-cutting ideas were tried here and BOTH measured a ~15-18% property-recall loss on
50
+ # Qwen (the current default), concentrated in `excepts` (carve-outs) -- so neither was adopted. (1) A coarse
51
+ # "aspect gate" that pruned groups per clause (removed entirely). (2) Batching two clauses per group call (recall
52
+ # 0.818 at samples=1, 0.844 at samples=4 -- still a real regression, not variance). Splitting the model's attention
53
+ # across clauses, or pruning groups, both under-extract the hard cross-cutting fields. So every group always runs,
54
+ # one clause per call; the only recall-preserving cost lever is `is_extractable_span` (fewer spans reach here).
55
+
56
+
57
+ import re as _re
58
+
59
+ _MAX_VALUE_CHARS = 240 # a clause PROPERTY value is short ("12_months", "Delaware"); longer = leaked prose
60
+ _TAG_LIKE = _re.compile(r"<[A-Za-z_][\w-]*>") # a value must never contain XML tags (model reasoning/prompt leak)
61
+
62
+
63
+ def _sane(value: Any) -> Any:
64
+ """Drop a garbage string value (a leaked chain-of-thought / prompt echo): too long, or containing XML tags.
65
+ Non-string values pass through (enums/sub-models/lists can't carry this leak). A dropped value -> None so the
66
+ field falls back to its default (never store reasoning text as a clause property)."""
67
+ if isinstance(value, str) and (len(value) > _MAX_VALUE_CHARS or _TAG_LIKE.search(value)):
68
+ return None
69
+ return value
70
+
71
+
72
+ def _uninformative(v: Any) -> bool:
73
+ """A scalar value carrying no information: None/empty, or an enum OTHER escape (dropped downstream anyway)."""
74
+ if v is None:
75
+ return True
76
+ s = str(getattr(v, "value", v)).strip()
77
+ return s == "" or s in ("Other", "OTHER") or "Unknown" in s
78
+
79
+
80
+ def _item_key(item: Any) -> Any:
81
+ """A hashable identity for a list item so the union can dedup: an enum by its value, a sub-model by its dump."""
82
+ if hasattr(item, "model_dump"):
83
+ return tuple(sorted((k, str(v)) for k, v in item.model_dump().items()))
84
+ return getattr(item, "value", item)
85
+
86
+
87
+ def _union_lists(lists: list[Any]) -> list[Any]:
88
+ """Order-preserving UNION of a list-valued field across samples -- the fix for granite's list under-enumeration
89
+ (each sample may emit a different subset; the union recovers the full set)."""
90
+ out: list[Any] = []
91
+ seen: set[Any] = set()
92
+ for lst in lists:
93
+ for item in (lst or []):
94
+ k = _item_key(item)
95
+ if k not in seen:
96
+ seen.add(k)
97
+ out.append(item)
98
+ return out
99
+
100
+
101
+ def _combine_group(samples: list[Clause | None], fields: tuple[str, ...]) -> dict[str, Any]:
102
+ """Combine a group's N sampled passes into one field dict: LIST-valued fields are UNIONed across samples
103
+ (list-completeness), scalar/nested fields take the first informative (non-OTHER) sane value. A single sample
104
+ (N=1) reduces to the prior behavior."""
105
+ valid = [r for r in samples if r is not None]
106
+ out: dict[str, Any] = {}
107
+ for f in fields:
108
+ kind, _sub, _hint = _classify(Clause.model_fields[f].annotation)
109
+ vals = [getattr(r, f) for r in valid]
110
+ if kind in ("list_scalar", "nested_list"):
111
+ merged = _union_lists(vals)
112
+ if merged:
113
+ out[f] = merged
114
+ else:
115
+ sane = [_sane(v) for v in vals]
116
+ pick = next((v for v in sane if v is not None and not _uninformative(v)),
117
+ next((v for v in sane if v is not None), None))
118
+ if pick is not None:
119
+ out[f] = pick
120
+ return out
121
+
122
+
123
+ def _group_has_list(fields: tuple[str, ...]) -> bool:
124
+ """True if the group has any LIST-valued field (list_scalar / nested_list) -- the fields where under-
125
+ enumeration bites, and the only ones a cross-model union is worth paying a second model for."""
126
+ return any(_classify(Clause.model_fields[f].annotation)[0] in ("list_scalar", "nested_list") for f in fields)
127
+
128
+
129
+ async def atag_extract_clause(text: str, model_id: str, *, document_reference: str = "",
130
+ temperature: float = 0.0, samples: int | None = None,
131
+ list_model: str | None = None) -> Clause:
132
+ """Extract a clause's typed properties as ONE `Clause`, function-independently: EVERY thematic tag-parse pass
133
+ over the `Clause` schema, merged. Each pass degrades on its own (build_tag_structured re-asks then
134
+ omits-to-default); a failing pass leaves its group at defaults (never fails the whole clause). (The former
135
+ aspect gate that pruned groups was removed in issue 0036 -- it measured a ~18% property-recall loss on Qwen.)
136
+
137
+ `samples` (env `RAG_INGEST_CLAUSE_SAMPLES`, default 1) runs each group N times and UNIONs the LIST-valued
138
+ fields across samples -- the inference-time fix for granite's list under-enumeration.
139
+
140
+ `list_model` (env `RAG_INGEST_LIST_MODEL`, DEFAULT gemma) enables a CROSS-MODEL union: for LIST-bearing groups
141
+ ONLY, also run a second (stronger, complementary) model and union its list values with the main model's.
142
+ granite and gemma under-enumerate DIFFERENT items, so their union is more complete than either alone (it fixes
143
+ the CONSISTENT misses same-model multi-sample can't); scoping the second model to list-bearing groups keeps its
144
+ cost off the ~half of groups with no list field. Scalars prefer the main model (its results are unioned first).
145
+ ON by default; disable with `RAG_INGEST_LIST_MODEL=off`."""
146
+ n = samples if samples is not None else max(1, int(os.environ.get("RAG_INGEST_CLAUSE_SAMPLES", "1")))
147
+ # cross-model list model: explicit ARGUMENT wins, else env, else the profile's general model (gemma). Any of
148
+ # them may be "off"/"none"/"" to disable -- so the default-on model is configurable, never a buried hardcode.
149
+ _raw = list_model if list_model is not None else os.environ.get("RAG_INGEST_LIST_MODEL", DEFAULT_GENERAL)
150
+ lm = None if not _raw or str(_raw).strip().lower() in ("none", "off") else str(_raw).strip()
151
+ stemp = temperature if n == 1 else max(temperature, 0.5) # diversity across samples for the union to help
152
+
153
+ async def _pass(fields: tuple[str, ...], model: str) -> Clause | None:
154
+ try:
155
+ return await build_tag_structured(
156
+ model, Clause, fields=set(fields), temperature=stemp, label="clause-group",
157
+ ).ainvoke(_GROUP_PROMPT.format(text=text))
158
+ except Exception: # noqa: BLE001 - a persistently-failing pass degrades to defaults, not a hard error
159
+ return None
160
+
161
+ async def _group(fields: tuple[str, ...]) -> dict[str, Any]:
162
+ models = [model_id] # main model first so its scalar values win 'first informative'
163
+ if lm and lm != model_id and _group_has_list(fields):
164
+ models.append(lm) # cross-model union, LIST-bearing groups only (cost-scoped)
165
+ runs = [r for m in models for r in await asyncio.gather(*[_pass(fields, m) for _ in range(n)])]
166
+ return _combine_group(runs, fields)
167
+
168
+ dicts = await asyncio.gather(*[_group(f) for f in CLAUSE_GROUPS.values()])
169
+ merged: dict[str, Any] = {}
170
+ for d in dicts:
171
+ merged.update(d)
172
+ merged["document_reference"] = document_reference or None
173
+ return Clause(**merged)
174
+
175
+
176
+ def tag_extract_clause(text: str, model_id: str, *, document_reference: str = "",
177
+ temperature: float = 0.0, samples: int | None = None,
178
+ list_model: str | None = None) -> Clause:
179
+ """Sync wrapper over `atag_extract_clause` (parity with the docling-graph `extract_clause`)."""
180
+ return asyncio.run(atag_extract_clause(
181
+ text, model_id, document_reference=document_reference, temperature=temperature,
182
+ samples=samples, list_model=list_model))
@@ -0,0 +1,6 @@
1
+ """The single ArcadeDB store behind the query-skill seam (FR-S.1, FR-S.5).
2
+
3
+ Holds both the hybrid retrieval index and the knowledge graph in one multi-model database.
4
+ Reached only through the query-skill interface so the store implementation is swappable
5
+ (Graphify for a prototype graph, and the eval-gated LanceDB fallback for the retrieval leg).
6
+ """