rag-wright 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (184) hide show
  1. rag_wright/__init__.py +13 -0
  2. rag_wright/api/__init__.py +33 -0
  3. rag_wright/api/config.py +59 -0
  4. rag_wright/api/discover.py +70 -0
  5. rag_wright/api/documents.py +39 -0
  6. rag_wright/api/ids.py +31 -0
  7. rag_wright/api/invoke.py +99 -0
  8. rag_wright/api/kg.py +61 -0
  9. rag_wright/api/mcp.py +94 -0
  10. rag_wright/api/usage.py +30 -0
  11. rag_wright/api/workspace.py +85 -0
  12. rag_wright/capabilities/__init__.py +8 -0
  13. rag_wright/capabilities/answer_generator.py +427 -0
  14. rag_wright/capabilities/ard.py +286 -0
  15. rag_wright/capabilities/assertion_extraction.py +79 -0
  16. rag_wright/capabilities/chunk_read.py +58 -0
  17. rag_wright/capabilities/chunk_write.py +163 -0
  18. rag_wright/capabilities/claim_extraction.py +153 -0
  19. rag_wright/capabilities/clause_exception_linking.py +117 -0
  20. rag_wright/capabilities/compliance_judgment.py +322 -0
  21. rag_wright/capabilities/compliance_store.py +87 -0
  22. rag_wright/capabilities/contract_kg_serve.py +156 -0
  23. rag_wright/capabilities/contract_kg_store.py +251 -0
  24. rag_wright/capabilities/dg_extraction.py +585 -0
  25. rag_wright/capabilities/disambiguation.py +163 -0
  26. rag_wright/capabilities/document_parse.py +87 -0
  27. rag_wright/capabilities/document_scope.py +49 -0
  28. rag_wright/capabilities/embedding.py +164 -0
  29. rag_wright/capabilities/embedding_profiles.py +43 -0
  30. rag_wright/capabilities/entity_resolution.py +154 -0
  31. rag_wright/capabilities/fusion.py +64 -0
  32. rag_wright/capabilities/graph_extraction.py +243 -0
  33. rag_wright/capabilities/graph_query.py +73 -0
  34. rag_wright/capabilities/graph_storage.py +111 -0
  35. rag_wright/capabilities/highlight_serve.py +142 -0
  36. rag_wright/capabilities/hybrid_search.py +65 -0
  37. rag_wright/capabilities/invoke.py +31 -0
  38. rag_wright/capabilities/jev_decision.py +38 -0
  39. rag_wright/capabilities/manifests.py +872 -0
  40. rag_wright/capabilities/okf_navigate.py +456 -0
  41. rag_wright/capabilities/parsing.py +286 -0
  42. rag_wright/capabilities/property_boosted_retrieval.py +125 -0
  43. rag_wright/capabilities/query_function_classifier.py +94 -0
  44. rag_wright/capabilities/query_understanding.py +109 -0
  45. rag_wright/capabilities/registry.py +262 -0
  46. rag_wright/capabilities/remote_encoders.py +94 -0
  47. rag_wright/capabilities/requirement_extraction.py +247 -0
  48. rag_wright/capabilities/reranking.py +123 -0
  49. rag_wright/capabilities/retrieval_core.py +126 -0
  50. rag_wright/capabilities/rlm_chunking.py +808 -0
  51. rag_wright/capabilities/rlm_synthesis.py +316 -0
  52. rag_wright/capabilities/scan_quality.py +136 -0
  53. rag_wright/capabilities/span_relevance_judgment.py +191 -0
  54. rag_wright/capabilities/vision_to_text.py +85 -0
  55. rag_wright/capabilities/vlm_ocr.py +85 -0
  56. rag_wright/contracts/__init__.py +6 -0
  57. rag_wright/contracts/chunk.py +79 -0
  58. rag_wright/contracts/compliance.py +303 -0
  59. rag_wright/contracts/contract_meta.py +27 -0
  60. rag_wright/contracts/extraction.py +130 -0
  61. rag_wright/contracts/function.py +167 -0
  62. rag_wright/contracts/function_routing.py +91 -0
  63. rag_wright/contracts/highlight.py +74 -0
  64. rag_wright/contracts/identifiers.py +153 -0
  65. rag_wright/contracts/jurisdiction.py +96 -0
  66. rag_wright/contracts/ontology.py +142 -0
  67. rag_wright/contracts/property.py +201 -0
  68. rag_wright/contracts/provenance.py +78 -0
  69. rag_wright/contracts/query_intent.py +53 -0
  70. rag_wright/contracts/span.py +76 -0
  71. rag_wright/contracts/value_match.py +84 -0
  72. rag_wright/corpus/__init__.py +0 -0
  73. rag_wright/corpus/canonicalize.py +116 -0
  74. rag_wright/corpus/cuad.py +153 -0
  75. rag_wright/corpus/cuad_ingestion.py +72 -0
  76. rag_wright/corpus/document_parser.py +299 -0
  77. rag_wright/corpus/edgar.py +231 -0
  78. rag_wright/corpus/gcs_ingestion.py +120 -0
  79. rag_wright/corpus/http.py +110 -0
  80. rag_wright/corpus/selection.py +152 -0
  81. rag_wright/mcp/__init__.py +11 -0
  82. rag_wright/mcp/compliance_server.py +299 -0
  83. rag_wright/mcp/intra_document_qa_server.py +170 -0
  84. rag_wright/mcp/relational_qa_server.py +171 -0
  85. rag_wright/mcp/session_store.py +64 -0
  86. rag_wright/mcp/typed_property_retrieval_server.py +191 -0
  87. rag_wright/models/__init__.py +8 -0
  88. rag_wright/models/profiles.py +331 -0
  89. rag_wright/models/seam.py +497 -0
  90. rag_wright/models/tag_structured.py +285 -0
  91. rag_wright/models/tracing.py +179 -0
  92. rag_wright/models/usage.py +102 -0
  93. rag_wright/okf/__init__.py +11 -0
  94. rag_wright/okf/compile.py +292 -0
  95. rag_wright/okf/document.py +47 -0
  96. rag_wright/okf/enrich.py +176 -0
  97. rag_wright/okf/links.py +190 -0
  98. rag_wright/okf/lint.py +105 -0
  99. rag_wright/ontology/__init__.py +6 -0
  100. rag_wright/ontology/_generated_template_meta.py +60 -0
  101. rag_wright/ontology/_generated_vocab.py +52 -0
  102. rag_wright/ontology/clause_template.py +964 -0
  103. rag_wright/ontology/codegen.py +84 -0
  104. rag_wright/ontology/compliance_bridge.ttl +186 -0
  105. rag_wright/ontology/contract_bridge.ttl +2685 -0
  106. rag_wright/ontology/contract_taxonomy.py +24 -0
  107. rag_wright/ontology/derive.py +58 -0
  108. rag_wright/ontology/loader.py +435 -0
  109. rag_wright/ontology/packs/ftc_16cfr255.ttl +29 -0
  110. rag_wright/ontology/registry.py +87 -0
  111. rag_wright/ontology/template_introspect.py +100 -0
  112. rag_wright/py.typed +0 -0
  113. rag_wright/reference/__init__.py +2 -0
  114. rag_wright/reference/compliance.py +41 -0
  115. rag_wright/reference/contract_seam.py +123 -0
  116. rag_wright/skills/__init__.py +7 -0
  117. rag_wright/skills/claim_extraction/SKILL.md +47 -0
  118. rag_wright/skills/claim_extraction/__init__.py +1 -0
  119. rag_wright/skills/claim_extraction/template.py +50 -0
  120. rag_wright/skills/compliance_judgment/SKILL.md +59 -0
  121. rag_wright/skills/corpus_ingest/SKILL.md +106 -0
  122. rag_wright/skills/extraction_semantic_judge/SKILL.md +51 -0
  123. rag_wright/skills/extraction_semantic_judge/__init__.py +1 -0
  124. rag_wright/skills/generation/SKILL.md +64 -0
  125. rag_wright/skills/generation/__init__.py +1 -0
  126. rag_wright/skills/generic_compliance_judgment/SKILL.md +58 -0
  127. rag_wright/skills/okf_navigate/SKILL.md +137 -0
  128. rag_wright/skills/requirement_extraction/SKILL.md +47 -0
  129. rag_wright/skills/requirement_extraction/__init__.py +1 -0
  130. rag_wright/skills/requirement_extraction/template.py +50 -0
  131. rag_wright/skills/rlm/SKILL.md +186 -0
  132. rag_wright/skills/rlm/__init__.py +31 -0
  133. rag_wright/skills/rlm/agent.py +292 -0
  134. rag_wright/skills/span_relevance_judgment/SKILL.md +67 -0
  135. rag_wright/skills/vision_to_text/SKILL.md +36 -0
  136. rag_wright/skills/vision_to_text/__init__.py +1 -0
  137. rag_wright/spans/__init__.py +1 -0
  138. rag_wright/spans/boundary.py +78 -0
  139. rag_wright/spans/clause_function_classifier.py +490 -0
  140. rag_wright/spans/clause_kg_extractor.py +337 -0
  141. rag_wright/spans/cuad_labels.py +81 -0
  142. rag_wright/spans/dim_classifier.py +158 -0
  143. rag_wright/spans/dim_fleet.json +411 -0
  144. rag_wright/spans/function_classifier.py +77 -0
  145. rag_wright/spans/function_families.py +62 -0
  146. rag_wright/spans/hybrid_classifier.py +103 -0
  147. rag_wright/spans/legalbert_classifier.py +83 -0
  148. rag_wright/spans/model_capabilities.py +107 -0
  149. rag_wright/spans/new_function_labels.py +111 -0
  150. rag_wright/spans/page_map.py +68 -0
  151. rag_wright/spans/property_extractor.py +365 -0
  152. rag_wright/spans/property_grounding.py +182 -0
  153. rag_wright/spans/reclassify.py +77 -0
  154. rag_wright/spans/scarce_function_labels.py +105 -0
  155. rag_wright/spans/segment.py +341 -0
  156. rag_wright/spans/semantic_judge.py +197 -0
  157. rag_wright/spans/symbolic_validation.py +131 -0
  158. rag_wright/spans/tag_clause_extractor.py +182 -0
  159. rag_wright/store/__init__.py +6 -0
  160. rag_wright/store/arcadedb.py +1135 -0
  161. rag_wright/store/chunk_text.py +66 -0
  162. rag_wright/store/seam.py +213 -0
  163. rag_wright/subgraphs/__init__.py +0 -0
  164. rag_wright/subgraphs/async_ingestion.py +204 -0
  165. rag_wright/subgraphs/compliance_check.py +1042 -0
  166. rag_wright/subgraphs/compliance_ingestion.py +306 -0
  167. rag_wright/subgraphs/contract_ingestion_pipeline.py +999 -0
  168. rag_wright/subgraphs/graph_extraction.py +102 -0
  169. rag_wright/subgraphs/intra_document_qa.py +328 -0
  170. rag_wright/subgraphs/observability.py +140 -0
  171. rag_wright/subgraphs/query_constraint_extraction.py +73 -0
  172. rag_wright/subgraphs/relational_qa.py +165 -0
  173. rag_wright/subgraphs/requirement_extraction.py +137 -0
  174. rag_wright/subgraphs/scaffold.py +65 -0
  175. rag_wright/subgraphs/semantic_chunking.py +183 -0
  176. rag_wright/subgraphs/typed_clause_extraction.py +172 -0
  177. rag_wright/subgraphs/typed_property_retrieval.py +278 -0
  178. rag_wright/util/__init__.py +1 -0
  179. rag_wright/util/concurrent.py +153 -0
  180. rag_wright/util/spacy_model.py +45 -0
  181. rag_wright-0.1.0.dist-info/METADATA +168 -0
  182. rag_wright-0.1.0.dist-info/RECORD +184 -0
  183. rag_wright-0.1.0.dist-info/WHEEL +4 -0
  184. rag_wright-0.1.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,1135 @@
1
+ """The ArcadeDB implementation of the store seam (T13, FR-S.1, FR-S.5).
2
+
3
+ One multi-model database holds both the hybrid retrieval index (the `Chunk` vertex type, carrying
4
+ the dense summary vector and the sparse full-text vector) and the knowledge graph (the `Entity`
5
+ vertex type), so a chunk and its extracted entities share one store and one `chunk_id` with no
6
+ cross-store join (FR-S.1). The hybrid index has two legs: a dense `LSM_VECTOR` (HNSW) index over the
7
+ summary vector and a sparse `LSM_SPARSE_VECTOR` index over the full-text vector.
8
+
9
+ Grounded against `arcadedb_python` 0.4.0 (`SyncClient`, `DatabaseDao`) and the live ArcadeDB
10
+ 26.7.2 server: the sparse index requires the sparse vector stored as two parallel arrays,
11
+ `sparse_indices` (ARRAY_OF_INTEGERS) and `sparse_weights` (ARRAY_OF_FLOATS), not a single map, so
12
+ T3's `sparse_vector: dict[int, float]` is decomposed at the store boundary (T20 writes it). ArcadeDB
13
+ rejects `CREATE ... IF NOT EXISTS` in this dialect position, so idempotency is by schema
14
+ introspection: create only what is absent.
15
+ """
16
+
17
+ from __future__ import annotations
18
+
19
+ import json
20
+ import os
21
+ from typing import Any, Iterable, Optional
22
+
23
+ from arcadedb_python import DatabaseDao, SyncClient
24
+
25
+ from rag_wright.contracts.chunk import BGE_M3_DENSE_DIM, ChunkRecord, MetadataValue
26
+ from rag_wright.contracts.provenance import ConfidenceTag
27
+ from rag_wright.corpus.canonicalize import normalize_entity_name # issue 0030: name -> entity clustering key
28
+ from rag_wright.ontology.loader import ( # ADR-0067: KG schema from the ontology
29
+ load_kg_schema, # P5b: domain vertex/edge types
30
+ load_typed_edges, # P5a: typed-edge map
31
+ )
32
+ from rag_wright.contracts.span import SpanRecord
33
+ from rag_wright.store.seam import NOT_NULL, GraphEdge, GraphNode
34
+
35
+ CHUNK_TYPE = "Chunk"
36
+ ENTITY_TYPE = "Entity"
37
+ REL_EDGE_TYPE = "Relationship" # entity -> entity relationship edge (the graph's primary content)
38
+ MENTIONS_EDGE_TYPE = "Mentions" # chunk -> entity provenance edge (FR-S.1: chunk and entities connect)
39
+
40
+ # FR-R (ADR-0025/0026) property graph: clause node -> typed property edge -> shared property-value node.
41
+ # Distinct from the generic Entity graph. Value nodes are deduped by (dimension,value); the controlled
42
+ # vocabulary is already canonical, so no entity-resolution clustering is needed.
43
+ CLAUSE_TYPE = "Clause"
44
+ PROPVALUE_TYPE = "PropertyValue"
45
+ PROPERTY_EDGE_TYPE = "HasProperty" # legacy flat edge (ADR-0025/0026); superseded by the KG-3 typed edges
46
+
47
+ # KG-3 (ADR-0033): the TYPED property-edge layer that replaces the single generic HasProperty edge (KG-0
48
+ # gate Q3: build typed, retire the flat edge). Each PropertyDimension maps to its sanctioned typed edge; the
49
+ # shared PropertyValue node (deduped by value_key) is UNCHANGED -- identity preserved, so the upgrade is
50
+ # additive on nodes and rebuilt on edges. Every typed edge still carries the assertion provenance (FR-S.4)
51
+ # plus a predicate IRI (ODRL for the deontic edges, our bridge IRI otherwise).
52
+ # ADR-0067 P5a: the typed-edge map (dimension -> KG edge type) + the predicate IRIs are AUTHORITATIVE in
53
+ # contract_bridge.ttl (cbr:kgEdge / cbr:KgEdgeType); loaded here, not a Python literal. Edit the ttl to retarget.
54
+ _DIM_EDGE_STR, _EDGE_PREDICATE_IRI = load_typed_edges()
55
+ # distinct edge types (deterministic order; a set / for counts + DDL, never order-dependent). The dim->edge map
56
+ # stays str-keyed: only the clause-KG writer indexed it by `PropertyDimension`, and that moved to the
57
+ # `capabilities/contract_kg_store.py` extension (DD-1b), so the engine store needs no `PropertyDimension`.
58
+ TYPED_PROPERTY_EDGE_TYPES: tuple[str, ...] = tuple(sorted(set(_DIM_EDGE_STR.values())))
59
+
60
+
61
+ def _edge_predicate_iri(edge_type: str) -> str:
62
+ """The predicate IRI stamped on a typed edge (ODRL for the deontic edges, the bridge IRI otherwise) --
63
+ from contract_bridge.ttl (ADR-0067 P5a)."""
64
+ return _EDGE_PREDICATE_IRI[edge_type]
65
+
66
+ # Candidates fetched per leg before fusion. RRF reorders within this pool, so it is set well above a
67
+ # typical final `k` to give fusion (and any metadata filter) room to work; the fused list is then
68
+ # cut to `k`. Tuned at GATE-2 against the golden set if recall calls for it.
69
+ DEFAULT_CANDIDATE_POOL = 100
70
+ SCOPED_CANDIDATE_POOL = 1000 # issue 0031: a larger KNN pool when a `documents` scope filters AFTER the vector
71
+ # legs, so a small workspace does not under-fill k (the vector functions do not pre-filter; ADR-0008)
72
+
73
+ SPAN_TYPE = "Span" # FR-R (ADR-0025): the operative-span hybrid index; dense+sparse over the span text
74
+ CONTRACT_TYPE = "Contract" # CU-B3 (ADR-0029): contract-level metadata (the CUAD document lookup unit)
75
+ # (issue 0028 / ADR-0091: the `PartyTo` edge was retired -- written on every ingest, read by nothing; party->clause
76
+ # is reached via CONTRACTS_WITH provenance + the contract-scoped clause KG.)
77
+ IS_EXCEPTION_TO_EDGE_TYPE = "IsExceptionTo" # ADR-0044: exception clause (Uncapped) -> the Cap clause it excepts
78
+ REQUIREMENT_TYPE = "Requirement" # CC-5 (compliance §13): a deontic regulatory rule (its own DB, ragwright_compliance)
79
+
80
+ # Expected index names follow ArcadeDB's `Type[prop]` / `Type[p1,p2]` convention.
81
+ _DENSE_INDEX = f"{CHUNK_TYPE}[dense]"
82
+ _SPARSE_INDEX = f"{CHUNK_TYPE}[sparse_indices,sparse_weights]"
83
+ _CHUNK_ID_INDEX = f"{CHUNK_TYPE}[chunk_id]"
84
+ _ENTITY_ID_INDEX = f"{ENTITY_TYPE}[entity_id]"
85
+ _SPAN_ID_INDEX = f"{SPAN_TYPE}[span_id]"
86
+ _SPAN_DENSE_INDEX = f"{SPAN_TYPE}[dense]"
87
+ _SPAN_SPARSE_INDEX = f"{SPAN_TYPE}[sparse_indices,sparse_weights]"
88
+ # ADR-0067 P5b: the domain vertex UNIQUE id indexes (Clause/PropertyValue/Contract) are pack-declared
89
+ # (cbr:uniqueIndexOn) and built by the generic ensure_schema loop, not hardcoded here.
90
+
91
+
92
+ def _property_value_key(dimension: str, value: str) -> str:
93
+ """The shared `PropertyValue` node identity: the canonical (dimension, value). The controlled
94
+ vocabulary is already canonical, so dedup across clauses is a deterministic upsert by this key."""
95
+ return f"{dimension}:{value}"
96
+
97
+
98
+ def _sql_str(value: str) -> str:
99
+ """A single-quoted ArcadeDB SQL string literal (backslash, quote, and control whitespace escaped).
100
+
101
+ ArcadeDB's SQL tokenizer rejects a raw newline / carriage-return / tab inside a string literal (a
102
+ "token recognition error at ..."), so those are backslash-escaped alongside the quote and backslash.
103
+ Backslash is escaped first so the escapes added afterward each carry a single backslash. Discovered
104
+ ingesting ACORD's multi-paragraph clauses (T33): without this, every clause containing a newline
105
+ silently dead-lettered.
106
+ """
107
+ return (
108
+ "'"
109
+ + value.replace("\\", "\\\\")
110
+ .replace("'", "\\'")
111
+ .replace("\n", "\\n")
112
+ .replace("\r", "\\r")
113
+ .replace("\t", "\\t")
114
+ + "'"
115
+ )
116
+
117
+
118
+ def _float_array(values: Iterable[float]) -> str:
119
+ return "[" + ",".join(repr(float(x)) for x in values) + "]"
120
+
121
+
122
+ def _str_array(values: Iterable[str]) -> str:
123
+ return "[" + ",".join(_sql_str(v) for v in values) + "]"
124
+
125
+
126
+ def _kg_sql_value(value: object) -> str:
127
+ """Serialize a scalar for `kg_read` (DD-1a): bool/int/float native, everything else a quoted string."""
128
+ if isinstance(value, bool):
129
+ return "true" if value else "false"
130
+ if isinstance(value, int):
131
+ return str(value)
132
+ if isinstance(value, float):
133
+ return repr(value)
134
+ return _sql_str(str(value))
135
+
136
+
137
+ def _kg_sql_array(values: list) -> str:
138
+ return "[" + ",".join(_kg_sql_value(v) for v in values) + "]"
139
+
140
+
141
+ def _kg_sql(value: object) -> str:
142
+ """Type-driven serialization for `kg_write` EDGE props + undeclared fields: None->null, scalars native/quoted,
143
+ list/tuple -> a nested array literal."""
144
+ if value is None:
145
+ return "null"
146
+ if isinstance(value, bool):
147
+ return "true" if value else "false"
148
+ if isinstance(value, int):
149
+ return str(value)
150
+ if isinstance(value, float):
151
+ return repr(value)
152
+ if isinstance(value, (list, tuple)):
153
+ return "[" + ",".join(_kg_sql(v) for v in value) + "]"
154
+ return _sql_str(str(value))
155
+
156
+
157
+ def _kg_encode(value: object, declared_type: Optional[str]) -> str:
158
+ """Encode a `kg_write` NODE prop by its PACK-DECLARED storage type (DD-1b): the declared type is what
159
+ disambiguates a list stored as a native array (`ARRAY_OF_*`) from one stored as a JSON string (`STRING`) --
160
+ e.g. `pages` (array) vs `bbox` (JSON string). None->null; an undeclared field falls back to type-driven."""
161
+ if value is None:
162
+ return "null"
163
+ dt = (declared_type or "").upper()
164
+ if dt == "STRING":
165
+ return _sql_str(value if isinstance(value, str) else json.dumps(value))
166
+ if dt in ("INTEGER", "LONG", "SHORT", "BYTE"):
167
+ return str(int(value))
168
+ if dt in ("FLOAT", "DOUBLE", "DECIMAL"):
169
+ return repr(float(value))
170
+ if dt == "BOOLEAN":
171
+ return "true" if value else "false"
172
+ if dt == "ARRAY_OF_INTEGERS":
173
+ return "[" + ",".join(str(int(x)) for x in value) + "]"
174
+ if dt == "ARRAY_OF_FLOATS":
175
+ return _float_array(value)
176
+ if dt == "ARRAY_OF_STRINGS":
177
+ return _str_array(value)
178
+ return _kg_sql(value)
179
+
180
+
181
+ # DD-1b: the non-pack KG vertex property storage types, centralized so `ensure_compliance_schema` and `kg_write`'s
182
+ # encoder read ONE source (the contract-pack vertices -- Clause/PropertyValue/Contract -- come from `load_kg_schema`).
183
+ _ENGINE_VERTEX_PROPERTY_TYPES: dict[str, dict[str, str]] = {
184
+ REQUIREMENT_TYPE: {
185
+ "requirement_id": "STRING", "source": "STRING", "citation": "STRING", "deontic_type": "STRING",
186
+ "actor": "STRING", "requirement_text": "STRING", "evidence_standard": "STRING", "severity": "STRING",
187
+ "applicability_json": "STRING", "confidence": "STRING", "pages": "ARRAY_OF_INTEGERS", "bbox": "STRING"},
188
+ }
189
+
190
+
191
+ def _doc_id_of(chunk_id: str) -> str:
192
+ """The source-document id embedded in a chunk/span/clause id (issue 0031). The id scheme is
193
+ `<source_doc_id>:<index>:<hash>` and `source_doc_id` is delimiter-safe (no ':', enforced by `ChunkId`),
194
+ so the document id is exactly the prefix before the first ':'. Empty in -> empty out."""
195
+ return (chunk_id or "").split(":", 1)[0]
196
+
197
+
198
+ def _edge_provenance_assignments(chunk_id: str) -> str:
199
+ """The `SET` fragment writing an edge's provenance `chunk_id` and its DERIVED `source_doc_id` as ONE
200
+ matched pair from a SINGLE source (issue 0031 ask #2). `source_doc_id` is a denormalization of `chunk_id`
201
+ for index-backed scoping; keeping the derivation in this one place is what makes the two impossible to
202
+ drift. A drifted pair (a `source_doc_id` disagreeing with its `chunk_id`) is an INVISIBLE cross-matter
203
+ leak -- the scoping filter would silently admit an out-of-scope edge -- so every edge writer MUST emit the
204
+ pair through here, never assign the two fields independently."""
205
+ return f"chunk_id = {_sql_str(chunk_id)}, source_doc_id = {_sql_str(_doc_id_of(chunk_id))}"
206
+
207
+
208
+ def _sql_literal(value: MetadataValue) -> str:
209
+ """A SQL literal for a filterable metadata scalar (bool checked before int: `bool` subclasses `int`)."""
210
+ if isinstance(value, bool):
211
+ return "true" if value else "false"
212
+ if isinstance(value, (int, float)):
213
+ return repr(value)
214
+ return _sql_str(str(value))
215
+
216
+
217
+ def _stale_property_statements(span_ids: list[str]) -> list[str]:
218
+ """ADR-0048 Phase A mark-stale (pure): one `UPDATE ... SET confidence=AMBIGUOUS WHERE span_id IN (...) AND
219
+ confidence <> AMBIGUOUS` per typed property edge type, for the given spans. Separated from the DB call so the
220
+ SQL is unit-tested with no store. Empty span list -> no statements (the caller no-ops)."""
221
+ if not span_ids:
222
+ return []
223
+ id_list = "[" + ",".join(_sql_str(s) for s in span_ids) + "]"
224
+ amb = _sql_str(ConfidenceTag.AMBIGUOUS.value)
225
+ return [
226
+ f"UPDATE {edge_type} SET confidence = {amb} WHERE span_id IN {id_list} AND confidence <> {amb}"
227
+ for edge_type in TYPED_PROPERTY_EDGE_TYPES
228
+ ]
229
+
230
+
231
+ class ArcadeDBStore:
232
+ """The default store: schema management over a single ArcadeDB database."""
233
+
234
+ def __init__(self, client: SyncClient, database: str, *, pack_ttl: str | None = None) -> None:
235
+ self._client = client
236
+ self._database = database
237
+ self._db = DatabaseDao(client, database)
238
+ # AC-journey: the DOMAIN pack `.ttl` whose KG vertex/edge types `ensure_schema` creates and whose property
239
+ # storage types `kg_write` encodes by. `None` = the engine's reference CONTRACT pack (today's behavior); a
240
+ # new domain points this at its OWN pack, so "config + .ttl" creates that domain's schema with no engine edit.
241
+ self._pack_ttl = pack_ttl
242
+
243
+ @classmethod
244
+ def from_env(cls, *, database: str | None = None, reset: bool = False,
245
+ pack_ttl: str | None = None) -> "ArcadeDBStore":
246
+ """Build a store from `ARCADEDB_*` env, creating the database if absent.
247
+
248
+ `database` overrides `ARCADEDB_DATABASE` (used to point tests at a scratch database).
249
+ `reset=True` drops and recreates the database first, for a clean-slate test.
250
+ `pack_ttl` selects the domain pack schema (None = the contract reference pack).
251
+ """
252
+ return cls.from_config(
253
+ os.environ["ARCADEDB_HOST"], os.environ["ARCADEDB_PORT"],
254
+ os.environ["ARCADEDB_USER"], os.environ["ARCADEDB_PASSWORD"],
255
+ database=database or os.environ["ARCADEDB_DATABASE"],
256
+ protocol=os.getenv("ARCADEDB_PROTOCOL", "http"), # `https` for the Modal-hosted KG (EC-2)
257
+ reset=reset, pack_ttl=pack_ttl)
258
+
259
+ @classmethod
260
+ def from_config(cls, host: str, port: str, user: str, password: str, *, database: str,
261
+ protocol: str = "http", reset: bool = False,
262
+ pack_ttl: str | None = None) -> "ArcadeDBStore":
263
+ """Build a store from EXPLICIT connection params (EP-API-1: the de-env'd twin of `from_env`, so engine
264
+ config flows as data, not `os.environ`). Creates the database if absent; `reset=True` drops + recreates it.
265
+ `pack_ttl` selects the domain pack schema (None = the contract reference pack)."""
266
+ client = SyncClient(host, port, protocol=protocol, username=user, password=password)
267
+ if reset and DatabaseDao.exists(client, database):
268
+ DatabaseDao.delete(client, database)
269
+ if not DatabaseDao.exists(client, database):
270
+ DatabaseDao.create(client, database)
271
+ return cls(client, database, pack_ttl=pack_ttl)
272
+
273
+ # --- seam surface ---------------------------------------------------------------------------
274
+
275
+ def ensure_schema(self) -> None:
276
+ """Create the chunk-record and graph-node types and the hybrid indexes, idempotently."""
277
+ types = self.type_names()
278
+ if CHUNK_TYPE not in types:
279
+ self._command(f"CREATE VERTEX TYPE {CHUNK_TYPE}")
280
+ self._command(f"CREATE PROPERTY {CHUNK_TYPE}.chunk_id STRING")
281
+ self._command(f"CREATE PROPERTY {CHUNK_TYPE}.source_doc_id STRING")
282
+ self._command(f"CREATE PROPERTY {CHUNK_TYPE}.dense ARRAY_OF_FLOATS")
283
+ self._command(f"CREATE PROPERTY {CHUNK_TYPE}.sparse_indices ARRAY_OF_INTEGERS")
284
+ self._command(f"CREATE PROPERTY {CHUNK_TYPE}.sparse_weights ARRAY_OF_FLOATS")
285
+ if ENTITY_TYPE not in types:
286
+ self._command(f"CREATE VERTEX TYPE {ENTITY_TYPE}")
287
+ self._command(f"CREATE PROPERTY {ENTITY_TYPE}.entity_id STRING") # node key (canonical id or surrogate)
288
+ self._command(f"CREATE PROPERTY {ENTITY_TYPE}.chunk_id STRING")
289
+ self._command(f"CREATE PROPERTY {ENTITY_TYPE}.canonical_id STRING") # ADR-0067 P5c: the resolver's canonical id, or '' if unlinked
290
+ self._command(f"CREATE PROPERTY {ENTITY_TYPE}.name STRING")
291
+ self._command(f"CREATE PROPERTY {ENTITY_TYPE}.entity_type STRING")
292
+ self._command(f"CREATE PROPERTY {ENTITY_TYPE}.confidence STRING")
293
+ if REL_EDGE_TYPE not in types:
294
+ self._command(f"CREATE EDGE TYPE {REL_EDGE_TYPE}")
295
+ # issue 0031: the source-document id (derived from the edge's provenance chunk_id) so a graph
296
+ # traversal can be scoped to a workspace's documents (`WHERE source_doc_id IN [...]`).
297
+ self._command(f"CREATE PROPERTY {REL_EDGE_TYPE}.source_doc_id STRING")
298
+ if MENTIONS_EDGE_TYPE not in types:
299
+ self._command(f"CREATE EDGE TYPE {MENTIONS_EDGE_TYPE}")
300
+ if SPAN_TYPE not in types: # FR-R (ADR-0025): operative-span hybrid index
301
+ self._command(f"CREATE VERTEX TYPE {SPAN_TYPE}")
302
+ self._command(f"CREATE PROPERTY {SPAN_TYPE}.span_id STRING")
303
+ self._command(f"CREATE PROPERTY {SPAN_TYPE}.parent_chunk_id STRING")
304
+ self._command(f"CREATE PROPERTY {SPAN_TYPE}.parent_okf_path STRING")
305
+ self._command(f"CREATE PROPERTY {SPAN_TYPE}.span_index INTEGER")
306
+ self._command(f"CREATE PROPERTY {SPAN_TYPE}.text STRING")
307
+ self._command(f"CREATE PROPERTY {SPAN_TYPE}.function STRING") # the PRIMARY function-classifier tag (T56)
308
+ self._command(f"CREATE PROPERTY {SPAN_TYPE}.functions STRING") # T55/ADR-0114: top-k soft tags, JSON list
309
+ self._command(f"CREATE PROPERTY {SPAN_TYPE}.dense ARRAY_OF_FLOATS")
310
+ self._command(f"CREATE PROPERTY {SPAN_TYPE}.sparse_indices ARRAY_OF_INTEGERS")
311
+ self._command(f"CREATE PROPERTY {SPAN_TYPE}.sparse_weights ARRAY_OF_FLOATS")
312
+ self._command(f"CREATE PROPERTY {SPAN_TYPE}.contract_id STRING") # CU-B2: within-contract filter
313
+ self._command(f"CREATE PROPERTY {SPAN_TYPE}.doc_start INTEGER") # CU-B2: doc-absolute char offset
314
+ self._command(f"CREATE PROPERTY {SPAN_TYPE}.doc_end INTEGER") # CU-B2: exclusive (citation)
315
+ self._command(f"CREATE PROPERTY {SPAN_TYPE}.pages ARRAY_OF_INTEGERS") # issue 0032: source page(s)
316
+ self._command(f"CREATE PROPERTY {SPAN_TYPE}.bbox STRING") # issue 0032: best-effort [l,t,r,b] JSON
317
+ # ADR-0067 P5b: the DOMAIN vertex types (Clause / PropertyValue / Contract) + structural edges
318
+ # (HasProperty / IsExceptionTo) are declared in the pack ttl (load_kg_schema); the engine creates
319
+ # whatever the pack declares, so a new domain ships its own node schema without editing this method.
320
+ vertex_types, structural_edges = load_kg_schema(getattr(self, "_pack_ttl", None))
321
+ for vt in vertex_types:
322
+ if vt.name not in types:
323
+ self._command(f"CREATE VERTEX TYPE {vt.name}")
324
+ for pname, ptype in vt.properties:
325
+ self._command(f"CREATE PROPERTY {vt.name}.{pname} {ptype}")
326
+ # edge types: the structural edges (pack) + the typed property edges (P5a, ttl-driven)
327
+ for edge in sorted(structural_edges) + list(TYPED_PROPERTY_EDGE_TYPES):
328
+ if edge not in types:
329
+ self._command(f"CREATE EDGE TYPE {edge}")
330
+
331
+ indexes = self.index_names()
332
+ if _CHUNK_ID_INDEX not in indexes:
333
+ self._command(f"CREATE INDEX ON {CHUNK_TYPE} (chunk_id) UNIQUE")
334
+ if _DENSE_INDEX not in indexes:
335
+ self._command(
336
+ f"CREATE INDEX ON {CHUNK_TYPE} (dense) LSM_VECTOR "
337
+ f"METADATA {{ dimensions: {BGE_M3_DENSE_DIM}, similarity: 'COSINE' }}"
338
+ )
339
+ if _SPARSE_INDEX not in indexes:
340
+ self._command(
341
+ f"CREATE INDEX ON {CHUNK_TYPE} (sparse_indices, sparse_weights) LSM_SPARSE_VECTOR"
342
+ )
343
+ if _ENTITY_ID_INDEX not in indexes:
344
+ self._command(f"CREATE INDEX ON {ENTITY_TYPE} (entity_id) UNIQUE")
345
+ if _SPAN_ID_INDEX not in indexes:
346
+ self._command(f"CREATE INDEX ON {SPAN_TYPE} (span_id) UNIQUE")
347
+ if _SPAN_DENSE_INDEX not in indexes:
348
+ self._command(
349
+ f"CREATE INDEX ON {SPAN_TYPE} (dense) LSM_VECTOR "
350
+ f"METADATA {{ dimensions: {BGE_M3_DENSE_DIM}, similarity: 'COSINE' }}"
351
+ )
352
+ if _SPAN_SPARSE_INDEX not in indexes:
353
+ self._command(
354
+ f"CREATE INDEX ON {SPAN_TYPE} (sparse_indices, sparse_weights) LSM_SPARSE_VECTOR"
355
+ )
356
+ # ADR-0067 P5b: the domain vertex UNIQUE id indexes (pack-declared, `cbr:uniqueIndexOn`)
357
+ for vt in vertex_types:
358
+ if vt.unique_index and f"{vt.name}[{vt.unique_index}]" not in indexes:
359
+ self._command(f"CREATE INDEX ON {vt.name} ({vt.unique_index}) UNIQUE")
360
+
361
+ def type_names(self) -> set[str]:
362
+ return {row["name"] for row in self._query("SELECT name FROM schema:types")}
363
+
364
+ def property_names(self, type_name: str) -> set[str]:
365
+ for row in self._query("SELECT name, properties FROM schema:types"):
366
+ if row.get("name") == type_name:
367
+ return {prop["name"] for prop in row.get("properties", [])}
368
+ return set()
369
+
370
+ def index_names(self) -> set[str]:
371
+ return {row["name"] for row in self._query("SELECT name FROM schema:indexes")}
372
+
373
+ def ping(self) -> bool:
374
+ return DatabaseDao.exists(self._client, self._database)
375
+
376
+ def close(self) -> None:
377
+ # The HTTP client holds no persistent connection to release.
378
+ pass
379
+
380
+ # --- write-side (T20) -----------------------------------------------------------------------
381
+
382
+ def upsert_chunk(self, record: ChunkRecord) -> None:
383
+ """Upsert a chunk record by `chunk_id`. The sparse vector is decomposed into the two parallel
384
+ arrays the `LSM_SPARSE_VECTOR` index binds (ADR-0007); the dense vector's length is enforced
385
+ by the `LSM_VECTOR` index."""
386
+ chunk_id = record.chunk_id.value
387
+ token_ids = sorted(record.sparse_vector) # deterministic order across the paired arrays
388
+ dense = _float_array(record.dense_vector)
389
+ sparse_indices = "[" + ",".join(str(i) for i in token_ids) + "]"
390
+ sparse_weights = _float_array(record.sparse_vector[i] for i in token_ids)
391
+ self._command(
392
+ f"UPDATE {CHUNK_TYPE} SET"
393
+ f" chunk_id = {_sql_str(chunk_id)},"
394
+ f" source_doc_id = {_sql_str(record.chunk_id.source_doc_id)},"
395
+ f" summary = {_sql_str(record.summary)},"
396
+ f" dense = {dense},"
397
+ f" sparse_indices = {sparse_indices},"
398
+ f" sparse_weights = {sparse_weights},"
399
+ f" keywords = {_str_array(record.keywords)},"
400
+ f" entity_mentions = {_str_array(record.entity_mentions)}"
401
+ f" UPSERT WHERE chunk_id = {_sql_str(chunk_id)}"
402
+ )
403
+
404
+ def get_chunk(self, chunk_id: str) -> Any:
405
+ rows = self._query(
406
+ f"SELECT chunk_id, source_doc_id, summary FROM {CHUNK_TYPE} "
407
+ f"WHERE chunk_id = {_sql_str(chunk_id)}"
408
+ )
409
+ return rows[0] if rows else None
410
+
411
+ # --- generic typed-node read (DD-1a, ADR-0117): the backend-agnostic primitive the domain store
412
+ # extensions delegate to, so a domain pack never writes ArcadeDB SQL -----------------------------
413
+
414
+ def kg_read(
415
+ self,
416
+ node_type: str,
417
+ *,
418
+ where: Optional[dict[str, object]] = None,
419
+ fields: Optional[list[str]] = None,
420
+ distinct: Optional[str] = None,
421
+ order_by: Optional[str] = None,
422
+ limit: Optional[int] = None,
423
+ ) -> list[dict]:
424
+ """Read typed nodes of `node_type` (see `Store.kg_read`). A list `where` value that is empty is
425
+ scope-to-nothing -> `[]` without a query (never an invalid `IN []`). Clauses AND-ed in insertion order."""
426
+ clauses: list[str] = []
427
+ for field, value in (where or {}).items():
428
+ if isinstance(value, (list, tuple, set)):
429
+ vals = list(value)
430
+ if not vals:
431
+ return []
432
+ clauses.append(f"{field} IN {_kg_sql_array(vals)}")
433
+ else:
434
+ clauses.append(f"{field} = {_kg_sql_value(value)}")
435
+ proj = f"DISTINCT({distinct}) AS {distinct}" if distinct else (", ".join(fields) if fields else "*")
436
+ sql = f"SELECT {proj} FROM {node_type}"
437
+ if clauses:
438
+ sql += " WHERE " + " AND ".join(clauses)
439
+ if order_by:
440
+ sql += f" ORDER BY {order_by}"
441
+ if limit is not None:
442
+ sql += f" LIMIT {int(limit)}"
443
+ return self._query(sql)
444
+
445
+ def kg_edges(
446
+ self,
447
+ from_type: Optional[str] = None,
448
+ *,
449
+ where: Optional[dict[str, object]] = None,
450
+ key_range: Optional[tuple[str, object, object]] = None,
451
+ direction: str = "out",
452
+ edge_type: Optional[str] = None,
453
+ edge_where: Optional[dict[str, object]] = None,
454
+ target_where: Optional[dict[str, object]] = None,
455
+ select: dict[str, str],
456
+ ) -> list[dict]:
457
+ """Generic edge traversal (see `Store.kg_edges`): node-start MATCH (out/in) when a start selector is
458
+ given, else a direct edge-table scan. An empty membership anywhere scopes to nothing -> `[]`."""
459
+
460
+ def _terms(d: Optional[dict[str, object]]) -> Optional[list[str]]:
461
+ out: list[str] = []
462
+ for field, value in (d or {}).items():
463
+ if value is NOT_NULL:
464
+ out.append(f"{field} IS NOT NULL")
465
+ elif isinstance(value, (list, tuple, set)):
466
+ vals = list(value)
467
+ if not vals:
468
+ return None # empty membership -> scope-to-nothing
469
+ out.append(f"{field} IN {_kg_sql_array(vals)}")
470
+ else:
471
+ out.append(f"{field} = {_kg_sql_value(value)}")
472
+ return out
473
+
474
+ c_terms, e_terms, v_terms = _terms(where), _terms(edge_where), _terms(target_where)
475
+ if c_terms is None or e_terms is None or v_terms is None:
476
+ return []
477
+ if key_range is not None:
478
+ f, lo, hi = key_range
479
+ c_terms = c_terms + [f"{f} >= {_kg_sql_value(lo)}", f"{f} < {_kg_sql_value(hi)}"]
480
+ returns = ", ".join(f"{expr} AS {alias}" for alias, expr in select.items())
481
+
482
+ if not (where or key_range is not None): # --- edge-scan idiom ---
483
+ if not edge_type:
484
+ raise ValueError("kg_edges edge-scan needs an edge_type (no start selector given)")
485
+ sql = f"SELECT {returns} FROM {edge_type}"
486
+ if e_terms:
487
+ sql += " WHERE " + " AND ".join(e_terms)
488
+ return self._query(sql)
489
+
490
+ # --- node-start MATCH traversal ---
491
+ if from_type is None:
492
+ raise ValueError("kg_edges traversal needs a from_type")
493
+ etok = f"'{edge_type}'" if edge_type else ""
494
+ if direction == "out":
495
+ edge_step, far_step = f"outE({etok})", "inV()"
496
+ elif direction == "in":
497
+ edge_step, far_step = f"inE({etok})", "outV()"
498
+ else:
499
+ raise ValueError(f"kg_edges direction must be 'out' or 'in', got {direction!r}")
500
+
501
+ def _blk(body: str, term_list: list[str]) -> str:
502
+ return "{" + body + (f", where: ({' AND '.join(term_list)})" if term_list else "") + "}"
503
+
504
+ c_blk = _blk(f"type: {from_type}, as: c", c_terms)
505
+ e_blk = _blk("as: e", e_terms)
506
+ v_blk = _blk("as: v", v_terms)
507
+ return self._query(f"MATCH {c_blk}.{edge_step}{e_blk}.{far_step}{v_blk} RETURN {returns}")
508
+
509
+ def _property_types(self, type_name: str) -> dict[str, str]:
510
+ """`{property -> declared storage type}` for a KG node type, used by `kg_write` to encode each prop. Sourced
511
+ from the pack schema (`load_kg_schema`) + the centralized non-pack declarations. Cached per store."""
512
+ cache = getattr(self, "_prop_types_cache", None)
513
+ if cache is None:
514
+ cache = {name: dict(props) for name, props in _ENGINE_VERTEX_PROPERTY_TYPES.items()}
515
+ vertex_types, _ = load_kg_schema(getattr(self, "_pack_ttl", None))
516
+ for vt in vertex_types:
517
+ cache[vt.name] = dict(vt.properties)
518
+ self._prop_types_cache = cache
519
+ return cache.get(type_name, {})
520
+
521
+ def kg_write(self, nodes, edges=()) -> None:
522
+ """Upsert typed `nodes` (by `key_field`) then create typed `edges`, all in one transaction (see
523
+ `Store.kg_write`). Node props encode by the type's pack-declared storage type; edge props are type-driven."""
524
+ statements: list[str] = []
525
+ for n in nodes:
526
+ types = self._property_types(n.type)
527
+ sets = ", ".join(f"{k} = {_kg_encode(v, types.get(k))}" for k, v in n.props.items())
528
+ key_sql = _kg_encode(n.props[n.key_field], types.get(n.key_field))
529
+ statements.append(f"UPDATE {n.type} SET {sets} UPSERT WHERE {n.key_field} = {key_sql}")
530
+ for e in edges:
531
+ stmt = (
532
+ f"CREATE EDGE {e.type}"
533
+ f" FROM (SELECT FROM {e.from_type} WHERE {e.from_key_field} = {_kg_sql(e.from_key)})"
534
+ f" TO (SELECT FROM {e.to_type} WHERE {e.to_key_field} = {_kg_sql(e.to_key)})")
535
+ if e.props:
536
+ stmt += " SET " + ", ".join(f"{k} = {_kg_sql(v)}" for k, v in e.props.items())
537
+ statements.append(stmt)
538
+ if statements:
539
+ self._db.execute_transaction(statements)
540
+
541
+ # --- operative-span write/search (FR-R, ADR-0025) -------------------------------------------
542
+
543
+ def upsert_span(self, record: SpanRecord) -> None:
544
+ """Upsert an operative-span record by `span_id` (dense + sparse over the span text). Same sparse
545
+ decomposition into the two parallel arrays the `LSM_SPARSE_VECTOR` index binds as `upsert_chunk`."""
546
+ token_ids = sorted(record.sparse_vector) # deterministic order across the paired arrays
547
+ dense = _float_array(record.dense_vector)
548
+ sparse_indices = "[" + ",".join(str(i) for i in token_ids) + "]"
549
+ sparse_weights = _float_array(record.sparse_vector[i] for i in token_ids)
550
+ doc_start = "null" if record.doc_start is None else int(record.doc_start)
551
+ doc_end = "null" if record.doc_end is None else int(record.doc_end)
552
+ pages = "[" + ",".join(str(int(p)) for p in record.pages) + "]" # issue 0032: source page(s)
553
+ functions_json = _sql_str(json.dumps(list(record.functions))) # T55/ADR-0114: top-k soft tags (primary-first)
554
+ bbox = "null" if record.bbox is None else _sql_str(json.dumps(list(record.bbox))) # best-effort [l,t,r,b]
555
+ self._command(
556
+ f"UPDATE {SPAN_TYPE} SET"
557
+ f" span_id = {_sql_str(record.span_id)},"
558
+ f" parent_chunk_id = {_sql_str(record.parent_chunk_id)},"
559
+ f" parent_okf_path = {_sql_str(record.parent_okf_path)},"
560
+ f" span_index = {int(record.span_index)},"
561
+ f" text = {_sql_str(record.text)},"
562
+ f" function = {_sql_str(record.function)},"
563
+ f" dense = {dense},"
564
+ f" sparse_indices = {sparse_indices},"
565
+ f" sparse_weights = {sparse_weights},"
566
+ f" contract_id = {_sql_str(record.contract_id)}," # CU-B2: citation + within-contract filter
567
+ f" doc_start = {doc_start},"
568
+ f" doc_end = {doc_end},"
569
+ f" pages = {pages}," # issue 0032 (CU-B5): source page(s) for the citation highlight
570
+ f" functions = {functions_json}," # T55/ADR-0114: top-k soft tags (JSON list, primary-first)
571
+ f" bbox = {bbox}"
572
+ f" UPSERT WHERE span_id = {_sql_str(record.span_id)}"
573
+ )
574
+
575
+ def contract_by_id(self, contract_id: str) -> dict | None:
576
+ """CU-B3: look up a contract's metadata by id (the row, or None if absent)."""
577
+ rows = self.kg_read(CONTRACT_TYPE, fields=[
578
+ "contract_id", "name", "agreement_type", "parties_json", "agreement_date", "effective_date",
579
+ "source_doc_id", "content_hash", "page_count"], where={"contract_id": contract_id})
580
+ return rows[0] if rows else None
581
+
582
+ # --- Compliance module (CC-5, §13): the Requirement KG, in its OWN database (ragwright_compliance) ---
583
+
584
+ def ensure_compliance_schema(self) -> None:
585
+ """Create the compliance schema: the `Requirement` vertex type + a UNIQUE index on `requirement_id`.
586
+ Additive + idempotent (create only what is absent, by introspection). Intended for a SEPARATE database
587
+ (`ragwright_compliance`) so the contract KG stays clean; touches no existing type or identifier."""
588
+ if REQUIREMENT_TYPE in self.type_names():
589
+ return
590
+ self._command(f"CREATE VERTEX TYPE {REQUIREMENT_TYPE}")
591
+ for prop, ptype in _ENGINE_VERTEX_PROPERTY_TYPES[REQUIREMENT_TYPE].items(): # DD-1b: one source of types
592
+ self._command(f"CREATE PROPERTY {REQUIREMENT_TYPE}.{prop} {ptype}") # incl. pages (array) + bbox (JSON str)
593
+ self._command(f"CREATE INDEX ON {REQUIREMENT_TYPE} (requirement_id) UNIQUE")
594
+
595
+ def all_requirements(self, sources: Optional[Iterable[str]] = None) -> list[dict]:
596
+ """Stored `Requirement` rows (CC-6 loads these to match a claim's scope against applicability).
597
+
598
+ `sources=None` returns every row (store-wide, unchanged). Issue 0007: when a list of policy `source`s is
599
+ given, the filter is pushed into the QUERY (`WHERE source IN [...]`) so a store holding thousands of rows
600
+ across many policies/tenants never fetches the ones outside the scope -- scale-ready, not an in-memory
601
+ filter. An empty scope (`sources=[]`) returns `[]` without a query (scope-to-nothing; also avoids an
602
+ invalid `IN []`)."""
603
+ if sources is not None:
604
+ sources = list(sources)
605
+ if not sources:
606
+ return [] # empty scope -> [] without a query
607
+ return self.kg_read(REQUIREMENT_TYPE, fields=[
608
+ "requirement_id", "source", "citation", "deontic_type", "actor", "requirement_text",
609
+ "evidence_standard", "severity", "applicability_json", "confidence", "pages", "bbox"],
610
+ where=({"source": sources} if sources is not None else None))
611
+
612
+ def requirement_sources(self) -> set[str]:
613
+ """Issue 0007: the DISTINCT set of policy `source`s present in the Requirement KG -- powers unknown-source
614
+ validation (naming a policy that does not exist) WITHOUT loading any requirement rows. The Requirement type
615
+ may not exist yet on a fresh DB -> empty set."""
616
+ if REQUIREMENT_TYPE not in self.type_names():
617
+ return set()
618
+ rows = self._query(f"SELECT DISTINCT(source) AS s FROM {REQUIREMENT_TYPE}")
619
+ return {r["s"] for r in rows if r.get("s")}
620
+
621
+ def ingested_citations(self, source: str) -> set[str]:
622
+ """COMP-ASYNC-1 resume (PROD-2 #2): the set of `citation`s that ALREADY have >=1 `Requirement` for `source`
623
+ -- the compliance analogue of a present `Contract` node. A section in this set was successfully ingested
624
+ (a FAILED or genuinely-empty section wrote 0 requirements, so it is absent and correctly re-runs). The
625
+ Requirement type may not exist yet on a fresh DB -> empty set."""
626
+ if REQUIREMENT_TYPE not in self.type_names():
627
+ return set()
628
+ rows = self._query(
629
+ f"SELECT DISTINCT(citation) AS c FROM {REQUIREMENT_TYPE} WHERE source = {_sql_str(source)}")
630
+ return {r["c"] for r in rows if r.get("c")}
631
+
632
+ def spans_by_contract(self, contract_id: str, functions: list[str]) -> list[dict]:
633
+ """CU-B3: the within-contract typed filter -- every span in `contract_id` whose `function` is in
634
+ `functions`, ordered by document position (the CUAD serve retrieval; empty `functions` -> []).
635
+ Returns citation-ready rows (span_id, parent pointer, text, function, doc offsets)."""
636
+ if not functions:
637
+ return []
638
+ return self.kg_read(SPAN_TYPE, fields=[
639
+ "span_id", "parent_chunk_id", "parent_okf_path", "span_index", "text", "function",
640
+ "contract_id", "doc_start", "doc_end", "pages", "bbox"],
641
+ where={"contract_id": contract_id, "function": functions}, order_by="doc_start")
642
+
643
+ def all_spans_by_contract(self, contract_id: str) -> list[dict]:
644
+ """CU-C2: EVERY span in a contract (all functions incl NONE), with its dense vector, ordered by
645
+ document position. For the out-of-taxonomy semantic fallback: the contract is small (hundreds of
646
+ spans), so ranking happens in Python -- a global ANN + contract filter would miss, since one
647
+ contract is ~1% of the corpus. Returns citation-ready rows plus `dense`."""
648
+ return self._query(
649
+ f"SELECT span_id, parent_chunk_id, span_index, text, function, contract_id,"
650
+ f" doc_start, doc_end, pages, bbox, dense FROM {SPAN_TYPE}" # issue 0032: page citation
651
+ f" WHERE contract_id = {_sql_str(contract_id)} ORDER BY doc_start"
652
+ )
653
+
654
+ def span_hybrid_search(
655
+ self,
656
+ dense_query: list[float],
657
+ sparse_query: dict[int, float],
658
+ *,
659
+ k: int,
660
+ function: str | None = None,
661
+ documents: list[str] | None = None,
662
+ ) -> list[dict]:
663
+ """RRF-fused dense+sparse search over the `Span` index, optionally restricted to one `function` tag
664
+ (the function-classifier's routing filter, FR-R) and/or a `documents` set (issue 0031: a workspace
665
+ scope -- `Span.contract_id IN [...]`, the source-document id). Mirrors `hybrid_search`; returns span_id
666
+ + the parent pointer so the caller can follow the span back to its clause for the rerank stage.
667
+
668
+ Issue 0031: when `documents` is given, the vector legs pull a LARGER pool (`SCOPED_CANDIDATE_POOL`)
669
+ before the `contract_id IN [...]` cut, because the KNN ranks across the whole index and only then is
670
+ scoped -- a small workspace could otherwise under-fill `k` from the default pool. `documents=[]` is a
671
+ valid scope-to-nothing -> no query (`[]`)."""
672
+ if documents is not None and not documents:
673
+ return [] # scope-to-nothing: never issue an invalid `IN []`
674
+ leg_k = max(k, SCOPED_CANDIDATE_POOL if documents else DEFAULT_CANDIDATE_POOL)
675
+ token_ids = sorted(sparse_query)
676
+ sparse_indices = "[" + ",".join(str(i) for i in token_ids) + "]"
677
+ sparse_weights = _float_array(sparse_query[i] for i in token_ids)
678
+ dense = _float_array(dense_query)
679
+ fused = (
680
+ "SELECT expand(`vector.fuse`("
681
+ f"`vector.neighbors`('{_SPAN_DENSE_INDEX}', {dense}, {leg_k}), "
682
+ f"`vector.sparseNeighbors`('{_SPAN_SPARSE_INDEX}', {sparse_indices}, {sparse_weights}, {leg_k}), "
683
+ "{ fusion: 'RRF' }))"
684
+ )
685
+ clauses = []
686
+ if function:
687
+ clauses.append(f"function = {_sql_str(function)}")
688
+ if documents:
689
+ clauses.append(f"contract_id IN {_str_array(documents)}")
690
+ where = f" WHERE {' AND '.join(clauses)}" if clauses else ""
691
+ return self._query(
692
+ f"SELECT span_id, parent_chunk_id, parent_okf_path, function FROM ({fused}){where} LIMIT {k}"
693
+ )
694
+
695
+ def span_dense_search(
696
+ self,
697
+ dense_query: list[float],
698
+ *,
699
+ k: int,
700
+ documents: list[str] | None = None,
701
+ ) -> list[dict]:
702
+ """Issue 0041 (dense floor): PURE-DENSE nearest-neighbour search over the `Span` dense index -- the dense
703
+ leg of `span_hybrid_search` WITHOUT the sparse leg or RRF fusion, so a strong semantic match a short
704
+ common-token query's sparse leg would crowd out of the fused pool is still recoverable. Returns the same
705
+ row shape as `span_hybrid_search` (span_id + parent pointers + function), in descending cosine order.
706
+ `documents` scopes to a workspace exactly as the hybrid search does (`contract_id IN [...]`, with the
707
+ larger scoped pool before the cut); `documents=[]` is scope-to-nothing."""
708
+ if documents is not None and not documents:
709
+ return []
710
+ leg_k = max(k, SCOPED_CANDIDATE_POOL if documents else DEFAULT_CANDIDATE_POOL)
711
+ dense = _float_array(dense_query)
712
+ neighbours = f"SELECT expand(`vector.neighbors`('{_SPAN_DENSE_INDEX}', {dense}, {leg_k}))"
713
+ where = f" WHERE contract_id IN {_str_array(documents)}" if documents else ""
714
+ return self._query(
715
+ f"SELECT span_id, parent_chunk_id, parent_okf_path, function FROM ({neighbours}){where} LIMIT {k}"
716
+ )
717
+
718
+ def span_texts(self, span_ids: list[str]) -> dict[str, str]:
719
+ """The operative-span text for each span_id (batched), for citing a retrieved span. {span_id: text}."""
720
+ if not span_ids:
721
+ return {}
722
+ id_list = "[" + ",".join(_sql_str(s) for s in span_ids) + "]"
723
+ rows = self._query(f"SELECT span_id, text FROM {SPAN_TYPE} WHERE span_id IN {id_list}")
724
+ return {r["span_id"]: r.get("text", "") for r in rows}
725
+
726
+ def mark_span_properties_ambiguous(self, span_ids: list[str]) -> int:
727
+ """ADR-0048 Phase A mark-stale: set `confidence = AMBIGUOUS` on every typed property edge of these spans.
728
+ A clause whose PRIMARY function flipped had its properties extracted for the OLD function, so they are
729
+ stale until Phase B re-extraction -- downgraded (kept but flagged) exactly as the ADR-0040 judges do, so
730
+ the soft-boost down-weights them meanwhile. Keyed by the ADR-0025 `edge.span_id`. Idempotent (skips
731
+ already-AMBIGUOUS). Returns the number of edges downgraded (best-effort from the driver's row count)."""
732
+ if not span_ids:
733
+ return 0
734
+ total = 0
735
+ for stmt in _stale_property_statements(span_ids):
736
+ res = self._command(stmt)
737
+ if isinstance(res, list):
738
+ for row in res:
739
+ if isinstance(row, dict) and "count" in row:
740
+ total += int(row["count"])
741
+ return total
742
+
743
+ def chunk_count(self) -> int:
744
+ rows = self._query(f"SELECT count(*) AS n FROM {CHUNK_TYPE}")
745
+ return int(rows[0]["n"]) if rows else 0
746
+
747
+ # --- query-side (T21) -----------------------------------------------------------------------
748
+
749
+ def hybrid_search(
750
+ self,
751
+ dense_query: list[float],
752
+ sparse_query: dict[int, float],
753
+ *,
754
+ k: int,
755
+ filters: dict[str, MetadataValue] | None = None,
756
+ ) -> list[dict]:
757
+ """Server-side RRF fusion of the dense and sparse legs, honoring equality metadata filters.
758
+
759
+ Grounded and proven end to end at T14: `vector.fuse` fuses a dense `vector.neighbors` leg and a
760
+ sparse `vector.sparseNeighbors` leg with the RRF strategy; `expand` flattens the fused list into
761
+ rows. Each leg is fetched to `DEFAULT_CANDIDATE_POOL` (or `k` if larger) so fusion and the filter
762
+ have a real pool to work over; the fused, ranked result is then filtered and cut to `k`. Dotted
763
+ function names are backtick-quoted. The `arcadedb-python` API does not wrap these functions
764
+ (ADR-0007/ADR-0008), so they are issued through the grounded `query()` method.
765
+ """
766
+ leg_k = max(k, DEFAULT_CANDIDATE_POOL)
767
+ token_ids = sorted(sparse_query) # deterministic order across the paired arrays
768
+ sparse_indices = "[" + ",".join(str(i) for i in token_ids) + "]"
769
+ sparse_weights = _float_array(sparse_query[i] for i in token_ids)
770
+ dense = _float_array(dense_query)
771
+ fused = (
772
+ "SELECT expand(`vector.fuse`("
773
+ f"`vector.neighbors`('{_DENSE_INDEX}', {dense}, {leg_k}), "
774
+ f"`vector.sparseNeighbors`('{_SPARSE_INDEX}', {sparse_indices}, {sparse_weights}, {leg_k}), "
775
+ "{ fusion: 'RRF' }))"
776
+ )
777
+ where = ""
778
+ if filters:
779
+ clauses = " AND ".join(f"{col} = {_sql_literal(v)}" for col, v in filters.items())
780
+ where = f" WHERE {clauses}"
781
+ return self._query(f"SELECT chunk_id, source_doc_id FROM ({fused}){where} LIMIT {k}")
782
+
783
+ # --- graph-write (T25) ----------------------------------------------------------------------
784
+
785
+ def write_graph(self, nodes: list[GraphNode], edges: list[GraphEdge]) -> None:
786
+ """Upsert entity nodes and create relationship edges in ONE transaction (FR-S.1). Each entity
787
+ is connected to its source chunk by a `Mentions` edge (for chunks that exist), so a chunk and
788
+ its extracted entities land together. Uses `execute_transaction` so a failure rolls back whole.
789
+
790
+ Idempotent (issue 0029 / ADR-0092): re-ingesting the same document CONVERGES instead of appending.
791
+ Nodes stay UPSERT (already idempotent). A `Mentions` edge is created only if absent on (chunk,
792
+ entity); a `Relationship` edge only if absent on (source, target, relationship_type, chunk_id) --
793
+ so a genuinely distinct edge from a different contract (a different provenance `chunk_id`) still
794
+ writes, while a re-ingest of the same document adds nothing. Existence is checked with read-only
795
+ pre-queries BEFORE the transaction; correctness no longer depends on a caller's `already_ingested`
796
+ guard (which stays a useful whole-pipeline short-circuit)."""
797
+ statements: list[str] = []
798
+ for node in nodes: # nodes first, so edge endpoints exist within the transaction
799
+ statements.append(
800
+ f"UPDATE {ENTITY_TYPE} SET"
801
+ f" entity_id = {_sql_str(node.node_key)},"
802
+ f" canonical_id = {_sql_str(node.entity_id)}," # ADR-0067 P5c: the resolver's canonical id
803
+ f" name = {_sql_str(node.name)},"
804
+ f" entity_type = {_sql_str(node.entity_type)},"
805
+ f" confidence = {_sql_str(node.confidence)},"
806
+ f" chunk_id = {_sql_str(node.chunk_id)}"
807
+ f" UPSERT WHERE entity_id = {_sql_str(node.node_key)}"
808
+ )
809
+ existing_chunks = self._existing_chunks({node.chunk_id for node in nodes})
810
+ seen_mentions: set[tuple[str, str]] = set() # de-dupe within this batch too (full convergence)
811
+ for node in nodes:
812
+ key = (node.chunk_id, node.node_key)
813
+ if node.chunk_id in existing_chunks and key not in seen_mentions \
814
+ and not self._mentions_edge_exists(node.chunk_id, node.node_key):
815
+ seen_mentions.add(key)
816
+ statements.append( # connect chunk -> entity (provenance edge)
817
+ f"CREATE EDGE {MENTIONS_EDGE_TYPE}"
818
+ f" FROM (SELECT FROM {CHUNK_TYPE} WHERE chunk_id = {_sql_str(node.chunk_id)})"
819
+ f" TO (SELECT FROM {ENTITY_TYPE} WHERE entity_id = {_sql_str(node.node_key)})"
820
+ )
821
+ seen_edges: set[tuple[str, str, str, str]] = set()
822
+ for edge in edges:
823
+ key = (edge.source_key, edge.target_key, edge.relationship_type, edge.chunk_id)
824
+ if key in seen_edges or self._relationship_edge_exists(
825
+ edge.source_key, edge.target_key, edge.relationship_type, edge.chunk_id):
826
+ continue
827
+ seen_edges.add(key)
828
+ statements.append(
829
+ f"CREATE EDGE {REL_EDGE_TYPE}"
830
+ f" FROM (SELECT FROM {ENTITY_TYPE} WHERE entity_id = {_sql_str(edge.source_key)})"
831
+ f" TO (SELECT FROM {ENTITY_TYPE} WHERE entity_id = {_sql_str(edge.target_key)})"
832
+ f" SET relationship_type = {_sql_str(edge.relationship_type)},"
833
+ f" confidence = {_sql_str(edge.confidence)}, {_edge_provenance_assignments(edge.chunk_id)}"
834
+ )
835
+ if statements:
836
+ self._db.execute_transaction(statements)
837
+
838
+ def _mentions_edge_exists(self, chunk_id: str, node_key: str) -> bool:
839
+ """True iff a `Mentions` edge already connects this chunk to this entity (issue 0029 idempotence).
840
+ Endpoint properties are reached with `outV()`/`inV()`: a plain `out.<prop>`/`in.<prop>` projection
841
+ returns NULL on this ArcadeDB (verified live), which would silently defeat the existence check."""
842
+ rows = self._query(
843
+ f"SELECT count(*) AS c FROM {MENTIONS_EDGE_TYPE}"
844
+ f" WHERE outV().chunk_id = {_sql_str(chunk_id)} AND inV().entity_id = {_sql_str(node_key)}")
845
+ return bool(rows) and (rows[0].get("c") or 0) > 0
846
+
847
+ def _relationship_edge_exists(self, source_key: str, target_key: str, rel_type: str, chunk_id: str) -> bool:
848
+ """True iff a `Relationship` edge already exists on (source, target, relationship_type, chunk_id)
849
+ -- the full provenance key, so distinct edges from different contracts are not collapsed (0029).
850
+ `outV()`/`inV()` reach the endpoint entity_ids (a bare `out.entity_id` projects NULL here)."""
851
+ rows = self._query(
852
+ f"SELECT count(*) AS c FROM {REL_EDGE_TYPE}"
853
+ f" WHERE relationship_type = {_sql_str(rel_type)}"
854
+ f" AND outV().entity_id = {_sql_str(source_key)} AND inV().entity_id = {_sql_str(target_key)}"
855
+ f" AND chunk_id = {_sql_str(chunk_id)}")
856
+ return bool(rows) and (rows[0].get("c") or 0) > 0
857
+
858
+ def add_affiliation_edges(self, nodes: list[GraphNode], edges: list[GraphEdge]) -> int:
859
+ """issue 0027 BACKFILL (edge-only): add `AFFILIATE_OF` edges to an ALREADY-INGESTED KG without re-writing
860
+ it. Creates an entity node ONLY IF ABSENT (never overwrites an existing node's name/id -- unlike
861
+ `write_graph`'s UPSERT, so a party node keeps its resolved identity), and creates each edge ONLY IF ABSENT.
862
+ Idempotent -- re-running adds nothing. Returns the number of edges created. Used by
863
+ `scripts/backfill_affiliations.py`; the live ingest path uses `write_graph` unchanged."""
864
+ for node in nodes:
865
+ if self._query(f"SELECT entity_id FROM {ENTITY_TYPE} WHERE entity_id = {_sql_str(node.node_key)} LIMIT 1"):
866
+ continue # keep the existing node exactly as it is
867
+ self._command(
868
+ f"INSERT INTO {ENTITY_TYPE} SET entity_id = {_sql_str(node.node_key)},"
869
+ f" canonical_id = {_sql_str(node.entity_id)}, name = {_sql_str(node.name)},"
870
+ f" entity_type = {_sql_str(node.entity_type)}, confidence = {_sql_str(node.confidence)},"
871
+ f" chunk_id = {_sql_str(node.chunk_id)}")
872
+ added = 0
873
+ for edge in edges:
874
+ exists = self._query( # outV()/inV(): a bare out.entity_id projects NULL on this ArcadeDB (issue 0029)
875
+ f"SELECT count(*) AS c FROM {REL_EDGE_TYPE}"
876
+ f" WHERE relationship_type = {_sql_str(edge.relationship_type)}"
877
+ f" AND outV().entity_id = {_sql_str(edge.source_key)} AND inV().entity_id = {_sql_str(edge.target_key)}")
878
+ if exists and (exists[0].get("c") or 0) > 0:
879
+ continue
880
+ self._command(
881
+ f"CREATE EDGE {REL_EDGE_TYPE}"
882
+ f" FROM (SELECT FROM {ENTITY_TYPE} WHERE entity_id = {_sql_str(edge.source_key)})"
883
+ f" TO (SELECT FROM {ENTITY_TYPE} WHERE entity_id = {_sql_str(edge.target_key)})"
884
+ f" SET relationship_type = {_sql_str(edge.relationship_type)},"
885
+ f" confidence = {_sql_str(edge.confidence)}, {_edge_provenance_assignments(edge.chunk_id)}")
886
+ added += 1
887
+ return added
888
+
889
+ def all_contracts(self) -> list[dict]:
890
+ """Every contract id in the store (e.g. for a corpus-wide backfill pass)."""
891
+ return self._query(f"SELECT contract_id FROM {CONTRACT_TYPE}")
892
+
893
+ def known_document_ids(self) -> set[str]:
894
+ """Issue 0031: the DISTINCT ids of every INGESTED document -- the `Contract` nodes (CU-B3), the
895
+ per-document registry written once per ingested source document. This is the validation set for a
896
+ `documents` scope: a scope naming a document that was never ingested raises (mirrors issue 0007's
897
+ `requirement_sources`), but a document that WAS ingested yet indexed nothing (unreadable / empty ->
898
+ no spans, no edges) is a KNOWN document and PASSES -- it simply contributes nothing to the sweep,
899
+ which the product reports through its coverage line rather than having one bad document raise and
900
+ break the whole matter's sweep. (Deliberately the ingested-document set, not the narrower union of
901
+ documents that produced spans or edges.) The Contract type may not exist yet on a fresh DB -> empty."""
902
+ if CONTRACT_TYPE not in self.type_names():
903
+ return set()
904
+ return {r["contract_id"] for r in self.all_contracts() if r.get("contract_id")}
905
+
906
+ def entities_by_name(self, name: str) -> list[dict]:
907
+ """Resolve a party NAME to its graph entities (issue 0030 / ADR-0093): the first step before
908
+ `graph_neighbors`/`graph_query`, which take an exact `start_entity_id` and cannot be reached from a
909
+ name otherwise. Returns `[{entity_id, name, entity_type}]` for every stored entity whose name
910
+ normalizes to the same clustering key as `name`, via the SAME `normalize_entity_name` the ingestion
911
+ side uses to cluster ('Acme Corp' / 'Acme Corporation' / 'ACME, Inc.' -> one key). A name may resolve
912
+ to SEVERAL nodes (a resolved node plus a not-yet-merged unlinked ref) -- all are returned, each usable
913
+ as a `start_entity_id`. Normalization is the engine's rule and is applied HERE (a raw or an already-
914
+ normalized name both work; the normalization is idempotent). Empty list on no match / a non-entity name.
915
+
916
+ (issue 0030 replaces the retired `all_entities`, which was removed with the KG-7 PartyTo capability but
917
+ was the only name->entity route; this puts the capability on the Store seam and owns the normalization
918
+ rather than forcing every caller to re-implement it against an engine internal.)"""
919
+ target = normalize_entity_name(name)
920
+ if not target: # empty / whitespace / non-entity: no lookup key
921
+ return []
922
+ rows = self._query(f"SELECT entity_id, name, entity_type FROM {ENTITY_TYPE}")
923
+ return [
924
+ {"entity_id": r["entity_id"], "name": r["name"], "entity_type": r["entity_type"]}
925
+ for r in rows
926
+ if normalize_entity_name(r.get("name") or "") == target
927
+ ]
928
+
929
+ # --- ADR-0044: the IS_EXCEPTION_TO derived carve-out relationship (exception clause -> Cap clause) ------
930
+
931
+ def clause_positions(self, functions: list[str]) -> list[dict]:
932
+ """Clauses of the given functions with their operative-span DOCUMENT offsets (via the clause-level
933
+ span_id, ADR-0042), for proximity-based exception linking. Rows: {clause_id, function, contract_id,
934
+ doc_start, doc_end}. A clause with no resolvable span (legacy/unbackfilled) is skipped."""
935
+ if not functions:
936
+ return []
937
+ fn_list = "[" + ",".join(_sql_str(f) for f in functions) + "]"
938
+ clauses = self._query(
939
+ f"SELECT clause_id, function, span_id FROM {CLAUSE_TYPE} WHERE function IN {fn_list}")
940
+ span_ids = [c["span_id"] for c in clauses if c.get("span_id")]
941
+ if not span_ids:
942
+ return []
943
+ id_list = "[" + ",".join(_sql_str(s) for s in span_ids) + "]"
944
+ spans = self._query(
945
+ f"SELECT span_id, doc_start, doc_end, contract_id FROM {SPAN_TYPE} WHERE span_id IN {id_list}")
946
+ by_span = {s["span_id"]: s for s in spans}
947
+ out: list[dict] = []
948
+ for c in clauses:
949
+ s = by_span.get(c.get("span_id"))
950
+ if s is None:
951
+ continue
952
+ out.append({"clause_id": c["clause_id"], "function": c["function"],
953
+ "contract_id": s.get("contract_id"), "doc_start": s.get("doc_start"),
954
+ "doc_end": s.get("doc_end")})
955
+ return out
956
+
957
+ def write_clause_exception_links(self, links: list) -> None:
958
+ """ADR-0044: write the `IsExceptionTo` edges (exception/Uncapped clause -> the Cap clause it excepts).
959
+ Idempotent: clears the existing IsExceptionTo layer first, so re-linking is safe and re-derivable. The
960
+ edge carries the INFERRED confidence (a derived, reasoned link, FR-S.4). One transaction."""
961
+ statements = (
962
+ [f"DELETE FROM {IS_EXCEPTION_TO_EDGE_TYPE} UNSAFE"]
963
+ if IS_EXCEPTION_TO_EDGE_TYPE in self.type_names() else [])
964
+ for link in links:
965
+ statements.append(
966
+ f"CREATE EDGE {IS_EXCEPTION_TO_EDGE_TYPE}"
967
+ f" FROM (SELECT FROM {CLAUSE_TYPE} WHERE clause_id = {_sql_str(link.exception_clause_id)})"
968
+ f" TO (SELECT FROM {CLAUSE_TYPE} WHERE clause_id = {_sql_str(link.cap_clause_id)})"
969
+ f" SET confidence = {_sql_str(link.confidence.value)}"
970
+ )
971
+ if statements:
972
+ self._db.execute_transaction(statements)
973
+
974
+ def graph_counts(self) -> dict[str, int]:
975
+ entities = self._query(f"SELECT count(*) AS n FROM {ENTITY_TYPE}")
976
+ rels = self._query(f"SELECT count(*) AS n FROM {REL_EDGE_TYPE}")
977
+ return {
978
+ "entities": int(entities[0]["n"]) if entities else 0,
979
+ "relationships": int(rels[0]["n"]) if rels else 0,
980
+ }
981
+
982
+ def graph_neighbors(
983
+ self, entity_id: str, *, relationship_type: str, max_hops: int, documents: list[str] | None = None
984
+ ) -> list[dict]:
985
+ """Traverse via ArcadeDB `MATCH` over `Relationship` edges (grounded live, T26): `bothE` binds
986
+ each edge (so its `chunk_id`/`confidence` are cited) and `bothV` the reached entity. One-hop and
987
+ two-hop are separate MATCH queries; `$matched` de-dups the two-hop return to the start. Braces
988
+ are concatenated in (they clash with f-string interpolation).
989
+
990
+ Issue 0031: `documents` scopes the traversal to a workspace's source documents -- EVERY edge on the
991
+ path must have `source_doc_id IN [...]` (applied to e1 AND e2, so a two-hop path cannot route THROUGH
992
+ an out-of-scope contract to reach an in-scope target). `None` = the whole graph; `[]` = scope-to-
993
+ nothing (no query)."""
994
+ if documents is not None and not documents:
995
+ return [] # scope-to-nothing: never issue an invalid `IN []`
996
+ eid = _sql_str(entity_id)
997
+ rel = _sql_str(relationship_type)
998
+ # each edge's where-clause: relationship_type, plus (0031) the document scope on the edge itself
999
+ doc_scope = f" and source_doc_id IN {_str_array(documents)}" if documents else ""
1000
+ edge_where = "(relationship_type = " + rel + doc_scope + ")"
1001
+ paths: list[dict] = []
1002
+
1003
+ one_hop = (
1004
+ "MATCH {type: " + ENTITY_TYPE + ", as: a, where: (entity_id = " + eid + ")}"
1005
+ ".bothE('" + REL_EDGE_TYPE + "'){as: e, where: " + edge_where + "}"
1006
+ ".bothV(){as: b, where: (entity_id <> " + eid + ")}"
1007
+ " RETURN b.entity_id AS target_id, b.name AS target_name,"
1008
+ " e.chunk_id AS c1, e.confidence AS cf1"
1009
+ )
1010
+ for row in self._query(one_hop):
1011
+ paths.append({
1012
+ "target_id": row["target_id"], "target_name": row["target_name"],
1013
+ "path_entity_ids": [entity_id, row["target_id"]],
1014
+ "path_chunk_ids": [row["c1"]], "path_confidences": [row["cf1"]], "hops": 1,
1015
+ })
1016
+
1017
+ if max_hops >= 2:
1018
+ two_hop = (
1019
+ "MATCH {type: " + ENTITY_TYPE + ", as: a, where: (entity_id = " + eid + ")}"
1020
+ ".bothE('" + REL_EDGE_TYPE + "'){as: e1, where: " + edge_where + "}"
1021
+ ".bothV(){as: b, where: (entity_id <> " + eid + ")}"
1022
+ ".bothE('" + REL_EDGE_TYPE + "'){as: e2, where: " + edge_where + "}"
1023
+ ".bothV(){as: cc, where: (entity_id <> " + eid
1024
+ + " and entity_id <> $matched.b.entity_id)}"
1025
+ " RETURN b.entity_id AS mid_id, e1.chunk_id AS e1c, e1.confidence AS e1cf,"
1026
+ " cc.entity_id AS target_id, cc.name AS target_name,"
1027
+ " e2.chunk_id AS e2c, e2.confidence AS e2cf"
1028
+ )
1029
+ for row in self._query(two_hop):
1030
+ paths.append({
1031
+ "target_id": row["target_id"], "target_name": row["target_name"],
1032
+ "path_entity_ids": [entity_id, row["mid_id"], row["target_id"]],
1033
+ "path_chunk_ids": [row["e1c"], row["e2c"]],
1034
+ "path_confidences": [row["e1cf"], row["e2cf"]], "hops": 2,
1035
+ })
1036
+ return paths
1037
+
1038
+ # --- property graph (T57c, FR-R) ------------------------------------------------------------
1039
+
1040
+ def clear_property_graph(self) -> None:
1041
+ """Delete all property-graph records (Clause / PropertyValue / HasProperty) while LEAVING the span
1042
+ index intact -- so a property re-extraction can start from scratch without re-embedding (T58 resume
1043
+ control). Edges first (UNSAFE bypasses the edge-safety check this dialect requires), then vertices."""
1044
+ self._command(f"DELETE FROM {PROPERTY_EDGE_TYPE} UNSAFE")
1045
+ self._command(f"DELETE FROM {CLAUSE_TYPE}")
1046
+ self._command(f"DELETE FROM {PROPVALUE_TYPE}")
1047
+
1048
+ def property_graph_counts(self) -> dict[str, int]:
1049
+ """Counts for introspection/tests: clauses, shared property-value nodes, and property edges."""
1050
+ clauses = self._query(f"SELECT count(*) AS n FROM {CLAUSE_TYPE}")
1051
+ values = self._query(f"SELECT count(*) AS n FROM {PROPVALUE_TYPE}")
1052
+ edges = self._query(f"SELECT count(*) AS n FROM {PROPERTY_EDGE_TYPE}")
1053
+ return {
1054
+ "clauses": int(clauses[0]["n"]) if clauses else 0,
1055
+ "property_values": int(values[0]["n"]) if values else 0,
1056
+ "property_edges": int(edges[0]["n"]) if edges else 0,
1057
+ }
1058
+
1059
+ def clause_property_values(self, clause_id: str) -> list[dict]:
1060
+ """The property values a clause asserts, each with the edge's provenance (dimension, value,
1061
+ confidence, span_id) -- the readback for tests and the shape T58's query builds on."""
1062
+ q = (
1063
+ "MATCH {type: " + CLAUSE_TYPE + ", as: c, where: (clause_id = " + _sql_str(clause_id) + ")}"
1064
+ ".outE('" + PROPERTY_EDGE_TYPE + "'){as: e}.inV(){as: v}"
1065
+ " RETURN v.dimension AS dimension, v.value AS value, e.confidence AS confidence,"
1066
+ " e.span_id AS span_id"
1067
+ )
1068
+ return self._query(q)
1069
+
1070
+ # --- KG-3 (ADR-0033): the TYPED unified clause KG (replaces the flat HasProperty write path) ---------
1071
+
1072
+ def clause_kg_counts(self) -> dict[str, int]:
1073
+ """Counts for the typed KG (introspection/tests): clauses, shared value nodes, and the total of the
1074
+ typed property edges across all typed edge types."""
1075
+ clauses = self._query(f"SELECT count(*) AS n FROM {CLAUSE_TYPE}")
1076
+ values = self._query(f"SELECT count(*) AS n FROM {PROPVALUE_TYPE}")
1077
+ typed = 0
1078
+ present = self.type_names() # a DB populated before a schema extension lacks the newer edge types
1079
+ for edge_type in TYPED_PROPERTY_EDGE_TYPES:
1080
+ if edge_type not in present:
1081
+ continue
1082
+ rows = self._query(f"SELECT count(*) AS n FROM {edge_type}")
1083
+ typed += int(rows[0]["n"]) if rows else 0
1084
+ return {
1085
+ "clauses": int(clauses[0]["n"]) if clauses else 0,
1086
+ "property_values": int(values[0]["n"]) if values else 0,
1087
+ "typed_edges": typed,
1088
+ }
1089
+
1090
+ def clear_clause_kg(self) -> None:
1091
+ """Delete the typed clause KG (all typed edges + the legacy flat edge + Clause + PropertyValue),
1092
+ leaving the span index intact -- the KG-3 counterpart of `clear_property_graph` for a clean
1093
+ re-extraction into the typed schema. Edges first (UNSAFE bypasses the edge-safety check)."""
1094
+ present = self.type_names() # skip edge types a pre-extension DB never created
1095
+ for edge_type in (*TYPED_PROPERTY_EDGE_TYPES, PROPERTY_EDGE_TYPE):
1096
+ if edge_type in present:
1097
+ self._command(f"DELETE FROM {edge_type} UNSAFE")
1098
+ self._command(f"DELETE FROM {CLAUSE_TYPE}")
1099
+ self._command(f"DELETE FROM {PROPVALUE_TYPE}")
1100
+
1101
+ def _contract_bounds(self, contract_id: str) -> tuple[str, str]:
1102
+ return _sql_str(contract_id + ":"), _sql_str(contract_id + ";")
1103
+
1104
+ def clauses_in_contract(self, contract_id: str) -> list[dict]:
1105
+ """Every clause in one contract (Leg-A scope): clause_id, function, folio_iri -- including clauses
1106
+ with no typed properties (still queryable by type)."""
1107
+ lo, hi = self._contract_bounds(contract_id)
1108
+ return self._query(
1109
+ f"SELECT clause_id, function, folio_iri, span_id FROM {CLAUSE_TYPE} "
1110
+ f"WHERE clause_id >= {lo} AND clause_id < {hi} ORDER BY clause_id"
1111
+ )
1112
+
1113
+ def _existing_chunks(self, chunk_ids: set[str]) -> set[str]:
1114
+ if not chunk_ids:
1115
+ return set()
1116
+ rows = self._query(
1117
+ f"SELECT chunk_id FROM {CHUNK_TYPE} WHERE chunk_id IN {_str_array(sorted(chunk_ids))}"
1118
+ )
1119
+ return {row["chunk_id"] for row in rows}
1120
+
1121
+ # --- test / lifecycle helper ----------------------------------------------------------------
1122
+
1123
+ def drop(self) -> None:
1124
+ """Delete the database (used to reset a scratch/test database)."""
1125
+ if DatabaseDao.exists(self._client, self._database):
1126
+ DatabaseDao.delete(self._client, self._database)
1127
+
1128
+ # --- internals ------------------------------------------------------------------------------
1129
+
1130
+ def _command(self, sql: str) -> Any:
1131
+ return self._db.query("sql", sql, is_command=True)
1132
+
1133
+ def _query(self, sql: str) -> list[dict]:
1134
+ result = self._db.query("sql", sql)
1135
+ return result if isinstance(result, list) else []