ltcai 11.2.0 → 11.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (247) hide show
  1. package/README.md +46 -53
  2. package/docs/CHANGELOG.md +61 -0
  3. package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
  4. package/docs/DEVELOPMENT.md +1 -1
  5. package/docs/MULTI_AGENT_RUNTIME.md +1 -1
  6. package/docs/ONBOARDING.md +1 -1
  7. package/docs/OPERATIONS.md +6 -2
  8. package/docs/PERMISSION_MODE.md +1 -1
  9. package/docs/TRUST_MODEL.md +1 -1
  10. package/docs/WHY_LATTICE.md +1 -1
  11. package/docs/kg-schema.md +2 -2
  12. package/docs/v11.3.0_PLAN.md +202 -0
  13. package/docs/v11.4.0_RUST_FOUNDATION_PLAN.md +176 -0
  14. package/lattice_brain/__init__.py +1 -1
  15. package/lattice_brain/graph/_kg_common/__init__.py +287 -0
  16. package/lattice_brain/graph/_kg_common/extraction.py +516 -0
  17. package/lattice_brain/graph/_kg_common/relations.py +161 -0
  18. package/lattice_brain/graph/_kg_common/text.py +479 -0
  19. package/lattice_brain/graph/discovery_index/__init__.py +35 -0
  20. package/lattice_brain/graph/discovery_index/cleanup.py +182 -0
  21. package/lattice_brain/graph/discovery_index/extract.py +137 -0
  22. package/lattice_brain/graph/discovery_index/scan.py +411 -0
  23. package/lattice_brain/graph/discovery_index/upsert.py +495 -0
  24. package/lattice_brain/graph/projection/__init__.py +42 -0
  25. package/lattice_brain/graph/projection/curation.py +500 -0
  26. package/lattice_brain/graph/{projection.py → projection/v2_schema.py} +15 -477
  27. package/lattice_brain/graph/retrieval/__init__.py +54 -0
  28. package/lattice_brain/graph/retrieval/context.py +197 -0
  29. package/lattice_brain/graph/retrieval/graph_view.py +319 -0
  30. package/lattice_brain/graph/retrieval/hybrid.py +488 -0
  31. package/lattice_brain/graph/retrieval/maintenance.py +121 -0
  32. package/lattice_brain/graph/retrieval/signals.py +95 -0
  33. package/lattice_brain/graph/retrieval_vector/__init__.py +42 -0
  34. package/lattice_brain/graph/retrieval_vector/fingerprint.py +97 -0
  35. package/lattice_brain/graph/retrieval_vector/indexing.py +347 -0
  36. package/lattice_brain/graph/retrieval_vector/search.py +560 -0
  37. package/lattice_brain/graph/retrieval_vector/status.py +374 -0
  38. package/lattice_brain/ingestion/__init__.py +130 -0
  39. package/lattice_brain/ingestion/_contract.py +90 -0
  40. package/lattice_brain/ingestion/constants.py +127 -0
  41. package/lattice_brain/ingestion/folder_scan.py +57 -0
  42. package/lattice_brain/ingestion/folders.py +258 -0
  43. package/lattice_brain/ingestion/hashing.py +26 -0
  44. package/lattice_brain/ingestion/jobs_api.py +107 -0
  45. package/lattice_brain/ingestion/models.py +80 -0
  46. package/lattice_brain/ingestion/pipeline.py +486 -0
  47. package/lattice_brain/ingestion/quality.py +209 -0
  48. package/lattice_brain/ingestion/routing.py +295 -0
  49. package/lattice_brain/multimodal/__init__.py +164 -0
  50. package/lattice_brain/multimodal/audio.py +77 -0
  51. package/lattice_brain/multimodal/common.py +118 -0
  52. package/lattice_brain/multimodal/images.py +498 -0
  53. package/lattice_brain/multimodal/ports.py +169 -0
  54. package/lattice_brain/multimodal/video.py +410 -0
  55. package/lattice_brain/portability/__init__.py +90 -0
  56. package/lattice_brain/portability/_contract.py +42 -0
  57. package/lattice_brain/portability/backups.py +338 -0
  58. package/lattice_brain/portability/bundles.py +136 -0
  59. package/lattice_brain/portability/constants.py +93 -0
  60. package/lattice_brain/portability/fsops.py +138 -0
  61. package/lattice_brain/portability/service.py +41 -0
  62. package/lattice_brain/{portability.py → portability/sharing.py} +44 -677
  63. package/lattice_brain/runtime/__init__.py +1 -1
  64. package/lattice_brain/runtime/multi_agent.py +1 -1
  65. package/latticeai/__init__.py +1 -1
  66. package/latticeai/api/chronicle.py +63 -0
  67. package/latticeai/core/agent/__init__.py +93 -0
  68. package/latticeai/core/agent/_contract.py +79 -0
  69. package/latticeai/core/agent/context.py +57 -0
  70. package/latticeai/core/agent/deps.py +125 -0
  71. package/latticeai/core/agent/execution.py +622 -0
  72. package/latticeai/core/agent/planning.py +145 -0
  73. package/latticeai/core/agent/recovery.py +157 -0
  74. package/latticeai/core/agent/runtime.py +210 -0
  75. package/latticeai/core/agent/verification.py +231 -0
  76. package/latticeai/core/embedding_providers/__init__.py +151 -0
  77. package/latticeai/core/embedding_providers/base.py +199 -0
  78. package/latticeai/core/embedding_providers/captions.py +162 -0
  79. package/latticeai/core/embedding_providers/profiles.py +126 -0
  80. package/latticeai/core/embedding_providers/text.py +350 -0
  81. package/latticeai/core/embedding_providers/vision.py +352 -0
  82. package/latticeai/core/file_generation/__init__.py +115 -0
  83. package/latticeai/core/file_generation/bundles.py +76 -0
  84. package/latticeai/core/file_generation/extraction.py +154 -0
  85. package/latticeai/core/file_generation/inference.py +235 -0
  86. package/latticeai/core/file_generation/orchestration.py +152 -0
  87. package/latticeai/core/file_generation/prompting.py +117 -0
  88. package/latticeai/core/file_generation/repair.py +114 -0
  89. package/latticeai/core/file_generation/sanitize.py +61 -0
  90. package/latticeai/core/file_generation/validation.py +201 -0
  91. package/latticeai/core/legacy_compatibility.py +1 -1
  92. package/latticeai/core/marketplace.py +1 -1
  93. package/latticeai/core/messages.py +9 -0
  94. package/latticeai/core/workspace_os_constants.py +1 -1
  95. package/latticeai/integrations/telegram_bot/__init__.py +123 -0
  96. package/latticeai/integrations/telegram_bot/__main__.py +17 -0
  97. package/latticeai/integrations/telegram_bot/config.py +86 -0
  98. package/latticeai/integrations/telegram_bot/dispatch.py +311 -0
  99. package/latticeai/integrations/telegram_bot/flows.py +478 -0
  100. package/latticeai/integrations/telegram_bot/helpers.py +322 -0
  101. package/latticeai/integrations/telegram_bot/screens.py +394 -0
  102. package/latticeai/models/router/__init__.py +88 -0
  103. package/latticeai/models/router/_contract.py +66 -0
  104. package/latticeai/models/router/branding.py +56 -0
  105. package/latticeai/models/router/catalog.py +69 -0
  106. package/latticeai/models/router/documents.py +199 -0
  107. package/latticeai/models/router/errors.py +37 -0
  108. package/latticeai/models/router/generation.py +258 -0
  109. package/latticeai/models/router/loading.py +291 -0
  110. package/latticeai/models/router/local_models.py +85 -0
  111. package/latticeai/models/router/registry.py +147 -0
  112. package/latticeai/runtime/build_phases/__init__.py +82 -0
  113. package/latticeai/runtime/build_phases/features.py +407 -0
  114. package/latticeai/runtime/build_phases/foundation.py +555 -0
  115. package/latticeai/runtime/build_phases/web.py +492 -0
  116. package/latticeai/runtime/runtime_context.py +1 -0
  117. package/latticeai/services/architecture_readiness.py +48 -19
  118. package/latticeai/services/brain_intelligence/__init__.py +58 -0
  119. package/latticeai/services/brain_intelligence/_contract.py +71 -0
  120. package/latticeai/services/brain_intelligence/consistency.py +193 -0
  121. package/latticeai/services/brain_intelligence/constants.py +47 -0
  122. package/latticeai/services/brain_intelligence/digest.py +258 -0
  123. package/latticeai/services/brain_intelligence/health.py +331 -0
  124. package/latticeai/services/brain_intelligence/proposals.py +264 -0
  125. package/latticeai/services/brain_intelligence/sampling.py +84 -0
  126. package/latticeai/services/brain_intelligence/service.py +48 -0
  127. package/latticeai/services/chronicle.py +557 -0
  128. package/latticeai/services/memory_service/__init__.py +52 -0
  129. package/latticeai/services/memory_service/_contract.py +100 -0
  130. package/latticeai/services/memory_service/brief.py +431 -0
  131. package/latticeai/services/memory_service/constants.py +57 -0
  132. package/latticeai/services/memory_service/maintenance.py +138 -0
  133. package/latticeai/services/memory_service/manager.py +186 -0
  134. package/latticeai/services/memory_service/proof.py +136 -0
  135. package/latticeai/services/memory_service/recall.py +225 -0
  136. package/latticeai/services/memory_service/service.py +48 -0
  137. package/latticeai/services/memory_service/stores.py +110 -0
  138. package/latticeai/services/model_runtime/__init__.py +322 -0
  139. package/latticeai/services/model_runtime/cloud.py +87 -0
  140. package/latticeai/services/model_runtime/download.py +282 -0
  141. package/latticeai/services/model_runtime/engines.py +341 -0
  142. package/latticeai/services/model_runtime/loading.py +178 -0
  143. package/latticeai/services/model_runtime/service.py +129 -0
  144. package/latticeai/services/model_runtime/state.py +131 -0
  145. package/latticeai/services/model_runtime/status.py +255 -0
  146. package/latticeai/services/product_readiness.py +15 -7
  147. package/latticeai/setup/wizard/__init__.py +126 -0
  148. package/latticeai/setup/wizard/catalog.py +172 -0
  149. package/latticeai/setup/wizard/detect.py +323 -0
  150. package/latticeai/setup/wizard/install.py +348 -0
  151. package/latticeai/setup/wizard/paths.py +168 -0
  152. package/latticeai/setup/wizard/plans.py +74 -0
  153. package/latticeai/setup/wizard/recommend.py +320 -0
  154. package/package.json +6 -2
  155. package/scripts/bump_version.py +14 -0
  156. package/scripts/capture_release_evidence.mjs +33 -21
  157. package/scripts/check_current_release_docs.mjs +1 -1
  158. package/scripts/check_i18n_namespace_coverage.mjs +41 -4
  159. package/scripts/check_max_file_lines.mjs +102 -0
  160. package/scripts/check_release_evidence_bound.mjs +30 -15
  161. package/scripts/check_screenshot_pixel_delta.py +34 -4
  162. package/scripts/check_server_i18n.mjs +1 -0
  163. package/scripts/generate_rust_parity_fixtures.py +562 -0
  164. package/scripts/lib/mock_server_fingerprint.mjs +94 -0
  165. package/scripts/release_screen_claims.json +31 -2
  166. package/src-tauri/Cargo.lock +361 -3
  167. package/src-tauri/Cargo.toml +6 -1
  168. package/src-tauri/src/backend.rs +349 -0
  169. package/src-tauri/src/folder.rs +33 -0
  170. package/src-tauri/src/main.rs +97 -399
  171. package/src-tauri/tauri.conf.json +1 -1
  172. package/static/app/asset-manifest.json +41 -37
  173. package/static/app/assets/Act-yYpYnn0v.js +1 -0
  174. package/static/app/assets/AdminConsole-DL3Cr5pL.js +1 -0
  175. package/static/app/assets/{Brain-tuhI4sOC.js → Brain-C1HBN0Wf.js} +2 -2
  176. package/static/app/assets/BrainHome-DoXRhUUC.js +2 -0
  177. package/static/app/assets/BrainSignals-6yR6ir5t.js +1 -0
  178. package/static/app/assets/Capture-CFIRsFNE.js +1 -0
  179. package/static/app/assets/Chronicle-BZbEgiwN.js +1 -0
  180. package/static/app/assets/CommandPalette-D2pMxC2I.js +1 -0
  181. package/static/app/assets/Library-DwO3yZST.js +1 -0
  182. package/static/app/assets/{LivingBrain-DBwhto14.js → LivingBrain-Jn1GK0-S.js} +1 -1
  183. package/static/app/assets/ProductFlow-B-w1R4Oo.js +1 -0
  184. package/static/app/assets/ReviewCard-6B27X8Vg.js +3 -0
  185. package/static/app/assets/System-DW8F-2xL.js +1 -0
  186. package/static/app/assets/arrow-left-DXvKg9U6.js +1 -0
  187. package/static/app/assets/{bot-Cia42c2h.js → bot-IM_E_Y12.js} +1 -1
  188. package/static/app/assets/brain-Ci1CkWjM.js +1 -0
  189. package/static/app/assets/{button-2j2Ijzgq.js → button-COwyqfHM.js} +1 -1
  190. package/static/app/assets/circle-check-DfInj-qD.js +1 -0
  191. package/static/app/assets/{circle-pause-BEFeWpVW.js → circle-pause-DEM4A1Y5.js} +1 -1
  192. package/static/app/assets/{circle-play-ujXMcHxl.js → circle-play-C9djDuLd.js} +1 -1
  193. package/static/app/assets/{cpu-k4awryFq.js → cpu-DFdo1gw-.js} +1 -1
  194. package/static/app/assets/{download-DFbLJ_ig.js → download-SnJL6oqk.js} +1 -1
  195. package/static/app/assets/{folder-open-7y_b6xkM.js → folder-open-CqZeDkjE.js} +1 -1
  196. package/static/app/assets/{hard-drive-Bidh02Kr.js → hard-drive-j1jJXYYf.js} +1 -1
  197. package/static/app/assets/{index-DwDl9-8Y.css → index-BLPb5lmE.css} +1 -1
  198. package/static/app/assets/index-_u5iUHDr.js +10 -0
  199. package/static/app/assets/input-B0lPdRQZ.js +1 -0
  200. package/static/app/assets/link-2-CoFbooHS.js +1 -0
  201. package/static/app/assets/{permissionCopy-Bpb83Hx9.js → permissionCopy-BsyLxtao.js} +1 -1
  202. package/static/app/assets/primitives-DEbN-d6p.js +1 -0
  203. package/static/app/assets/search-BybIWPNd.js +1 -0
  204. package/static/app/assets/{share-2-BH1M-WNi.js → share-2-CVtZ_ewX.js} +1 -1
  205. package/static/app/assets/{shield-alert-BlKdBXcG.js → shield-alert-CBi2GNWM.js} +1 -1
  206. package/static/app/assets/{textarea-CCWbUfFB.js → textarea-DNMpB5ih.js} +1 -1
  207. package/static/app/assets/{useFocusTrap-YdHQ7pJ1.js → useFocusTrap-C83t3GXF.js} +1 -1
  208. package/static/app/assets/useMutation-DtbJDoyz.js +1 -0
  209. package/static/app/assets/{useQuery-CXQiwbVT.js → useQuery-Dcp1OChy.js} +1 -1
  210. package/static/app/assets/utils-BlZr7Pd4.js +4 -0
  211. package/static/app/assets/workspace-jJY4RuAV.js +1 -0
  212. package/static/app/index.html +4 -4
  213. package/static/sw.js +1 -1
  214. package/lattice_brain/graph/_kg_common.py +0 -1331
  215. package/lattice_brain/graph/discovery_index.py +0 -1141
  216. package/lattice_brain/graph/retrieval.py +0 -1120
  217. package/lattice_brain/graph/retrieval_vector.py +0 -1293
  218. package/lattice_brain/ingestion.py +0 -1525
  219. package/lattice_brain/multimodal.py +0 -1258
  220. package/latticeai/core/agent.py +0 -1465
  221. package/latticeai/core/embedding_providers.py +0 -1196
  222. package/latticeai/core/file_generation.py +0 -1047
  223. package/latticeai/integrations/telegram_bot.py +0 -1390
  224. package/latticeai/models/router.py +0 -1007
  225. package/latticeai/runtime/build_phases.py +0 -1450
  226. package/latticeai/services/brain_intelligence.py +0 -1083
  227. package/latticeai/services/memory_service.py +0 -1177
  228. package/latticeai/services/model_runtime.py +0 -1281
  229. package/latticeai/setup/wizard.py +0 -1310
  230. package/static/app/assets/Act-AWf0SAKp.js +0 -1
  231. package/static/app/assets/AdminConsole-D0u8Tiyj.js +0 -1
  232. package/static/app/assets/BrainHome-Ts7G_Ila.js +0 -2
  233. package/static/app/assets/BrainSignals-jMYgQ2Ar.js +0 -1
  234. package/static/app/assets/Capture-CqOSzyPr.js +0 -1
  235. package/static/app/assets/CommandPalette-DC0Bzh-I.js +0 -1
  236. package/static/app/assets/Library-CX-bbhmK.js +0 -1
  237. package/static/app/assets/ProductFlow-BHA2cfKI.js +0 -1
  238. package/static/app/assets/ReviewCard-BUhCKRNM.js +0 -3
  239. package/static/app/assets/System-Bu2t5hn1.js +0 -1
  240. package/static/app/assets/arrow-left-Dzwa5zRb.js +0 -1
  241. package/static/app/assets/brain-DJMoqrwx.js +0 -1
  242. package/static/app/assets/index-BpYkzcVm.js +0 -10
  243. package/static/app/assets/input-DSlJJxRs.js +0 -1
  244. package/static/app/assets/primitives-BCx6TvfG.js +0 -1
  245. package/static/app/assets/search-Cgy8cCFJ.js +0 -1
  246. package/static/app/assets/utils-zqPZJxdx.js +0 -4
  247. package/static/app/assets/workspace-DXTihhfU.js +0 -1
@@ -0,0 +1,95 @@
1
+ """Honest context-quality and multimodal signals for a result set.
2
+
3
+ Plain functions over an already-computed match list — no store, no I/O — so
4
+ every mixin in the package (and the chat surfaces outside it) can build the
5
+ same signal the same way.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ # ruff: noqa: F403,F405
11
+ from .._kg_common import * # noqa: F403,F401
12
+
13
+ #: Node types that are a *thing you can look at or listen to*, not prose. A
14
+ #: match of one of these means the answer rests on more than text.
15
+ MULTIMODAL_NODE_TYPES = ("Image", "ImageText")
16
+
17
+
18
+ def multimodal_signal(matches: Iterable[Dict[str, Any]]) -> Optional[Dict[str, Any]]:
19
+ """``{"images": n, "types": [...]}`` when a result set includes pictures.
20
+
21
+ ``None`` when it does not: the context-quality contract stays four keys
22
+ wide for the ordinary all-text case, and a caller that sees the key knows
23
+ it means something rather than having to compare a zero.
24
+ """
25
+ images = 0
26
+ seen: List[str] = []
27
+ for match in matches:
28
+ node_type = str(match.get("type") or "")
29
+ if node_type in MULTIMODAL_NODE_TYPES:
30
+ images += 1
31
+ if node_type not in seen:
32
+ seen.append(node_type)
33
+ if not images:
34
+ return None
35
+ return {"images": images, "types": seen}
36
+
37
+
38
+ def context_quality_signal(
39
+ mode: str,
40
+ nodes: int,
41
+ *,
42
+ reason: Optional[str] = None,
43
+ vector: Optional[Dict[str, Any]] = None,
44
+ multimodal: Optional[Dict[str, Any]] = None,
45
+ ) -> Dict[str, Any]:
46
+ """Honest RAG context-quality signal (v9.8.0, additive contract).
47
+
48
+ Shape consumed by the chat metadata channel:
49
+ ``{"mode": "hybrid"|"lexical_only"|"none", "nodes": int, "limited": bool,
50
+ "reason": str|None}``. ``nodes == 0`` always collapses ``mode`` to
51
+ ``"none"``; ``limited`` is true whenever the context is thin (0–1 nodes)
52
+ or the vector side fell back to lexical-only retrieval. ``reason`` is a
53
+ short human-readable Korean phrase, only present when limited.
54
+
55
+ ``vector`` (v11.1.0) carries the vector channel's own honesty block —
56
+ which backend scored, whether it was approximate, whether the candidate
57
+ scan was truncated. "hybrid, 6 nodes" describes two different answers
58
+ depending on those bits, and the caller that has to say "I did not find
59
+ it" deserves to know which one it got. The key is present **only when
60
+ there is a caveat to report**: an exact, complete vector scan is the
61
+ contract's baseline assumption, so annotating it would be noise, and the
62
+ four-key shape stays exactly what existing consumers pin.
63
+
64
+ ``multimodal`` (v11.1.0) follows the same present-only-when-true rule and
65
+ says that part of this context is a picture. "6 nodes" reads differently
66
+ when two of them are screenshots whose text came out of OCR, and the
67
+ surface that has to explain the answer deserves to know.
68
+ """
69
+ nodes = max(0, int(nodes or 0))
70
+ mode = str(mode or "none")
71
+ if nodes == 0:
72
+ mode = "none"
73
+ if mode not in ("hybrid", "lexical_only", "none"):
74
+ mode = "lexical_only"
75
+ limited = nodes <= 1 or mode != "hybrid"
76
+ if reason is None and limited:
77
+ if nodes == 0:
78
+ reason = "그래프에서 관련 지식을 찾지 못했습니다"
79
+ elif mode == "lexical_only":
80
+ reason = "벡터 검색을 사용할 수 없어 키워드 검색 결과만 사용했습니다"
81
+ else:
82
+ reason = "그래프 기반 컨텍스트가 제한적입니다"
83
+ if not limited:
84
+ reason = None
85
+ signal: Dict[str, Any] = {
86
+ "mode": mode,
87
+ "nodes": nodes,
88
+ "limited": limited,
89
+ "reason": reason,
90
+ }
91
+ if vector is not None:
92
+ signal["vector"] = dict(vector)
93
+ if multimodal is not None:
94
+ signal["multimodal"] = dict(multimodal)
95
+ return signal
@@ -0,0 +1,42 @@
1
+ """The derived vector index: build it, report on it, search it.
2
+
3
+ v11.3.0 turned this module into a package. ``KnowledgeGraphVectorMixin`` is
4
+ now composed from four cohesive sub-mixins — embedder fingerprint, index
5
+ build, index status, and search — each moved here verbatim. Every name this
6
+ module exported before still resolves from
7
+ ``lattice_brain.graph.retrieval_vector``.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ # ruff: noqa: F403,F405
13
+ from .._kg_common import * # noqa: F403,F401
14
+ from .fingerprint import _VectorFingerprintMixin
15
+ from .indexing import _VectorIndexingMixin
16
+ from .search import ( # noqa: F401
17
+ DEFAULT_VECTOR_MAX_CANDIDATES,
18
+ VECTOR_MAX_CANDIDATES_CEILING,
19
+ VECTOR_MAX_CANDIDATES_ENV,
20
+ VECTOR_SCAN_BATCH,
21
+ _configured_vector_max_candidates,
22
+ _VectorSearchMixin,
23
+ )
24
+ from .status import _VectorStatusMixin
25
+
26
+
27
+ # Base order is most-composed-first (status → indexing → fingerprint, then
28
+ # search): each half names the ones it calls as its typing-only base, and C3
29
+ # needs a subclass ahead of the class it extends. The method sets are disjoint,
30
+ # so at runtime the order changes nothing.
31
+ class KnowledgeGraphVectorMixin(
32
+ _VectorStatusMixin,
33
+ _VectorIndexingMixin,
34
+ _VectorFingerprintMixin,
35
+ _VectorSearchMixin,
36
+ ):
37
+ """Vector-embedding index build/status/search, split out of retrieval.
38
+
39
+ Composed into KnowledgeGraphStore alongside KnowledgeGraphRetrievalMixin;
40
+ both mixins share the same instance, so vector methods still reach sibling
41
+ retrieval/write helpers (e.g. self._vector_text_for_node) through the MRO.
42
+ """
@@ -0,0 +1,97 @@
1
+ """Which embedder actually built the index, recorded in ``graph_meta``.
2
+
3
+ ``vector_search`` filters on the current model/dim, so swapping the embedder
4
+ silently yields zero vector rows; the fingerprint turns that into the honest
5
+ ``stale_embedder`` signal. Moved verbatim out of ``retrieval_vector.py``.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ from typing import TYPE_CHECKING
11
+
12
+ # ruff: noqa: F403,F405
13
+ from .._kg_common import * # noqa: F403,F401
14
+
15
+ # The cross-mixin surface (`_connect`, `_upsert_node`, …) is declared in
16
+ # `_kg_contract.KnowledgeGraphCore`. It is a typing-only base: at runtime this
17
+ # is `object`, so the MRO of `KnowledgeGraphStore` is unchanged.
18
+ if TYPE_CHECKING:
19
+ from .._kg_contract import KnowledgeGraphCore as _Core
20
+ else:
21
+ _Core = object
22
+
23
+
24
+ class _VectorFingerprintMixin(_Core):
25
+ """Embedder-fingerprint read/write. Composed into the public mixin."""
26
+
27
+ # ── embedder fingerprint (review Wave 2.2 — stale_embedder) ──────────────
28
+ # vector_search filters on the CURRENT model/dim, so swapping the embedder
29
+ # silently yields zero vector rows. The fingerprint persisted in graph_meta
30
+ # records which embedder actually built the index; a mismatch is surfaced
31
+ # as the honest ``stale_embedder`` signal instead of a silent degradation.
32
+
33
+ _EMBEDDER_FINGERPRINT_KEY = "embedder_fingerprint"
34
+
35
+ def _embedder_fingerprint_record(
36
+ self, conn: sqlite3.Connection
37
+ ) -> Optional[Dict[str, Any]]:
38
+ """Read the recorded embedder fingerprint from graph_meta (or None)."""
39
+ row = conn.execute(
40
+ "SELECT value FROM graph_meta WHERE key=?",
41
+ (self._EMBEDDER_FINGERPRINT_KEY,),
42
+ ).fetchone()
43
+ if not row:
44
+ return None
45
+ payload = _safe_loads(row["value"])
46
+ if not isinstance(payload, dict) or not payload.get("model_id"):
47
+ return None
48
+ try:
49
+ dim = int(payload.get("dim") or 0)
50
+ except (TypeError, ValueError):
51
+ dim = 0
52
+ return {"model_id": str(payload["model_id"]), "dim": dim}
53
+
54
+ def _write_embedder_fingerprint(self, conn: sqlite3.Connection) -> Dict[str, Any]:
55
+ """Persist the CURRENT embedder identity (same transaction as caller)."""
56
+ fingerprint = {
57
+ "model_id": self._embedding_model.model_id,
58
+ "dim": int(self._embedding_model.dim),
59
+ }
60
+ conn.execute(
61
+ "INSERT OR REPLACE INTO graph_meta(key, value) VALUES (?, ?)",
62
+ (self._EMBEDDER_FINGERPRINT_KEY, _json(fingerprint)),
63
+ )
64
+ return fingerprint
65
+
66
+ def record_embedder_fingerprint(self) -> Dict[str, Any]:
67
+ """Record the current embedder (model_id + dim) as the index builder."""
68
+ with self._connect() as conn:
69
+ return self._write_embedder_fingerprint(conn)
70
+
71
+ def embedder_fingerprint_status(self) -> Dict[str, Any]:
72
+ """Compare the current embedder against the recorded index fingerprint.
73
+
74
+ Returns ``{"current": {model_id, dim}, "recorded": {...} | None,
75
+ "stale_embedder": bool}``. ``stale_embedder`` is True only when a
76
+ fingerprint was recorded AND it differs from the current embedder —
77
+ an unrecorded index (legacy DBs, nothing indexed yet) is honestly
78
+ "unknown", never reported stale. Never raises.
79
+ """
80
+ current = {
81
+ "model_id": self._embedding_model.model_id,
82
+ "dim": int(self._embedding_model.dim),
83
+ }
84
+ recorded: Optional[Dict[str, Any]] = None
85
+ try:
86
+ with self._connect() as conn:
87
+ recorded = self._embedder_fingerprint_record(conn)
88
+ except Exception: # noqa: BLE001 — status must degrade, never raise
89
+ recorded = None
90
+ stale = bool(
91
+ recorded is not None
92
+ and (
93
+ recorded.get("model_id") != current["model_id"]
94
+ or recorded.get("dim") != current["dim"]
95
+ )
96
+ )
97
+ return {"current": current, "recorded": recorded, "stale_embedder": stale}
@@ -0,0 +1,347 @@
1
+ """Building the vector index: incremental upserts and full rebuilds.
2
+
3
+ Moved verbatim out of ``retrieval_vector.py`` (v11.3.0 decomposition).
4
+ """
5
+
6
+ from __future__ import annotations
7
+
8
+ from typing import TYPE_CHECKING
9
+
10
+ # ruff: noqa: F403,F405
11
+ from .._kg_common import * # noqa: F403,F401
12
+
13
+ # Typing-only base (runtime value is `object`, so the store's MRO is
14
+ # unchanged). Every build path records which embedder produced the rows, so
15
+ # this half calls the fingerprint half through `self`; naming it as the base
16
+ # states that assumption and carries the store contract
17
+ # (`_connect`, `_upsert_node`, …) along with it.
18
+ if TYPE_CHECKING:
19
+ from .fingerprint import _VectorFingerprintMixin as _Core
20
+ else:
21
+ _Core = object
22
+
23
+
24
+ class _VectorIndexingMixin(_Core):
25
+ """Vector index build/refresh. Composed into the public mixin."""
26
+
27
+ def _vector_text_hashes(self, conn: sqlite3.Connection) -> Dict[str, str]:
28
+ """``item_id -> text_hash`` for rows already embedded by *this* embedder.
29
+
30
+ The incremental rebuild's job is mostly deciding what it does *not*
31
+ have to do, and it used to ask that question with one ``SELECT`` per
32
+ candidate item — a round trip per node and per chunk on every run,
33
+ almost all of which answer "unchanged". One query returning two short
34
+ columns replaces all of them.
35
+
36
+ Rows written by a different embedder are left out, so they compare as
37
+ missing and get re-embedded, which is what an embedder swap requires.
38
+ """
39
+ return {
40
+ row["item_id"]: row["text_hash"]
41
+ for row in conn.execute(
42
+ """
43
+ SELECT item_id, text_hash
44
+ FROM vector_embeddings
45
+ WHERE embedding_model=? AND embedding_dim=?
46
+ """,
47
+ (self._embedding_model.model_id, self._embedding_model.dim),
48
+ ).fetchall()
49
+ }
50
+
51
+ def _iter_vector_source_items(
52
+ self,
53
+ conn: sqlite3.Connection,
54
+ *,
55
+ include_nodes: bool = True,
56
+ include_chunks: bool = True,
57
+ ) -> Iterator[Dict[str, Any]]:
58
+ """Stream the graph's embeddable text, one item at a time.
59
+
60
+ Yields rather than returns a list. Every caller consumes this exactly
61
+ once in a ``for``, and building the list first meant a rebuild held
62
+ the full text of every node and chunk in memory simultaneously — the
63
+ one shape guaranteed to fail on precisely the large graph that most
64
+ needs the index.
65
+ """
66
+ if include_nodes:
67
+ for row in conn.execute(
68
+ """
69
+ SELECT id, type, title, summary, metadata_json
70
+ FROM nodes
71
+ WHERE type <> 'Chunk'
72
+ ORDER BY updated_at DESC, id ASC
73
+ """
74
+ ).fetchall():
75
+ metadata = _safe_loads(row["metadata_json"])
76
+ text = self._vector_text_for_node(
77
+ title=row["title"],
78
+ summary=row["summary"] or "",
79
+ metadata=metadata,
80
+ )
81
+ if text:
82
+ yield {
83
+ "item_id": row["id"],
84
+ "item_type": "node",
85
+ "source_node": row["id"],
86
+ "text": text,
87
+ "metadata": {"node_type": row["type"], **metadata},
88
+ }
89
+ if include_chunks:
90
+ for row in conn.execute(
91
+ """
92
+ SELECT c.id, c.source_node AS parent_source_node, c.text, c.metadata_json
93
+ FROM chunks c
94
+ JOIN nodes n ON n.id=c.id
95
+ ORDER BY c.created_at DESC, c.id ASC
96
+ """
97
+ ).fetchall():
98
+ metadata = _safe_loads(row["metadata_json"])
99
+ text = _clean_text(row["text"] or "")
100
+ if text:
101
+ yield {
102
+ "item_id": row["id"],
103
+ "item_type": "chunk",
104
+ "source_node": row["id"],
105
+ "text": text,
106
+ "metadata": {
107
+ **metadata,
108
+ "parent_source_node": row["parent_source_node"],
109
+ },
110
+ }
111
+
112
+ def index_node_incremental(self, node_id: str) -> Dict[str, Any]:
113
+ """Embed/index only ``node_id`` and its chunks (incremental sync).
114
+
115
+ The item construction mirrors :meth:`_iter_vector_source_items` exactly
116
+ (same ids, same ``source_node``/``parent_source_node`` semantics), so
117
+ anything this method indexes is indistinguishable from a full
118
+ :meth:`rebuild_vector_index` pass — and anything it *fails* to index
119
+ stays visible as ``missing``/``stale`` backlog in :meth:`index_status`,
120
+ where a later rebuild picks it up.
121
+
122
+ Never raises: embedding-provider or storage failures are reported as
123
+ ``{"status": "failed", ...}`` so ingestion callers can degrade instead
124
+ of losing an already-persisted write.
125
+ """
126
+ node_id = str(node_id or "").strip()
127
+ started = time.perf_counter()
128
+ summary: Dict[str, Any] = {
129
+ "node_id": node_id,
130
+ "items_total": 0,
131
+ "items_indexed": 0,
132
+ "items_skipped": 0,
133
+ }
134
+ if not node_id:
135
+ return {**summary, "status": "skipped", "detail": "node_id required"}
136
+ try:
137
+ with self._connect() as conn:
138
+ row = conn.execute(
139
+ "SELECT id, type, title, summary, metadata_json FROM nodes WHERE id=?",
140
+ (node_id,),
141
+ ).fetchone()
142
+ if row is None:
143
+ return {**summary, "status": "skipped", "detail": "node not found"}
144
+ items: List[Dict[str, Any]] = []
145
+ if row["type"] != "Chunk":
146
+ metadata = _safe_loads(row["metadata_json"])
147
+ text = self._vector_text_for_node(
148
+ title=row["title"],
149
+ summary=row["summary"] or "",
150
+ metadata=metadata,
151
+ )
152
+ if text:
153
+ items.append(
154
+ {
155
+ "item_id": row["id"],
156
+ "item_type": "node",
157
+ "source_node": row["id"],
158
+ "text": text,
159
+ "metadata": {"node_type": row["type"], **metadata},
160
+ }
161
+ )
162
+ for chunk_row in conn.execute(
163
+ """
164
+ SELECT c.id, c.source_node AS parent_source_node, c.text, c.metadata_json
165
+ FROM chunks c
166
+ JOIN nodes n ON n.id=c.id
167
+ WHERE c.source_node=?
168
+ ORDER BY c.created_at ASC, c.id ASC
169
+ """,
170
+ (node_id,),
171
+ ).fetchall():
172
+ metadata = _safe_loads(chunk_row["metadata_json"])
173
+ text = _clean_text(chunk_row["text"] or "")
174
+ if text:
175
+ items.append(
176
+ {
177
+ "item_id": chunk_row["id"],
178
+ "item_type": "chunk",
179
+ "source_node": chunk_row["id"],
180
+ "text": text,
181
+ "metadata": {
182
+ **metadata,
183
+ "parent_source_node": chunk_row["parent_source_node"],
184
+ },
185
+ }
186
+ )
187
+ indexed = skipped = 0
188
+ for item in items:
189
+ if self._upsert_vector_item(conn, **item):
190
+ indexed += 1
191
+ else:
192
+ skipped += 1
193
+ if indexed and self._embedder_fingerprint_record(conn) is None:
194
+ # First successful vector write establishes the fingerprint;
195
+ # later incremental writes never overwrite it (only a full
196
+ # rebuild may flip it after an embedder swap).
197
+ self._write_embedder_fingerprint(conn)
198
+ summary.update(
199
+ {
200
+ "items_total": len(items),
201
+ "items_indexed": indexed,
202
+ "items_skipped": skipped,
203
+ }
204
+ )
205
+ return {
206
+ **summary,
207
+ "status": "indexed" if indexed else "noop",
208
+ "duration_ms": round((time.perf_counter() - started) * 1000, 2),
209
+ "embedding_model": self._embedding_model.model_id,
210
+ }
211
+ except Exception as exc: # noqa: BLE001 — incremental sync must never raise
212
+ return {
213
+ **summary,
214
+ "status": "failed",
215
+ "detail": str(exc),
216
+ "duration_ms": round((time.perf_counter() - started) * 1000, 2),
217
+ }
218
+
219
+ def rebuild_vector_index(
220
+ self,
221
+ *,
222
+ full: bool = False,
223
+ include_nodes: bool = True,
224
+ include_chunks: bool = True,
225
+ ) -> Dict[str, Any]:
226
+ """Rebuild the derived vector index without mutating graph content."""
227
+ op_id = f"vector-op:{_sha256_text(f'{time.time()}:{os.getpid()}')[:24]}"
228
+ requested_at = _now()
229
+ started = time.perf_counter()
230
+ try:
231
+ with self._connect() as conn:
232
+ conn.execute(
233
+ """
234
+ INSERT INTO vector_index_operations(
235
+ id, operation, status, requested_at, started_at, metadata_json
236
+ )
237
+ VALUES (?, ?, 'running', ?, ?, ?)
238
+ """,
239
+ (
240
+ op_id,
241
+ "rebuild_full" if full else "rebuild_incremental",
242
+ requested_at,
243
+ requested_at,
244
+ _json(
245
+ {
246
+ "include_nodes": include_nodes,
247
+ "include_chunks": include_chunks,
248
+ }
249
+ ),
250
+ ),
251
+ )
252
+ if full:
253
+ filters = []
254
+ if include_nodes:
255
+ filters.append("'node'")
256
+ if include_chunks:
257
+ filters.append("'chunk'")
258
+ if filters:
259
+ conn.execute(
260
+ f"DELETE FROM vector_embeddings WHERE item_type IN ({','.join(filters)})"
261
+ )
262
+ # After a full wipe nothing is current by definition, so the
263
+ # prefetch would only be a wasted scan of a table we just
264
+ # emptied. Incremental is where it pays: it turns "one SELECT
265
+ # per item, nearly all of which say unchanged" into one query.
266
+ known = {} if full else self._vector_text_hashes(conn)
267
+ total = indexed = skipped = 0
268
+ for item in self._iter_vector_source_items(
269
+ conn,
270
+ include_nodes=include_nodes,
271
+ include_chunks=include_chunks,
272
+ ):
273
+ total += 1
274
+ if known.get(item["item_id"]) == _sha256_text(_clean_text(item["text"])):
275
+ skipped += 1
276
+ continue
277
+ if self._upsert_vector_item(conn, **item):
278
+ indexed += 1
279
+ else:
280
+ skipped += 1
281
+ duration_ms = round((time.perf_counter() - started) * 1000, 2)
282
+ conn.execute(
283
+ """
284
+ UPDATE vector_index_operations
285
+ SET status='completed', completed_at=?, items_total=?,
286
+ items_indexed=?, items_skipped=?, metadata_json=?
287
+ WHERE id=?
288
+ """,
289
+ (
290
+ _now(),
291
+ total,
292
+ indexed,
293
+ skipped,
294
+ _json(
295
+ {
296
+ "include_nodes": include_nodes,
297
+ "include_chunks": include_chunks,
298
+ "duration_ms": duration_ms,
299
+ "embedding_model": self._embedding_model.model_id,
300
+ "embedding_dim": self._embedding_model.dim,
301
+ }
302
+ ),
303
+ op_id,
304
+ ),
305
+ )
306
+ # A successful rebuild (re)establishes which embedder built
307
+ # the index — this is the only path that may flip a recorded
308
+ # fingerprint after an embedder swap.
309
+ self._write_embedder_fingerprint(conn)
310
+ return {
311
+ "status": "completed",
312
+ "operation_id": op_id,
313
+ "full": bool(full),
314
+ "items_total": total,
315
+ "items_indexed": indexed,
316
+ "items_skipped": skipped,
317
+ "duration_ms": duration_ms,
318
+ "embedding_model": self._embedding_model.model_id,
319
+ "embedding_dim": self._embedding_model.dim,
320
+ }
321
+ except Exception as exc:
322
+ duration_ms = round((time.perf_counter() - started) * 1000, 2)
323
+ with self._connect() as conn:
324
+ conn.execute(
325
+ """
326
+ INSERT INTO vector_index_operations(
327
+ id, operation, status, requested_at, started_at, completed_at,
328
+ error_message, metadata_json
329
+ )
330
+ VALUES (?, ?, 'failed', ?, ?, ?, ?, ?)
331
+ ON CONFLICT(id) DO UPDATE SET
332
+ status='failed',
333
+ completed_at=excluded.completed_at,
334
+ error_message=excluded.error_message,
335
+ metadata_json=excluded.metadata_json
336
+ """,
337
+ (
338
+ op_id,
339
+ "rebuild_full" if full else "rebuild_incremental",
340
+ requested_at,
341
+ requested_at,
342
+ _now(),
343
+ str(exc),
344
+ _json({"duration_ms": duration_ms}),
345
+ ),
346
+ )
347
+ raise