ltcai 11.2.0 → 11.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (247) hide show
  1. package/README.md +46 -53
  2. package/docs/CHANGELOG.md +61 -0
  3. package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
  4. package/docs/DEVELOPMENT.md +1 -1
  5. package/docs/MULTI_AGENT_RUNTIME.md +1 -1
  6. package/docs/ONBOARDING.md +1 -1
  7. package/docs/OPERATIONS.md +6 -2
  8. package/docs/PERMISSION_MODE.md +1 -1
  9. package/docs/TRUST_MODEL.md +1 -1
  10. package/docs/WHY_LATTICE.md +1 -1
  11. package/docs/kg-schema.md +2 -2
  12. package/docs/v11.3.0_PLAN.md +202 -0
  13. package/docs/v11.4.0_RUST_FOUNDATION_PLAN.md +176 -0
  14. package/lattice_brain/__init__.py +1 -1
  15. package/lattice_brain/graph/_kg_common/__init__.py +287 -0
  16. package/lattice_brain/graph/_kg_common/extraction.py +516 -0
  17. package/lattice_brain/graph/_kg_common/relations.py +161 -0
  18. package/lattice_brain/graph/_kg_common/text.py +479 -0
  19. package/lattice_brain/graph/discovery_index/__init__.py +35 -0
  20. package/lattice_brain/graph/discovery_index/cleanup.py +182 -0
  21. package/lattice_brain/graph/discovery_index/extract.py +137 -0
  22. package/lattice_brain/graph/discovery_index/scan.py +411 -0
  23. package/lattice_brain/graph/discovery_index/upsert.py +495 -0
  24. package/lattice_brain/graph/projection/__init__.py +42 -0
  25. package/lattice_brain/graph/projection/curation.py +500 -0
  26. package/lattice_brain/graph/{projection.py → projection/v2_schema.py} +15 -477
  27. package/lattice_brain/graph/retrieval/__init__.py +54 -0
  28. package/lattice_brain/graph/retrieval/context.py +197 -0
  29. package/lattice_brain/graph/retrieval/graph_view.py +319 -0
  30. package/lattice_brain/graph/retrieval/hybrid.py +488 -0
  31. package/lattice_brain/graph/retrieval/maintenance.py +121 -0
  32. package/lattice_brain/graph/retrieval/signals.py +95 -0
  33. package/lattice_brain/graph/retrieval_vector/__init__.py +42 -0
  34. package/lattice_brain/graph/retrieval_vector/fingerprint.py +97 -0
  35. package/lattice_brain/graph/retrieval_vector/indexing.py +347 -0
  36. package/lattice_brain/graph/retrieval_vector/search.py +560 -0
  37. package/lattice_brain/graph/retrieval_vector/status.py +374 -0
  38. package/lattice_brain/ingestion/__init__.py +130 -0
  39. package/lattice_brain/ingestion/_contract.py +90 -0
  40. package/lattice_brain/ingestion/constants.py +127 -0
  41. package/lattice_brain/ingestion/folder_scan.py +57 -0
  42. package/lattice_brain/ingestion/folders.py +258 -0
  43. package/lattice_brain/ingestion/hashing.py +26 -0
  44. package/lattice_brain/ingestion/jobs_api.py +107 -0
  45. package/lattice_brain/ingestion/models.py +80 -0
  46. package/lattice_brain/ingestion/pipeline.py +486 -0
  47. package/lattice_brain/ingestion/quality.py +209 -0
  48. package/lattice_brain/ingestion/routing.py +295 -0
  49. package/lattice_brain/multimodal/__init__.py +164 -0
  50. package/lattice_brain/multimodal/audio.py +77 -0
  51. package/lattice_brain/multimodal/common.py +118 -0
  52. package/lattice_brain/multimodal/images.py +498 -0
  53. package/lattice_brain/multimodal/ports.py +169 -0
  54. package/lattice_brain/multimodal/video.py +410 -0
  55. package/lattice_brain/portability/__init__.py +90 -0
  56. package/lattice_brain/portability/_contract.py +42 -0
  57. package/lattice_brain/portability/backups.py +338 -0
  58. package/lattice_brain/portability/bundles.py +136 -0
  59. package/lattice_brain/portability/constants.py +93 -0
  60. package/lattice_brain/portability/fsops.py +138 -0
  61. package/lattice_brain/portability/service.py +41 -0
  62. package/lattice_brain/{portability.py → portability/sharing.py} +44 -677
  63. package/lattice_brain/runtime/__init__.py +1 -1
  64. package/lattice_brain/runtime/multi_agent.py +1 -1
  65. package/latticeai/__init__.py +1 -1
  66. package/latticeai/api/chronicle.py +63 -0
  67. package/latticeai/core/agent/__init__.py +93 -0
  68. package/latticeai/core/agent/_contract.py +79 -0
  69. package/latticeai/core/agent/context.py +57 -0
  70. package/latticeai/core/agent/deps.py +125 -0
  71. package/latticeai/core/agent/execution.py +622 -0
  72. package/latticeai/core/agent/planning.py +145 -0
  73. package/latticeai/core/agent/recovery.py +157 -0
  74. package/latticeai/core/agent/runtime.py +210 -0
  75. package/latticeai/core/agent/verification.py +231 -0
  76. package/latticeai/core/embedding_providers/__init__.py +151 -0
  77. package/latticeai/core/embedding_providers/base.py +199 -0
  78. package/latticeai/core/embedding_providers/captions.py +162 -0
  79. package/latticeai/core/embedding_providers/profiles.py +126 -0
  80. package/latticeai/core/embedding_providers/text.py +350 -0
  81. package/latticeai/core/embedding_providers/vision.py +352 -0
  82. package/latticeai/core/file_generation/__init__.py +115 -0
  83. package/latticeai/core/file_generation/bundles.py +76 -0
  84. package/latticeai/core/file_generation/extraction.py +154 -0
  85. package/latticeai/core/file_generation/inference.py +235 -0
  86. package/latticeai/core/file_generation/orchestration.py +152 -0
  87. package/latticeai/core/file_generation/prompting.py +117 -0
  88. package/latticeai/core/file_generation/repair.py +114 -0
  89. package/latticeai/core/file_generation/sanitize.py +61 -0
  90. package/latticeai/core/file_generation/validation.py +201 -0
  91. package/latticeai/core/legacy_compatibility.py +1 -1
  92. package/latticeai/core/marketplace.py +1 -1
  93. package/latticeai/core/messages.py +9 -0
  94. package/latticeai/core/workspace_os_constants.py +1 -1
  95. package/latticeai/integrations/telegram_bot/__init__.py +123 -0
  96. package/latticeai/integrations/telegram_bot/__main__.py +17 -0
  97. package/latticeai/integrations/telegram_bot/config.py +86 -0
  98. package/latticeai/integrations/telegram_bot/dispatch.py +311 -0
  99. package/latticeai/integrations/telegram_bot/flows.py +478 -0
  100. package/latticeai/integrations/telegram_bot/helpers.py +322 -0
  101. package/latticeai/integrations/telegram_bot/screens.py +394 -0
  102. package/latticeai/models/router/__init__.py +88 -0
  103. package/latticeai/models/router/_contract.py +66 -0
  104. package/latticeai/models/router/branding.py +56 -0
  105. package/latticeai/models/router/catalog.py +69 -0
  106. package/latticeai/models/router/documents.py +199 -0
  107. package/latticeai/models/router/errors.py +37 -0
  108. package/latticeai/models/router/generation.py +258 -0
  109. package/latticeai/models/router/loading.py +291 -0
  110. package/latticeai/models/router/local_models.py +85 -0
  111. package/latticeai/models/router/registry.py +147 -0
  112. package/latticeai/runtime/build_phases/__init__.py +82 -0
  113. package/latticeai/runtime/build_phases/features.py +407 -0
  114. package/latticeai/runtime/build_phases/foundation.py +555 -0
  115. package/latticeai/runtime/build_phases/web.py +492 -0
  116. package/latticeai/runtime/runtime_context.py +1 -0
  117. package/latticeai/services/architecture_readiness.py +48 -19
  118. package/latticeai/services/brain_intelligence/__init__.py +58 -0
  119. package/latticeai/services/brain_intelligence/_contract.py +71 -0
  120. package/latticeai/services/brain_intelligence/consistency.py +193 -0
  121. package/latticeai/services/brain_intelligence/constants.py +47 -0
  122. package/latticeai/services/brain_intelligence/digest.py +258 -0
  123. package/latticeai/services/brain_intelligence/health.py +331 -0
  124. package/latticeai/services/brain_intelligence/proposals.py +264 -0
  125. package/latticeai/services/brain_intelligence/sampling.py +84 -0
  126. package/latticeai/services/brain_intelligence/service.py +48 -0
  127. package/latticeai/services/chronicle.py +557 -0
  128. package/latticeai/services/memory_service/__init__.py +52 -0
  129. package/latticeai/services/memory_service/_contract.py +100 -0
  130. package/latticeai/services/memory_service/brief.py +431 -0
  131. package/latticeai/services/memory_service/constants.py +57 -0
  132. package/latticeai/services/memory_service/maintenance.py +138 -0
  133. package/latticeai/services/memory_service/manager.py +186 -0
  134. package/latticeai/services/memory_service/proof.py +136 -0
  135. package/latticeai/services/memory_service/recall.py +225 -0
  136. package/latticeai/services/memory_service/service.py +48 -0
  137. package/latticeai/services/memory_service/stores.py +110 -0
  138. package/latticeai/services/model_runtime/__init__.py +322 -0
  139. package/latticeai/services/model_runtime/cloud.py +87 -0
  140. package/latticeai/services/model_runtime/download.py +282 -0
  141. package/latticeai/services/model_runtime/engines.py +341 -0
  142. package/latticeai/services/model_runtime/loading.py +178 -0
  143. package/latticeai/services/model_runtime/service.py +129 -0
  144. package/latticeai/services/model_runtime/state.py +131 -0
  145. package/latticeai/services/model_runtime/status.py +255 -0
  146. package/latticeai/services/product_readiness.py +15 -7
  147. package/latticeai/setup/wizard/__init__.py +126 -0
  148. package/latticeai/setup/wizard/catalog.py +172 -0
  149. package/latticeai/setup/wizard/detect.py +323 -0
  150. package/latticeai/setup/wizard/install.py +348 -0
  151. package/latticeai/setup/wizard/paths.py +168 -0
  152. package/latticeai/setup/wizard/plans.py +74 -0
  153. package/latticeai/setup/wizard/recommend.py +320 -0
  154. package/package.json +6 -2
  155. package/scripts/bump_version.py +14 -0
  156. package/scripts/capture_release_evidence.mjs +33 -21
  157. package/scripts/check_current_release_docs.mjs +1 -1
  158. package/scripts/check_i18n_namespace_coverage.mjs +41 -4
  159. package/scripts/check_max_file_lines.mjs +102 -0
  160. package/scripts/check_release_evidence_bound.mjs +30 -15
  161. package/scripts/check_screenshot_pixel_delta.py +34 -4
  162. package/scripts/check_server_i18n.mjs +1 -0
  163. package/scripts/generate_rust_parity_fixtures.py +562 -0
  164. package/scripts/lib/mock_server_fingerprint.mjs +94 -0
  165. package/scripts/release_screen_claims.json +31 -2
  166. package/src-tauri/Cargo.lock +361 -3
  167. package/src-tauri/Cargo.toml +6 -1
  168. package/src-tauri/src/backend.rs +349 -0
  169. package/src-tauri/src/folder.rs +33 -0
  170. package/src-tauri/src/main.rs +97 -399
  171. package/src-tauri/tauri.conf.json +1 -1
  172. package/static/app/asset-manifest.json +41 -37
  173. package/static/app/assets/Act-yYpYnn0v.js +1 -0
  174. package/static/app/assets/AdminConsole-DL3Cr5pL.js +1 -0
  175. package/static/app/assets/{Brain-tuhI4sOC.js → Brain-C1HBN0Wf.js} +2 -2
  176. package/static/app/assets/BrainHome-DoXRhUUC.js +2 -0
  177. package/static/app/assets/BrainSignals-6yR6ir5t.js +1 -0
  178. package/static/app/assets/Capture-CFIRsFNE.js +1 -0
  179. package/static/app/assets/Chronicle-BZbEgiwN.js +1 -0
  180. package/static/app/assets/CommandPalette-D2pMxC2I.js +1 -0
  181. package/static/app/assets/Library-DwO3yZST.js +1 -0
  182. package/static/app/assets/{LivingBrain-DBwhto14.js → LivingBrain-Jn1GK0-S.js} +1 -1
  183. package/static/app/assets/ProductFlow-B-w1R4Oo.js +1 -0
  184. package/static/app/assets/ReviewCard-6B27X8Vg.js +3 -0
  185. package/static/app/assets/System-DW8F-2xL.js +1 -0
  186. package/static/app/assets/arrow-left-DXvKg9U6.js +1 -0
  187. package/static/app/assets/{bot-Cia42c2h.js → bot-IM_E_Y12.js} +1 -1
  188. package/static/app/assets/brain-Ci1CkWjM.js +1 -0
  189. package/static/app/assets/{button-2j2Ijzgq.js → button-COwyqfHM.js} +1 -1
  190. package/static/app/assets/circle-check-DfInj-qD.js +1 -0
  191. package/static/app/assets/{circle-pause-BEFeWpVW.js → circle-pause-DEM4A1Y5.js} +1 -1
  192. package/static/app/assets/{circle-play-ujXMcHxl.js → circle-play-C9djDuLd.js} +1 -1
  193. package/static/app/assets/{cpu-k4awryFq.js → cpu-DFdo1gw-.js} +1 -1
  194. package/static/app/assets/{download-DFbLJ_ig.js → download-SnJL6oqk.js} +1 -1
  195. package/static/app/assets/{folder-open-7y_b6xkM.js → folder-open-CqZeDkjE.js} +1 -1
  196. package/static/app/assets/{hard-drive-Bidh02Kr.js → hard-drive-j1jJXYYf.js} +1 -1
  197. package/static/app/assets/{index-DwDl9-8Y.css → index-BLPb5lmE.css} +1 -1
  198. package/static/app/assets/index-_u5iUHDr.js +10 -0
  199. package/static/app/assets/input-B0lPdRQZ.js +1 -0
  200. package/static/app/assets/link-2-CoFbooHS.js +1 -0
  201. package/static/app/assets/{permissionCopy-Bpb83Hx9.js → permissionCopy-BsyLxtao.js} +1 -1
  202. package/static/app/assets/primitives-DEbN-d6p.js +1 -0
  203. package/static/app/assets/search-BybIWPNd.js +1 -0
  204. package/static/app/assets/{share-2-BH1M-WNi.js → share-2-CVtZ_ewX.js} +1 -1
  205. package/static/app/assets/{shield-alert-BlKdBXcG.js → shield-alert-CBi2GNWM.js} +1 -1
  206. package/static/app/assets/{textarea-CCWbUfFB.js → textarea-DNMpB5ih.js} +1 -1
  207. package/static/app/assets/{useFocusTrap-YdHQ7pJ1.js → useFocusTrap-C83t3GXF.js} +1 -1
  208. package/static/app/assets/useMutation-DtbJDoyz.js +1 -0
  209. package/static/app/assets/{useQuery-CXQiwbVT.js → useQuery-Dcp1OChy.js} +1 -1
  210. package/static/app/assets/utils-BlZr7Pd4.js +4 -0
  211. package/static/app/assets/workspace-jJY4RuAV.js +1 -0
  212. package/static/app/index.html +4 -4
  213. package/static/sw.js +1 -1
  214. package/lattice_brain/graph/_kg_common.py +0 -1331
  215. package/lattice_brain/graph/discovery_index.py +0 -1141
  216. package/lattice_brain/graph/retrieval.py +0 -1120
  217. package/lattice_brain/graph/retrieval_vector.py +0 -1293
  218. package/lattice_brain/ingestion.py +0 -1525
  219. package/lattice_brain/multimodal.py +0 -1258
  220. package/latticeai/core/agent.py +0 -1465
  221. package/latticeai/core/embedding_providers.py +0 -1196
  222. package/latticeai/core/file_generation.py +0 -1047
  223. package/latticeai/integrations/telegram_bot.py +0 -1390
  224. package/latticeai/models/router.py +0 -1007
  225. package/latticeai/runtime/build_phases.py +0 -1450
  226. package/latticeai/services/brain_intelligence.py +0 -1083
  227. package/latticeai/services/memory_service.py +0 -1177
  228. package/latticeai/services/model_runtime.py +0 -1281
  229. package/latticeai/setup/wizard.py +0 -1310
  230. package/static/app/assets/Act-AWf0SAKp.js +0 -1
  231. package/static/app/assets/AdminConsole-D0u8Tiyj.js +0 -1
  232. package/static/app/assets/BrainHome-Ts7G_Ila.js +0 -2
  233. package/static/app/assets/BrainSignals-jMYgQ2Ar.js +0 -1
  234. package/static/app/assets/Capture-CqOSzyPr.js +0 -1
  235. package/static/app/assets/CommandPalette-DC0Bzh-I.js +0 -1
  236. package/static/app/assets/Library-CX-bbhmK.js +0 -1
  237. package/static/app/assets/ProductFlow-BHA2cfKI.js +0 -1
  238. package/static/app/assets/ReviewCard-BUhCKRNM.js +0 -3
  239. package/static/app/assets/System-Bu2t5hn1.js +0 -1
  240. package/static/app/assets/arrow-left-Dzwa5zRb.js +0 -1
  241. package/static/app/assets/brain-DJMoqrwx.js +0 -1
  242. package/static/app/assets/index-BpYkzcVm.js +0 -10
  243. package/static/app/assets/input-DSlJJxRs.js +0 -1
  244. package/static/app/assets/primitives-BCx6TvfG.js +0 -1
  245. package/static/app/assets/search-Cgy8cCFJ.js +0 -1
  246. package/static/app/assets/utils-zqPZJxdx.js +0 -4
  247. package/static/app/assets/workspace-DXTihhfU.js +0 -1
@@ -0,0 +1,562 @@
1
+ #!/usr/bin/env python3
2
+ """Build the committed Python↔Rust retrieval parity fixture.
3
+
4
+ ``rust/lattice-retrieval`` is a port, and a port is only worth having if
5
+ something keeps proving it is still one. This script is the Python half of that
6
+ proof: it builds a small, fully deterministic Brain with the **real** write path
7
+ (``KnowledgeGraphStore._upsert_node`` / ``_upsert_chunk`` / ``_upsert_edge``,
8
+ the real hash embedder, the real v2 projection and trigram FTS index), then runs
9
+ the real ``hybrid_search`` / ``search`` / ``vector_search`` over it and writes
10
+ their answers to ``rust/fixtures/golden/``.
11
+
12
+ Two consumers read what it writes:
13
+
14
+ * ``tests/unit/test_rust_parity_contract.py`` re-runs the Python engines against
15
+ the committed database and asserts the goldens still hold — so a change to
16
+ Python retrieval semantics fails loudly instead of silently invalidating the
17
+ contract the Rust side is pinned to;
18
+ * ``rust/lattice-retrieval/tests/parity.rs`` runs the Rust port against the same
19
+ database and the same goldens.
20
+
21
+ Determinism is the whole design constraint:
22
+
23
+ * every timestamp is written by the real code and then **backdated** to a fixed
24
+ value, so nothing in the fixture depends on when it was generated;
25
+ * ``hybrid_search``'s recency decay calls ``datetime.now()``, so the clock is
26
+ frozen at :data:`FROZEN_NOW` (recorded in the manifest for the Rust side);
27
+ * LLM concept extraction is forced off, so ``_topic_candidates`` always takes
28
+ the rule-based path a port can reproduce;
29
+ * every environment knob the retrieval stack reads is pinned to its default.
30
+
31
+ Usage::
32
+
33
+ .venv/bin/python scripts/generate_rust_parity_fixtures.py
34
+ """
35
+
36
+ from __future__ import annotations
37
+
38
+ import json
39
+ import os
40
+ import shutil
41
+ import sqlite3
42
+ import sys
43
+ from contextlib import contextmanager
44
+ from datetime import datetime
45
+ from pathlib import Path
46
+ from typing import Any, Callable, Dict, Iterator, List, Optional, Tuple
47
+
48
+ REPO_ROOT = Path(__file__).resolve().parents[1]
49
+ if str(REPO_ROOT) not in sys.path:
50
+ sys.path.insert(0, str(REPO_ROOT))
51
+
52
+ FIXTURE_DIR = REPO_ROOT / "rust" / "fixtures"
53
+ GOLDEN_DIR = FIXTURE_DIR / "golden"
54
+ STORE_PATH = FIXTURE_DIR / "parity_store.sqlite"
55
+
56
+ #: The wall clock ``hybrid_search`` sees. Recency decay is a function of "now",
57
+ #: so a golden generated against a moving clock is not a golden.
58
+ FROZEN_NOW = "2026-08-01T12:00:00"
59
+
60
+ #: Every environment variable the ported path reads, pinned to the default
61
+ #: configuration the port targets (brute backend, RRF off, graph expansion off,
62
+ #: cross-encoder rerank off, rewrite on).
63
+ PINNED_ENV: Dict[str, str] = {
64
+ "LATTICEAI_VECTOR_DIM": "384",
65
+ "LATTICEAI_VECTOR_INDEX": "brute",
66
+ "LATTICEAI_VECTOR_MAX_CANDIDATES": "10000",
67
+ "LATTICEAI_KG_READ_V2": "1",
68
+ "LATTICEAI_FUSION_WEIGHTS": "",
69
+ "LATTICEAI_FUSION_STRATEGY": "",
70
+ "LATTICEAI_FUSION_RRF": "",
71
+ "LATTICEAI_GRAPH_EXPANSION": "",
72
+ "LATTICEAI_CROSS_ENCODER_RERANK": "",
73
+ "LATTICEAI_QUERY_REWRITE": "",
74
+ "LATTICEAI_LLM_EXTRACTION": "false",
75
+ }
76
+
77
+ WS_ALPHA = "ws-alpha"
78
+ WS_BETA = "ws-beta"
79
+
80
+ # ── the corpus ───────────────────────────────────────────────────────────────
81
+ # (node_id, type, title, summary, metadata, workspace_id, updated_at)
82
+ #
83
+ # Shaped on purpose:
84
+ # * every type in ``search()``'s fixed ``type_boost`` set appears, and so do
85
+ # types outside it, so the boost is observable;
86
+ # * titles/summaries are half Korean and half English, because the tokenizer,
87
+ # the query classifier and the concept extractor all branch on script;
88
+ # * the ``tie:`` block is five rows sharing one timestamp and one type with
89
+ # nothing to match, which pins the (hits, type_boost, updated_at) → id ASC
90
+ # tie-break that both engines have to reproduce;
91
+ # * two workspaces plus NULL-workspace legacy rows cover all three scoping
92
+ # answers (no scoping / empty set / a specific workspace).
93
+ NODES: List[Tuple[str, str, str, str, Dict[str, Any], Optional[str], str]] = [
94
+ ("dec:fusion-alpha", "Decision", "Hybrid retrieval fusion stays alpha weighted",
95
+ "We decided the ranking keeps alpha fusion: lexical rank plus max normalized vector score.",
96
+ {"category": "retrieval", "owner": "jiwon"}, WS_ALPHA, "2026-07-20T09:00:00"),
97
+ ("dec:rust-foundation", "Decision", "Rust 기반 검색 이관 결정",
98
+ "검색 랭킹을 Rust로 이관하기로 결정했습니다. 회의 결과 패리티 증명을 먼저 만듭니다.",
99
+ {"category": "platform"}, WS_ALPHA, "2026-07-18T14:30:00"),
100
+ ("task:parity-harness", "Task", "Build the retrieval parity harness",
101
+ "Generate a golden fixture so the Rust ranking can be compared against the Python ranking.",
102
+ {"status": "open"}, WS_ALPHA, "2026-07-15T11:00:00"),
103
+ ("task:onboarding-checklist", "Task", "온보딩 체크리스트 정리",
104
+ "새로운 사용자가 처음 다섯 걸음을 마칠 수 있도록 온보딩 체크리스트를 정리합니다.",
105
+ {"status": "open"}, WS_BETA, "2026-06-30T08:20:00"),
106
+ ("file:ranking-notes", "File", "ranking-notes.md",
107
+ "Notes on ranking: lexical channel scores one over rank, vector channel is max normalized.",
108
+ {"filename": "ranking-notes.md", "ext": ".md"}, WS_ALPHA, "2026-07-10T16:45:00"),
109
+ ("doc:handbook", "Document", "Lattice AI 사용 안내서",
110
+ "제품 안내서입니다. 검색, 회의 기록, 온보딩 체크리스트를 한 곳에서 설명합니다.",
111
+ {"filename": "handbook.pdf", "ext": ".pdf", "source_node": "doc:handbook"},
112
+ WS_ALPHA, "2026-06-12T10:00:00"),
113
+ ("doc:retrieval-spec", "Document", "Retrieval specification",
114
+ "The retrieval specification describes the lexical channel, the vector channel and their fusion.",
115
+ {"filename": "retrieval-spec.pdf", "ext": ".pdf"}, WS_BETA, "2026-05-28T13:15:00"),
116
+ ("code:hybrid-search", "CodeFile", "hybrid.py",
117
+ "def hybrid_search(query): the ranking pipeline lives here and calls vector_search().",
118
+ {"filename": "hybrid.py", "ext": ".py"}, WS_ALPHA, "2026-07-08T09:30:00"),
119
+ ("code:build-failure", "CodeFile", "build_pipeline.py",
120
+ "빌드 실패 원인을 남겨두는 파일입니다. 컴파일 오류가 나면 여기부터 확인합니다.",
121
+ {"filename": "build_pipeline.py", "ext": ".py"}, WS_BETA, "2026-04-22T18:05:00"),
122
+ ("sheet:metrics", "Spreadsheet", "retrieval-metrics.xlsx",
123
+ "Recall and precision per query class for the ranking experiments.",
124
+ {"filename": "retrieval-metrics.xlsx", "ext": ".xlsx"}, WS_ALPHA, "2026-05-06T09:45:00"),
125
+ ("deck:review", "SlideDeck", "분기 리뷰 발표자료",
126
+ "지난주 분기 리뷰에서 사용한 발표자료입니다. 검색 품질 지표를 담고 있습니다.",
127
+ {"filename": "quarterly-review.pptx", "ext": ".pptx"}, WS_ALPHA, "2026-07-02T15:00:00"),
128
+ ("img:whiteboard", "Image", "whiteboard-retrieval.png",
129
+ "Whiteboard photo from the retrieval design session.",
130
+ {"filename": "whiteboard-retrieval.png", "ext": ".png"}, WS_BETA, "2026-03-19T11:11:00"),
131
+ ("img:ocr-notes", "ImageText", "화이트보드 OCR 텍스트",
132
+ "회의 결정 사항: 랭킹은 alpha 융합을 유지한다.",
133
+ {"filename": "whiteboard-retrieval.png", "ocr": True}, WS_BETA, "2026-03-19T11:12:00"),
134
+ ("audio:standup", "Audio", "standup-2026-07-13.m4a",
135
+ "Standup recording where the parity harness was assigned.",
136
+ {"filename": "standup-2026-07-13.m4a", "ext": ".m4a"}, WS_ALPHA, "2026-07-13T09:05:00"),
137
+ ("page:onboarding", "Page", "온보딩 첫 다섯 걸음",
138
+ "온보딩 페이지입니다. 체크리스트와 안내 문구를 담습니다.", {}, None, "2026-06-01T12:00:00"),
139
+ ("slide:fusion", "Slide", "Fusion slide",
140
+ "One slide explaining alpha fusion of the lexical and vector channels.",
141
+ {}, None, "2026-05-14T12:00:00"),
142
+ ("concept:retrieval", "Concept", "Retrieval",
143
+ "Retrieval is the act of finding the right memory for a question.",
144
+ {}, WS_ALPHA, "2026-04-02T10:00:00"),
145
+ ("concept:ranking", "Concept", "Ranking",
146
+ "Ranking orders candidates so the best answer is first.", {}, WS_BETA, "2026-04-03T10:00:00"),
147
+ ("person:jiwon", "Person", "김지원 님",
148
+ "검색 품질 담당자입니다. 온보딩 개선도 함께 맡고 있습니다.",
149
+ {"role": "owner"}, WS_ALPHA, "2026-04-11T10:00:00"),
150
+ ("person:minseo", "Person", "박민서 님",
151
+ "빌드 파이프라인 담당자입니다.", {"role": "reviewer"}, WS_BETA, "2026-04-12T10:00:00"),
152
+ ("meeting:weekly", "Meeting", "주간 회의 2026-07-14",
153
+ "지난주 주간 회의 기록입니다. 검색 랭킹과 온보딩을 논의했습니다.",
154
+ {}, WS_ALPHA, "2026-07-14T10:00:00"),
155
+ ("meeting:kickoff", "Meeting", "Kickoff meeting",
156
+ "Kickoff meeting for the Rust foundation work.", {}, WS_BETA, "2026-02-09T10:00:00"),
157
+ ("chat:ranking", "Chat", "랭킹 관련 대화",
158
+ "랭킹이 왜 이렇게 나오는지 물어본 대화입니다.",
159
+ {"conversation_id": "conv-1"}, WS_ALPHA, "2026-07-05T21:00:00"),
160
+ ("source:repo", "Source", "github.com/lattice/ai",
161
+ "Source repository for the product.", {"source": "git"}, WS_ALPHA, "2026-02-20T10:00:00"),
162
+ ("repo:lattice", "Repository", "lattice-ai",
163
+ "The monorepo holding the Python worker and the Rust workspace.",
164
+ {}, WS_ALPHA, "2026-02-21T10:00:00"),
165
+ ("org:lattice", "Organization", "Lattice", "The organization.", {}, None, "2026-02-22T10:00:00"),
166
+ ("workflow:nightly", "Workflow", "Nightly reindex",
167
+ "A workflow that reindexes the vector store overnight.", {}, WS_BETA, "2026-03-02T10:00:00"),
168
+ ("agent:librarian", "Agent", "Librarian agent",
169
+ "An agent that files new documents into the graph.", {}, WS_BETA, "2026-03-03T10:00:00"),
170
+ ("error:timeout", "Error", "Search timeout",
171
+ "An error where the ranking took longer than the request budget.",
172
+ {}, WS_ALPHA, "2026-03-04T10:00:00"),
173
+ ("feature:command-palette", "Feature", "Command palette",
174
+ "The palette that opens search from anywhere.", {}, WS_ALPHA, "2026-03-05T10:00:00"),
175
+ ("topic:quality", "Topic", "검색 품질",
176
+ "검색 품질에 대한 주제입니다.", {}, WS_ALPHA, "2026-03-06T10:00:00"),
177
+ # ── the tie block: identical type, identical timestamp, nothing to match ──
178
+ ("tie:a", "Concept", "Tie candidate A", "", {}, WS_ALPHA, "2026-02-01T00:00:00"),
179
+ ("tie:b", "Concept", "Tie candidate B", "", {}, WS_ALPHA, "2026-02-01T00:00:00"),
180
+ ("tie:c", "Concept", "Tie candidate C", "", {}, WS_BETA, "2026-02-01T00:00:00"),
181
+ ("tie:d", "Concept", "Tie candidate D", "", {}, None, "2026-02-01T00:00:00"),
182
+ ("tie:e", "Concept", "Tie candidate E", "", {}, None, "2026-02-01T00:00:00"),
183
+ # ── boosted twins on the same timestamp: type_boost breaks nothing here,
184
+ # id ASC does, and that is the assertion.
185
+ ("twin:doc-a", "Document", "Twin document A", "동일한 시각의 문서 A", {}, WS_ALPHA,
186
+ "2026-02-05T00:00:00"),
187
+ ("twin:doc-b", "Document", "Twin document B", "동일한 시각의 문서 B", {}, WS_ALPHA,
188
+ "2026-02-05T00:00:00"),
189
+ ("twin:concept-a", "Concept", "Twin concept A", "동일한 시각의 개념 A", {}, WS_ALPHA,
190
+ "2026-02-05T00:00:00"),
191
+ ("legacy:global-note", "Document", "Legacy global note",
192
+ "A legacy row with no workspace, visible only with include_legacy_global.",
193
+ {}, None, "2026-06-20T10:00:00"),
194
+ ]
195
+
196
+ # (chunk_id, parent_node_id, index, node_title, text, chunk_fields, workspace, updated_at)
197
+ #
198
+ # The shape mirrors ``KnowledgeGraphIngestMixin``: every chunk is BOTH a
199
+ # ``Chunk`` node (so the lexical lane can match it and workspace scoping applies)
200
+ # and a ``chunks`` row with its own embedding (so the vector lane returns it and
201
+ # has to roll it up to its parent). Getting that duality wrong is the whole
202
+ # reason chunk-heavy queries are in the query set.
203
+ CHUNKS: List[Tuple[str, str, int, str, str, Dict[str, Any], Optional[str], str]] = [
204
+ ("chunk:handbook:1", "doc:handbook", 0, "handbook.pdf chunk 1",
205
+ "온보딩 체크리스트: 첫째, 폴더를 연결합니다. 둘째, 질문을 합니다. 셋째, 근거를 확인합니다.",
206
+ {"heading_path": "안내서 > 온보딩", "page": 3, "page_end": 4, "start_char": 0},
207
+ WS_ALPHA, "2026-06-12T10:00:01"),
208
+ ("chunk:handbook:2", "doc:handbook", 1, "handbook.pdf chunk 2",
209
+ "검색 화면에서는 회의 결정 사항을 한 번에 찾을 수 있습니다.",
210
+ {"heading_path": "안내서 > 검색", "page": 7, "start_char": 1200},
211
+ WS_ALPHA, "2026-06-12T10:00:02"),
212
+ ("chunk:spec:1", "doc:retrieval-spec", 0, "retrieval-spec.pdf chunk 1",
213
+ "The lexical channel scores one over rank. The vector channel is max normalized before fusion.",
214
+ {"heading_path": "Retrieval > Fusion", "page": 2, "start_char": 0},
215
+ WS_BETA, "2026-05-28T13:15:01"),
216
+ ("chunk:spec:2", "doc:retrieval-spec", 1, "retrieval-spec.pdf chunk 2",
217
+ "Ranking ties are broken by node id ascending, which keeps the answer stable across runs.",
218
+ {"start_char": 900}, WS_BETA, "2026-05-28T13:15:02"),
219
+ ]
220
+
221
+ # (from, to, type)
222
+ EDGES: List[Tuple[str, str, str]] = [
223
+ ("dec:fusion-alpha", "concept:retrieval", "mentions"),
224
+ ("dec:fusion-alpha", "concept:ranking", "mentions"),
225
+ ("task:parity-harness", "dec:rust-foundation", "relates_to"),
226
+ ("doc:handbook", "page:onboarding", "contains"),
227
+ ("meeting:weekly", "dec:fusion-alpha", "discusses"),
228
+ ("person:jiwon", "task:parity-harness", "owns"),
229
+ ]
230
+
231
+ # ── the query set ────────────────────────────────────────────────────────────
232
+ # ``allowed`` is None (no scoping), [] (a caller who may read nothing) or a list
233
+ # of workspace ids; ``legacy`` is ``include_legacy_global``.
234
+ QUERIES: List[Dict[str, Any]] = [
235
+ {"key": "en_fact", "query": "hybrid retrieval ranking"},
236
+ {"key": "ko_fact", "query": "회의 결정 사항"},
237
+ {"key": "en_code", "query": "vector_search() returns"},
238
+ {"key": "ko_code", "query": "빌드 실패 원인"},
239
+ {"key": "en_person", "query": "who owns the onboarding checklist"},
240
+ {"key": "ko_person", "query": "담당자 누구"},
241
+ {"key": "en_recency", "query": "recent decisions last week"},
242
+ {"key": "ko_recency", "query": "지난주 회의 기록"},
243
+ {"key": "en_filler", "query": " what is the retrieval specification please "},
244
+ {"key": "ko_filler", "query": "온보딩 체크리스트 좀 알려줘"},
245
+ {"key": "short_query", "query": "ai"},
246
+ {"key": "no_hit", "query": "zzqq wumpus nonsense"},
247
+ {"key": "tie_heavy", "query": "Tie candidate"},
248
+ {"key": "chunk_heavy", "query": "온보딩 체크리스트"},
249
+ {"key": "quoted", "query": 'said "retrieval" twice'},
250
+ {"key": "empty_query", "query": ""},
251
+ {"key": "ws_empty", "query": "hybrid retrieval ranking", "allowed": []},
252
+ {"key": "ws_empty_legacy", "query": "회의 결정 사항", "allowed": [], "legacy": True},
253
+ {"key": "ws_alpha", "query": "hybrid retrieval ranking", "allowed": [WS_ALPHA]},
254
+ {"key": "ws_alpha_legacy", "query": "회의 결정 사항", "allowed": [WS_ALPHA], "legacy": True},
255
+ {"key": "ws_beta", "query": "온보딩 체크리스트", "allowed": [WS_BETA]},
256
+ # A vector floor nothing clears: the lane returns zero rows, which is the
257
+ # only way to reach the stale-embedder probe and the lexical-only fusion
258
+ # label without breaking the store.
259
+ {"key": "min_vector_floor", "query": "hybrid retrieval ranking", "min_vector": 0.95},
260
+ # A small top_k so the rerank window (top_k * 2) is narrower than the
261
+ # candidate list and the cut is observable.
262
+ {"key": "top_k_small", "query": "hybrid retrieval ranking", "top_k": 3},
263
+ # An explicitly pinned alpha: no policy, so no class, no rewrite and — on a
264
+ # query that would otherwise be recency-classed — no age decay either.
265
+ {"key": "alpha_pinned", "query": "지난주 회의 기록", "alpha": 0.2},
266
+ # A limit below the FTS hit count, so `ORDER BY rank LIMIT ?` decides which
267
+ # rows exist at all. This is the one place bm25 ordering is observable
268
+ # (`search()` re-sorts by id afterwards), and therefore the one place a
269
+ # SQLite version difference between the two runtimes could show up.
270
+ {"key": "fts_rank_cut", "query": "Tie candidate", "limit": 2, "top_k": 2},
271
+ ]
272
+
273
+ #: Texts whose tokenizer output, hash pairs and full vectors pin the Rust
274
+ #: embedding port bit-for-bit (Korean, English, mixed, symbols, digits).
275
+ EMBEDDING_TEXTS: List[str] = [
276
+ "hybrid retrieval ranking",
277
+ "회의 결정 사항",
278
+ "온보딩 체크리스트 v2",
279
+ "vector_search() returns a dict",
280
+ "!!! ??? ...",
281
+ "MixedCase and snake_case and kebab-case",
282
+ "2026-07-20T09:00:00",
283
+ "a",
284
+ ]
285
+
286
+ #: Values that separate CPython's round-half-even from naive scaling, plus the
287
+ #: shapes real fusion arithmetic produces.
288
+ ROUNDING_VALUES: List[float] = [
289
+ 0.0, 1.0, 5e-07, 1.5e-06, 2.5e-06, 2.6535895, 1 / 3, 1 / 7,
290
+ 0.1234565, 0.1234575, 0.9999999999, 123456.7890625, -0.0000005,
291
+ 0.6 * 0.5 + 0.4 * (1 / 3), 0.35 * 0.25 + 0.65 * (1 / 7), 0.5 + 0.5 * 0.7071067811865476,
292
+ ]
293
+
294
+
295
+ @contextmanager
296
+ def pinned_environment() -> Iterator[None]:
297
+ """Pin every env knob the retrieval stack reads, then restore."""
298
+ previous = {key: os.environ.get(key) for key in PINNED_ENV}
299
+ os.environ.update(PINNED_ENV)
300
+ try:
301
+ yield
302
+ finally:
303
+ for key, value in previous.items():
304
+ if value is None:
305
+ os.environ.pop(key, None)
306
+ else:
307
+ os.environ[key] = value
308
+
309
+
310
+ @contextmanager
311
+ def frozen_clock() -> Iterator[None]:
312
+ """Freeze ``hybrid_search``'s ``datetime.now()`` at :data:`FROZEN_NOW`.
313
+
314
+ Only the recency-class age decay reads the clock, and it reads it through
315
+ the ``datetime`` name that ``hybrid.py`` imported — so rebinding that name
316
+ is the whole patch.
317
+ """
318
+ from lattice_brain.graph.retrieval import hybrid as hybrid_module
319
+
320
+ frozen = datetime.fromisoformat(FROZEN_NOW)
321
+
322
+ class _FrozenDatetime(datetime):
323
+ @classmethod
324
+ def now(cls, tz=None): # noqa: ARG003 — mirrors datetime.now's signature
325
+ return frozen
326
+
327
+ original = hybrid_module.datetime
328
+ hybrid_module.datetime = _FrozenDatetime
329
+ try:
330
+ yield
331
+ finally:
332
+ hybrid_module.datetime = original
333
+
334
+
335
+ @contextmanager
336
+ def rules_only_extraction() -> Iterator[None]:
337
+ """Force ``_topic_candidates`` down its rule-based path.
338
+
339
+ The LLM path needs a bound router, which no fixture run has — but "no router
340
+ happens to be bound" is an accident, and this contract cannot rest on one.
341
+ """
342
+ from lattice_brain.graph._kg_common import extraction
343
+
344
+ original = extraction.ENABLE_LLM_EXTRACTION
345
+ extraction.ENABLE_LLM_EXTRACTION = False
346
+ try:
347
+ yield
348
+ finally:
349
+ extraction.ENABLE_LLM_EXTRACTION = original
350
+
351
+
352
+ def open_store(db_path: Path):
353
+ """A ``KnowledgeGraphStore`` over ``db_path`` (blobs beside it)."""
354
+ from lattice_brain.graph import KnowledgeGraphStore
355
+
356
+ return KnowledgeGraphStore(Path(db_path), Path(db_path).parent / "blobs")
357
+
358
+
359
+ def _backdate(conn: sqlite3.Connection, node_id: str, stamp: str) -> None:
360
+ conn.execute("UPDATE nodes SET created_at=?, updated_at=? WHERE id=?", (stamp, stamp, node_id))
361
+ conn.execute(
362
+ "UPDATE nodes_v2 SET created_at=?, updated_at=? WHERE id=?", (stamp, stamp, node_id)
363
+ )
364
+
365
+
366
+ def build_store(db_path: Path) -> None:
367
+ """Create the fixture database from scratch with the real write path."""
368
+ for sibling in (db_path, Path(f"{db_path}-wal"), Path(f"{db_path}-shm")):
369
+ if sibling.exists():
370
+ sibling.unlink()
371
+ blob_dir = db_path.parent / "blobs"
372
+ if blob_dir.exists():
373
+ shutil.rmtree(blob_dir)
374
+
375
+ store = open_store(db_path)
376
+ with store._connect() as conn:
377
+ for node_id, node_type, title, summary, metadata, workspace, stamp in NODES:
378
+ store._upsert_node(
379
+ conn, node_id, node_type, title, summary, metadata, workspace_id=workspace
380
+ )
381
+ _backdate(conn, node_id, stamp)
382
+ for chunk_id, parent, index, title, text, fields, workspace, stamp in CHUNKS:
383
+ metadata = {"index": index, "source_node": parent, **fields}
384
+ store._upsert_node(
385
+ conn, chunk_id, "Chunk", title, text[:500], metadata, workspace_id=workspace
386
+ )
387
+ store._upsert_chunk(
388
+ conn, chunk_id=chunk_id, source_node=parent, text=text, metadata=metadata
389
+ )
390
+ _backdate(conn, chunk_id, stamp)
391
+ for from_node, to_node, edge_type in EDGES:
392
+ store._upsert_edge(conn, from_node, to_node, edge_type, 1.0, {})
393
+ conn.execute(
394
+ "UPDATE edges SET created_at=? WHERE from_node=? AND to_node=?",
395
+ ("2026-07-01T00:00:00", from_node, to_node),
396
+ )
397
+ # ``indexed_at`` decides the candidate scan order (and, when the cap
398
+ # bites, which candidates exist at all), so it is assigned explicitly
399
+ # rather than inherited from the clock.
400
+ item_ids = [row["item_id"] for row in conn.execute(
401
+ "SELECT item_id FROM vector_embeddings ORDER BY item_id ASC"
402
+ ).fetchall()]
403
+ for seq, item_id in enumerate(item_ids):
404
+ stamp = f"2026-03-01T{seq // 60:02d}:{seq % 60:02d}:00"
405
+ conn.execute(
406
+ "UPDATE vector_embeddings SET indexed_at=? WHERE item_id=?", (stamp, item_id)
407
+ )
408
+ store.record_embedder_fingerprint()
409
+
410
+ # Leave one self-contained file behind: checkpoint the WAL, drop back to a
411
+ # rollback journal so no -wal/-shm sidecar has to be committed, and compact.
412
+ with sqlite3.connect(str(db_path)) as conn:
413
+ conn.execute("PRAGMA wal_checkpoint(TRUNCATE)")
414
+ conn.execute("PRAGMA journal_mode=DELETE")
415
+ with sqlite3.connect(str(db_path)) as conn:
416
+ conn.execute("VACUUM")
417
+ conn.close()
418
+ for sibling in (Path(f"{db_path}-wal"), Path(f"{db_path}-shm")):
419
+ if sibling.exists():
420
+ sibling.unlink()
421
+ if blob_dir.exists():
422
+ shutil.rmtree(blob_dir)
423
+
424
+
425
+ def _allowed(spec: Dict[str, Any]):
426
+ allowed = spec.get("allowed")
427
+ return None if allowed is None else set(allowed)
428
+
429
+
430
+ ENGINES: Dict[str, Callable[[Any, Dict[str, Any]], Dict[str, Any]]] = {
431
+ "hybrid": lambda store, spec: store.hybrid_search(
432
+ spec["query"],
433
+ top_k=spec.get("top_k", 20),
434
+ alpha=spec.get("alpha"),
435
+ allowed_workspaces=_allowed(spec),
436
+ include_legacy_global=spec.get("legacy", False),
437
+ min_vector_score=spec.get("min_vector", 0.0),
438
+ ),
439
+ "keyword": lambda store, spec: store.search(
440
+ spec["query"],
441
+ spec.get("limit", 30),
442
+ allowed_workspaces=_allowed(spec),
443
+ include_legacy_global=spec.get("legacy", False),
444
+ ),
445
+ "vector": lambda store, spec: store.vector_search(
446
+ spec["query"], limit=spec.get("limit", 30), min_score=spec.get("min_score", 0.0)
447
+ ),
448
+ }
449
+
450
+
451
+ def run_engine(store, engine: str, spec: Dict[str, Any]) -> Dict[str, Any]:
452
+ """Run one engine for one query spec under the frozen clock."""
453
+ with frozen_clock(), rules_only_extraction():
454
+ return ENGINES[engine](store, spec)
455
+
456
+
457
+ def golden_path(engine: str, key: str) -> Path:
458
+ return GOLDEN_DIR / f"{engine}__{key}.json"
459
+
460
+
461
+ def golden_payload(engine: str, spec: Dict[str, Any], result: Dict[str, Any]) -> Dict[str, Any]:
462
+ return {
463
+ "engine": engine,
464
+ "key": spec["key"],
465
+ "query": spec["query"],
466
+ "params": {
467
+ "top_k": spec.get("top_k", 20),
468
+ "alpha": spec.get("alpha"),
469
+ "limit": spec.get("limit", 30),
470
+ "min_score": spec.get("min_score", 0.0),
471
+ "min_vector_score": spec.get("min_vector", 0.0),
472
+ "allowed_workspaces": spec.get("allowed"),
473
+ "include_legacy_global": spec.get("legacy", False),
474
+ },
475
+ "result": result,
476
+ }
477
+
478
+
479
+ def _dump(path: Path, payload: Any) -> None:
480
+ path.write_text(
481
+ json.dumps(payload, ensure_ascii=False, sort_keys=True, indent=2) + "\n",
482
+ encoding="utf-8",
483
+ )
484
+
485
+
486
+ def embeddings_golden() -> Dict[str, Any]:
487
+ """Tokenizer output, hash pairs and full vectors for the pinned texts."""
488
+ from lattice_brain.embeddings import LocalEmbeddingModel, _hash_to_index, _tokenize
489
+
490
+ model = LocalEmbeddingModel()
491
+ cases = []
492
+ for text in EMBEDDING_TEXTS:
493
+ features = _tokenize(text)
494
+ cases.append(
495
+ {
496
+ "text": text,
497
+ "features": features,
498
+ "hashes": [
499
+ {"feature": feature, "index": index, "sign": sign}
500
+ for feature, (index, sign) in (
501
+ (feature, _hash_to_index(feature, model.dim))
502
+ for feature in features[:8]
503
+ )
504
+ ],
505
+ "vector": model.embed(text),
506
+ "encoded_hex": model.encode(model.embed(text)).hex(),
507
+ }
508
+ )
509
+ return {"model_id": model.model_id, "dim": model.dim, "cases": cases}
510
+
511
+
512
+ def rounding_golden() -> List[Dict[str, float]]:
513
+ """``round(x, 6)`` for values where the tie rule is observable."""
514
+ return [{"input": value, "expected": round(value, 6)} for value in ROUNDING_VALUES]
515
+
516
+
517
+ def manifest() -> Dict[str, Any]:
518
+ from lattice_brain.embeddings import LocalEmbeddingModel
519
+
520
+ model = LocalEmbeddingModel()
521
+ return {
522
+ "frozen_now": FROZEN_NOW,
523
+ "store": STORE_PATH.name,
524
+ "embedding_model": model.model_id,
525
+ "embedding_dim": model.dim,
526
+ "engines": sorted(ENGINES),
527
+ "queries": QUERIES,
528
+ "pinned_env": PINNED_ENV,
529
+ }
530
+
531
+
532
+ def main() -> int:
533
+ FIXTURE_DIR.mkdir(parents=True, exist_ok=True)
534
+ if GOLDEN_DIR.exists():
535
+ shutil.rmtree(GOLDEN_DIR)
536
+ GOLDEN_DIR.mkdir(parents=True, exist_ok=True)
537
+ with pinned_environment():
538
+ build_store(STORE_PATH)
539
+ store = open_store(STORE_PATH)
540
+ written = 0
541
+ for spec in QUERIES:
542
+ for engine in sorted(ENGINES):
543
+ result = run_engine(store, engine, spec)
544
+ _dump(golden_path(engine, spec["key"]), golden_payload(engine, spec, result))
545
+ written += 1
546
+ _dump(GOLDEN_DIR / "embeddings_golden.json", embeddings_golden())
547
+ _dump(GOLDEN_DIR / "rounding_golden.json", rounding_golden())
548
+ _dump(GOLDEN_DIR / "manifest.json", manifest())
549
+ # ``KnowledgeGraphStore.__init__`` creates its blob directory eagerly; the
550
+ # fixture has no blobs, and an empty directory beside a committed artefact is
551
+ # just a thing for the next reader to wonder about.
552
+ blob_dir = STORE_PATH.parent / "blobs"
553
+ if blob_dir.is_dir() and not any(blob_dir.iterdir()):
554
+ blob_dir.rmdir()
555
+ size_kb = STORE_PATH.stat().st_size / 1024
556
+ print(f"store: {STORE_PATH.relative_to(REPO_ROOT)} ({size_kb:.1f} KiB)")
557
+ print(f"golden: {written} engine files + embeddings + rounding + manifest")
558
+ return 0
559
+
560
+
561
+ if __name__ == "__main__":
562
+ raise SystemExit(main())
@@ -0,0 +1,94 @@
1
+ /**
2
+ * Fingerprint of the release-capture mock API surface.
3
+ *
4
+ * Release screenshots are shot against the visual mock server, so the evidence
5
+ * is only trustworthy while that server still returns the payloads it returned
6
+ * during capture. `capture_release_evidence.mjs` records this fingerprint in
7
+ * SCREENSHOT_INDEX.md and `check_release_evidence_bound.mjs` re-computes it on
8
+ * every lint; a mismatch means the mock changed after capture.
9
+ *
10
+ * v11.3.0: `tests/visual/mock_server.cjs` became a thin entry that composes the
11
+ * route modules under `tests/visual/mock_server/`. Both scripts used to hash the
12
+ * entry alone, so every payload edit — which now lands in a route module — was
13
+ * invisible to the gate. They share this module precisely so the two sides of
14
+ * the binding cannot drift apart again.
15
+ *
16
+ * The digest covers the entry plus every `*.cjs` in the directory, ordered by
17
+ * repo-relative POSIX path, and mixes each path into the hash so a rename is a
18
+ * change even when the bytes are identical.
19
+ */
20
+ import { createHash } from "node:crypto";
21
+ import fs from "node:fs";
22
+ import path from "node:path";
23
+
24
+ /** Repo-relative path of the entry file (POSIX separators). */
25
+ export const MOCK_SERVER_ENTRY = "tests/visual/mock_server.cjs";
26
+ /** Repo-relative path of the directory holding the composed route modules. */
27
+ export const MOCK_SERVER_DIR = "tests/visual/mock_server";
28
+ /** Human-readable name for the whole surface, for error messages. */
29
+ export const MOCK_SERVER_LABEL = `${MOCK_SERVER_ENTRY} + ${MOCK_SERVER_DIR}/*.cjs`;
30
+
31
+ function toPosix(relativePath) {
32
+ return relativePath.split(path.sep).join("/");
33
+ }
34
+
35
+ /**
36
+ * Every file the fingerprint covers, as absolute paths in digest order.
37
+ * Returns null when the entry file is missing (nothing to bind to).
38
+ */
39
+ export function mockServerFiles(repoRoot) {
40
+ const entry = path.join(repoRoot, ...MOCK_SERVER_ENTRY.split("/"));
41
+ if (!fs.existsSync(entry)) {
42
+ return null;
43
+ }
44
+ const dir = path.join(repoRoot, ...MOCK_SERVER_DIR.split("/"));
45
+ const modules = fs.existsSync(dir)
46
+ ? fs
47
+ .readdirSync(dir)
48
+ .filter((name) => name.endsWith(".cjs"))
49
+ .map((name) => path.join(dir, name))
50
+ : [];
51
+ // "tests/visual/mock_server.cjs" sorts before "tests/visual/mock_server/x.cjs"
52
+ // ("." < "/"), so the entry always leads — but sort explicitly rather than
53
+ // relying on that, and sort on the POSIX form so Windows agrees with CI.
54
+ return [entry, ...modules].sort((a, b) => {
55
+ const left = toPosix(path.relative(repoRoot, a));
56
+ const right = toPosix(path.relative(repoRoot, b));
57
+ return left < right ? -1 : left > right ? 1 : 0;
58
+ });
59
+ }
60
+
61
+ /**
62
+ * `{ sha256, mtime, bytes, files }` for the mock API surface, or null when the
63
+ * entry file is missing.
64
+ *
65
+ * - `sha256` digest over `<relative path>\0<per-file sha256>\n` for each file
66
+ * - `mtime` newest mtime in the set (the last edit that could have moved a payload)
67
+ * - `bytes` total size of the set
68
+ * - `files` how many files the digest covers
69
+ */
70
+ export function mockServerFingerprint(repoRoot) {
71
+ const files = mockServerFiles(repoRoot);
72
+ if (!files) {
73
+ return null;
74
+ }
75
+ const digest = createHash("sha256");
76
+ let bytes = 0;
77
+ let newest = 0;
78
+ for (const file of files) {
79
+ const body = fs.readFileSync(file);
80
+ const stat = fs.statSync(file);
81
+ bytes += body.length;
82
+ newest = Math.max(newest, stat.mtime.getTime());
83
+ digest.update(toPosix(path.relative(repoRoot, file)));
84
+ digest.update("\0");
85
+ digest.update(createHash("sha256").update(body).digest("hex"));
86
+ digest.update("\n");
87
+ }
88
+ return {
89
+ sha256: digest.digest("hex"),
90
+ mtime: new Date(newest).toISOString(),
91
+ bytes,
92
+ files: files.length,
93
+ };
94
+ }