ltcai 11.4.0 → 11.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +48 -44
- package/docs/CHANGELOG.md +26 -0
- package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
- package/docs/DEVELOPMENT.md +1 -1
- package/docs/ONBOARDING.md +1 -1
- package/docs/OPERATIONS.md +1 -1
- package/docs/TRUST_MODEL.md +1 -1
- package/docs/WHY_LATTICE.md +1 -1
- package/docs/kg-schema.md +1 -1
- package/docs/v11.4.0_RUST_FOUNDATION_PLAN.md +11 -6
- package/docs/v11.5.0_RUST_COMPLETE_PLAN.md +145 -0
- package/lattice_brain/__init__.py +1 -1
- package/lattice_brain/runtime/multi_agent.py +1 -1
- package/latticeai/__init__.py +1 -1
- package/latticeai/api/index_jobs.py +145 -0
- package/latticeai/core/legacy_compatibility.py +1 -1
- package/latticeai/core/marketplace.py +1 -1
- package/latticeai/core/messages.py +5 -0
- package/latticeai/core/workspace_os_constants.py +1 -1
- package/latticeai/runtime/build_phases/features.py +14 -0
- package/latticeai/services/architecture_readiness.py +1 -1
- package/latticeai/services/product_readiness.py +1 -1
- package/package.json +1 -1
- package/scripts/check_current_release_docs.mjs +1 -1
- package/scripts/check_server_i18n.mjs +1 -0
- package/scripts/chunking_parity_corpus.py +449 -0
- package/scripts/generate_agent_parity_fixtures.py +752 -0
- package/scripts/generate_chunking_parity_fixtures.py +259 -0
- package/scripts/generate_rust_parity_fixtures.py +525 -90
- package/scripts/release_screen_claims.json +11 -0
- package/src-tauri/Cargo.lock +47 -4
- package/src-tauri/Cargo.toml +11 -4
- package/src-tauri/src/backend.rs +251 -140
- package/src-tauri/src/main.rs +16 -4
- package/src-tauri/src/topology.rs +356 -0
- package/src-tauri/tauri.conf.json +1 -1
- package/static/app/asset-manifest.json +40 -40
- package/static/app/assets/{Act-yYpYnn0v.js → Act-CWnxSCgN.js} +1 -1
- package/static/app/assets/{AdminConsole-DL3Cr5pL.js → AdminConsole-BEQYU6kF.js} +1 -1
- package/static/app/assets/{Brain-C1HBN0Wf.js → Brain-DWu1BhFg.js} +1 -1
- package/static/app/assets/{BrainHome-DoXRhUUC.js → BrainHome-95Hilr9R.js} +1 -1
- package/static/app/assets/{BrainSignals-6yR6ir5t.js → BrainSignals-QdeqCpAF.js} +1 -1
- package/static/app/assets/{Capture-CFIRsFNE.js → Capture-BHpCxnzb.js} +1 -1
- package/static/app/assets/{Chronicle-BZbEgiwN.js → Chronicle-B4xYKoed.js} +1 -1
- package/static/app/assets/{CommandPalette-D2pMxC2I.js → CommandPalette-BVXnttSz.js} +1 -1
- package/static/app/assets/{Library-DwO3yZST.js → Library-DgYcHome.js} +1 -1
- package/static/app/assets/{LivingBrain-Jn1GK0-S.js → LivingBrain-CrJLDbf7.js} +1 -1
- package/static/app/assets/{ProductFlow-B-w1R4Oo.js → ProductFlow-DFlScKoJ.js} +1 -1
- package/static/app/assets/{ReviewCard-6B27X8Vg.js → ReviewCard-Cy5f48Pj.js} +1 -1
- package/static/app/assets/{System-DW8F-2xL.js → System-NF8IfhTa.js} +1 -1
- package/static/app/assets/arrow-left-DwkSYrjR.js +1 -0
- package/static/app/assets/{bot-IM_E_Y12.js → bot-CucuhLhm.js} +1 -1
- package/static/app/assets/{brain-Ci1CkWjM.js → brain-BBnSryW_.js} +1 -1
- package/static/app/assets/{button-COwyqfHM.js → button-C2GUj2Ai.js} +1 -1
- package/static/app/assets/circle-check-CxOVPwYq.js +1 -0
- package/static/app/assets/{circle-pause-DEM4A1Y5.js → circle-pause-CbkWzBmG.js} +1 -1
- package/static/app/assets/{circle-play-C9djDuLd.js → circle-play-7lEaqHdJ.js} +1 -1
- package/static/app/assets/{cpu-DFdo1gw-.js → cpu-DAlCXlIy.js} +1 -1
- package/static/app/assets/{download-SnJL6oqk.js → download-RNhuuJwh.js} +1 -1
- package/static/app/assets/{folder-open-CqZeDkjE.js → folder-open-CLW4odzM.js} +1 -1
- package/static/app/assets/{hard-drive-j1jJXYYf.js → hard-drive-NKEiDIAJ.js} +1 -1
- package/static/app/assets/{index-_u5iUHDr.js → index-DMurvUuR.js} +3 -3
- package/static/app/assets/{input-B0lPdRQZ.js → input-D2UhPC1X.js} +1 -1
- package/static/app/assets/{link-2-CoFbooHS.js → link-2-6amKbP_P.js} +1 -1
- package/static/app/assets/{permissionCopy-BsyLxtao.js → permissionCopy-Cu9TZtdR.js} +1 -1
- package/static/app/assets/{primitives-DEbN-d6p.js → primitives-gPsccucr.js} +1 -1
- package/static/app/assets/search-Cj_TKk_2.js +1 -0
- package/static/app/assets/{share-2-CVtZ_ewX.js → share-2-Bau7KkPq.js} +1 -1
- package/static/app/assets/{shield-alert-CBi2GNWM.js → shield-alert-BufNYypi.js} +1 -1
- package/static/app/assets/{textarea-DNMpB5ih.js → textarea-BQnVWhYs.js} +1 -1
- package/static/app/assets/{useFocusTrap-C83t3GXF.js → useFocusTrap-B3_w60si.js} +1 -1
- package/static/app/assets/{useMutation-DtbJDoyz.js → useMutation-BHhCflT6.js} +1 -1
- package/static/app/assets/{useQuery-Dcp1OChy.js → useQuery-rBWfI-5t.js} +1 -1
- package/static/app/assets/{utils-BlZr7Pd4.js → utils-V_5-wxr5.js} +1 -1
- package/static/app/assets/{workspace-jJY4RuAV.js → workspace-K1zjYUHj.js} +1 -1
- package/static/app/index.html +3 -3
- package/static/sw.js +1 -1
- package/static/app/assets/arrow-left-DXvKg9U6.js +0 -1
- package/static/app/assets/circle-check-DfInj-qD.js +0 -1
- package/static/app/assets/search-BybIWPNd.js +0 -1
|
@@ -9,24 +9,24 @@ the real hash embedder, the real v2 projection and trigram FTS index), then runs
|
|
|
9
9
|
the real ``hybrid_search`` / ``search`` / ``vector_search`` over it and writes
|
|
10
10
|
their answers to ``rust/fixtures/golden/``.
|
|
11
11
|
|
|
12
|
+
v11.5.0 widens it past search: the same store also carries the conversation
|
|
13
|
+
corpus, and the same generator drives the KG relationship/traverse reads, the
|
|
14
|
+
service-layer graph and three-channel hybrid, the durable history reads and the
|
|
15
|
+
context assembler (:data:`SUITES`).
|
|
16
|
+
|
|
12
17
|
Two consumers read what it writes:
|
|
13
18
|
|
|
14
19
|
* ``tests/unit/test_rust_parity_contract.py`` re-runs the Python engines against
|
|
15
20
|
the committed database and asserts the goldens still hold — so a change to
|
|
16
|
-
Python
|
|
17
|
-
|
|
21
|
+
Python semantics fails loudly instead of silently invalidating the contract
|
|
22
|
+
the Rust side is pinned to;
|
|
18
23
|
* ``rust/lattice-retrieval/tests/parity.rs`` runs the Rust port against the same
|
|
19
24
|
database and the same goldens.
|
|
20
25
|
|
|
21
|
-
Determinism is the whole design constraint:
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
* ``hybrid_search``'s recency decay calls ``datetime.now()``, so the clock is
|
|
26
|
-
frozen at :data:`FROZEN_NOW` (recorded in the manifest for the Rust side);
|
|
27
|
-
* LLM concept extraction is forced off, so ``_topic_candidates`` always takes
|
|
28
|
-
the rule-based path a port can reproduce;
|
|
29
|
-
* every environment knob the retrieval stack reads is pinned to its default.
|
|
26
|
+
Determinism is the whole design constraint: every timestamp is written by the
|
|
27
|
+
real code and then **backdated**; the two ``datetime.now()`` calls the ports
|
|
28
|
+
reach are frozen at :data:`FROZEN_NOW` (recorded in the manifest); LLM concept
|
|
29
|
+
extraction is forced off; every environment knob is pinned to its default.
|
|
30
30
|
|
|
31
31
|
Usage::
|
|
32
32
|
|
|
@@ -36,6 +36,7 @@ Usage::
|
|
|
36
36
|
from __future__ import annotations
|
|
37
37
|
|
|
38
38
|
import json
|
|
39
|
+
import logging
|
|
39
40
|
import os
|
|
40
41
|
import shutil
|
|
41
42
|
import sqlite3
|
|
@@ -53,13 +54,11 @@ FIXTURE_DIR = REPO_ROOT / "rust" / "fixtures"
|
|
|
53
54
|
GOLDEN_DIR = FIXTURE_DIR / "golden"
|
|
54
55
|
STORE_PATH = FIXTURE_DIR / "parity_store.sqlite"
|
|
55
56
|
|
|
56
|
-
#: The wall clock
|
|
57
|
-
#: so a golden generated against a moving clock is not a golden.
|
|
57
|
+
#: The wall clock the ports see: a golden built against a moving clock is none.
|
|
58
58
|
FROZEN_NOW = "2026-08-01T12:00:00"
|
|
59
59
|
|
|
60
|
-
#: Every environment variable the ported path reads, pinned to the
|
|
61
|
-
#:
|
|
62
|
-
#: cross-encoder rerank off, rewrite on).
|
|
60
|
+
#: Every environment variable the ported path reads, pinned to the configuration
|
|
61
|
+
#: it targets (brute backend, RRF/expansion/rerank off, rewrite on).
|
|
63
62
|
PINNED_ENV: Dict[str, str] = {
|
|
64
63
|
"LATTICEAI_VECTOR_DIM": "384",
|
|
65
64
|
"LATTICEAI_VECTOR_INDEX": "brute",
|
|
@@ -80,16 +79,12 @@ WS_BETA = "ws-beta"
|
|
|
80
79
|
# ── the corpus ───────────────────────────────────────────────────────────────
|
|
81
80
|
# (node_id, type, title, summary, metadata, workspace_id, updated_at)
|
|
82
81
|
#
|
|
83
|
-
# Shaped on purpose:
|
|
84
|
-
#
|
|
85
|
-
#
|
|
86
|
-
#
|
|
87
|
-
#
|
|
88
|
-
#
|
|
89
|
-
# nothing to match, which pins the (hits, type_boost, updated_at) → id ASC
|
|
90
|
-
# tie-break that both engines have to reproduce;
|
|
91
|
-
# * two workspaces plus NULL-workspace legacy rows cover all three scoping
|
|
92
|
-
# answers (no scoping / empty set / a specific workspace).
|
|
82
|
+
# Shaped on purpose: every ``type_boost`` type appears and so do types outside it;
|
|
83
|
+
# titles/summaries are half Korean (the tokenizer, classifier and extractor all
|
|
84
|
+
# branch on script); the ``tie:`` block is five rows sharing one timestamp, one
|
|
85
|
+
# type and nothing to match, pinning the (hits, boost, updated_at) → id ASC
|
|
86
|
+
# tie-break; two workspaces plus NULL-workspace rows cover all three scoping
|
|
87
|
+
# answers (no scoping / empty set / a specific workspace).
|
|
93
88
|
NODES: List[Tuple[str, str, str, str, Dict[str, Any], Optional[str], str]] = [
|
|
94
89
|
("dec:fusion-alpha", "Decision", "Hybrid retrieval fusion stays alpha weighted",
|
|
95
90
|
"We decided the ranking keeps alpha fusion: lexical rank plus max normalized vector score.",
|
|
@@ -195,11 +190,10 @@ NODES: List[Tuple[str, str, str, str, Dict[str, Any], Optional[str], str]] = [
|
|
|
195
190
|
|
|
196
191
|
# (chunk_id, parent_node_id, index, node_title, text, chunk_fields, workspace, updated_at)
|
|
197
192
|
#
|
|
198
|
-
# The shape mirrors ``KnowledgeGraphIngestMixin``: every chunk is BOTH a
|
|
199
|
-
#
|
|
200
|
-
#
|
|
201
|
-
#
|
|
202
|
-
# reason chunk-heavy queries are in the query set.
|
|
193
|
+
# The shape mirrors ``KnowledgeGraphIngestMixin``: every chunk is BOTH a ``Chunk``
|
|
194
|
+
# node (lexical lane + workspace scoping) and a ``chunks`` row with its own
|
|
195
|
+
# embedding (vector lane, rolled up to its parent). Getting that duality wrong is
|
|
196
|
+
# the whole reason chunk-heavy queries are in the query set.
|
|
203
197
|
CHUNKS: List[Tuple[str, str, int, str, str, Dict[str, Any], Optional[str], str]] = [
|
|
204
198
|
("chunk:handbook:1", "doc:handbook", 0, "handbook.pdf chunk 1",
|
|
205
199
|
"온보딩 체크리스트: 첫째, 폴더를 연결합니다. 둘째, 질문을 합니다. 셋째, 근거를 확인합니다.",
|
|
@@ -218,14 +212,74 @@ CHUNKS: List[Tuple[str, str, int, str, str, Dict[str, Any], Optional[str], str]]
|
|
|
218
212
|
{"start_char": 900}, WS_BETA, "2026-05-28T13:15:02"),
|
|
219
213
|
]
|
|
220
214
|
|
|
221
|
-
# (from, to, type)
|
|
222
|
-
|
|
223
|
-
|
|
224
|
-
|
|
225
|
-
|
|
226
|
-
|
|
227
|
-
|
|
228
|
-
|
|
215
|
+
# (from, to, type, weight, created_at)
|
|
216
|
+
#
|
|
217
|
+
# Weights and timestamps are assigned rather than inherited: ``relationship_search``
|
|
218
|
+
# orders by ``weight DESC, created_at DESC, id ASC`` and ``traverse`` caps every BFS
|
|
219
|
+
# round with ``ORDER BY weight DESC, id ASC``, so an edge set sharing one weight and
|
|
220
|
+
# one clock proves nothing. Three shapes are deliberate — a weight tie broken by
|
|
221
|
+
# ``created_at`` (the ``org:lattice`` pair), a weight *and* clock tie broken by edge
|
|
222
|
+
# id (``topic:quality``/``deck:review``), and legacy-global endpoints for scoping.
|
|
223
|
+
# Types are canonicalized by the write door (``relates_to``/``owns`` → ``MENTIONS``),
|
|
224
|
+
# which is why the goldens record uppercase names the fixture never spells.
|
|
225
|
+
EDGES: List[Tuple[str, str, str, float, str]] = [
|
|
226
|
+
("dec:fusion-alpha", "concept:retrieval", "mentions", 0.9, "2026-07-01T00:00:00"),
|
|
227
|
+
("dec:fusion-alpha", "concept:ranking", "mentions", 0.8, "2026-07-02T00:00:00"),
|
|
228
|
+
("task:parity-harness", "dec:rust-foundation", "relates_to", 1.0, "2026-07-03T00:00:00"),
|
|
229
|
+
("doc:handbook", "page:onboarding", "contains", 0.7, "2026-07-04T00:00:00"),
|
|
230
|
+
("meeting:weekly", "dec:fusion-alpha", "discusses", 0.95, "2026-07-05T00:00:00"),
|
|
231
|
+
("person:jiwon", "task:parity-harness", "owns", 0.6, "2026-07-06T00:00:00"),
|
|
232
|
+
("person:minseo", "code:build-failure", "owns", 0.6, "2026-07-07T00:00:00"),
|
|
233
|
+
("meeting:weekly", "task:onboarding-checklist", "discusses", 0.55, "2026-07-08T00:00:00"),
|
|
234
|
+
("meeting:kickoff", "dec:rust-foundation", "discusses", 0.5, "2026-06-01T00:00:00"),
|
|
235
|
+
("doc:retrieval-spec", "concept:ranking", "mentions", 0.45, "2026-06-02T00:00:00"),
|
|
236
|
+
("concept:retrieval", "concept:ranking", "relates_to", 0.4, "2026-06-03T00:00:00"),
|
|
237
|
+
("file:ranking-notes", "dec:fusion-alpha", "relates_to", 0.35, "2026-06-04T00:00:00"),
|
|
238
|
+
("code:hybrid-search", "file:ranking-notes", "relates_to", 0.3, "2026-06-05T00:00:00"),
|
|
239
|
+
# Same weight, different clock → created_at DESC decides.
|
|
240
|
+
("org:lattice", "person:jiwon", "contains", 0.25, "2026-06-06T00:00:00"),
|
|
241
|
+
("org:lattice", "person:minseo", "contains", 0.25, "2026-06-07T00:00:00"),
|
|
242
|
+
("slide:fusion", "concept:retrieval", "mentions", 0.25, "2026-06-08T00:00:00"),
|
|
243
|
+
# Same weight AND same clock → the id ASC tie-break is the only thing left.
|
|
244
|
+
("topic:quality", "deck:review", "relates_to", 0.2, "2026-06-09T00:00:00"),
|
|
245
|
+
("deck:review", "meeting:weekly", "relates_to", 0.2, "2026-06-09T00:00:00"),
|
|
246
|
+
("task:onboarding-checklist", "page:onboarding", "relates_to", 0.15, "2026-05-01T00:00:00"),
|
|
247
|
+
("doc:handbook", "doc:retrieval-spec", "relates_to", 0.1, "2026-05-02T00:00:00"),
|
|
248
|
+
("tie:a", "tie:b", "relates_to", 1.0, "2026-07-03T00:00:00"),
|
|
249
|
+
]
|
|
250
|
+
|
|
251
|
+
# ── the conversation corpus (episodic memory, same database file) ────────────
|
|
252
|
+
# (conversation_id, role, content, user_email, nickname, source, timestamp,
|
|
253
|
+
# workspace_id, organization_id, extra)
|
|
254
|
+
#
|
|
255
|
+
# Every branch the history reads take: two users × three workspaces, NULL and
|
|
256
|
+
# empty-string workspaces (the legacy rows ``_scope_sql`` admits), rows with no
|
|
257
|
+
# ``conversation_id`` (the ``legacy-previous-history`` bucket), a whitespace-only
|
|
258
|
+
# first message (the ``새 대화`` placeholder and its later upgrade), an
|
|
259
|
+
# assistant-first conversation (no upgrade), an empty timestamp (the ``or ""``
|
|
260
|
+
# fallbacks), extra keys ``metadata_json`` merges flat, and ko/en content.
|
|
261
|
+
MESSAGES: List[Tuple[Any, ...]] = [
|
|
262
|
+
("conv-a", "user", "온보딩 체크리스트 어떻게 시작해?", "jiwon@lattice.ai", "지원", "web", "2026-07-20T09:00:00", WS_ALPHA, "org-1", {}),
|
|
263
|
+
("conv-a", "assistant", "먼저 폴더를 연결하세요. 그다음 질문하면 됩니다.", "jiwon@lattice.ai", None, "web", "2026-07-20T09:00:05", WS_ALPHA, "org-1", {}),
|
|
264
|
+
("conv-a", "user", "고마워", "jiwon@lattice.ai", "지원", "web", "2026-07-20T09:01:00", WS_ALPHA, "org-1", {"trace_id": "t-1"}),
|
|
265
|
+
("conv-b", "user", "How does hybrid retrieval ranking work?", "minseo@lattice.ai", "Minseo", "telegram", "2026-07-21T10:00:00", WS_BETA, "org-1", {}),
|
|
266
|
+
("conv-b", "assistant", "The lexical channel scores one over rank.", "minseo@lattice.ai", None, "telegram", "2026-07-21T10:00:07", WS_BETA, "org-1", {"tokens": 42, "cited": ["doc:retrieval-spec"]}),
|
|
267
|
+
("conv-c", "user", " \n ", None, None, None, "2026-07-22T08:00:00", None, None, {}),
|
|
268
|
+
("conv-c", "assistant", "무엇을 도와드릴까요?", None, None, None, "2026-07-22T08:00:05", None, None, {}),
|
|
269
|
+
("conv-c", "user", "지난주 회의 기록 보여줘", None, None, None, "2026-07-22T08:01:00", None, None, {}),
|
|
270
|
+
(None, "user", "이전 대화 기록입니다", "jiwon@lattice.ai", "지원", "web", "2026-06-01T09:00:00", "", None, {}),
|
|
271
|
+
(None, "assistant", "네, 확인했습니다.", "jiwon@lattice.ai", None, "web", "2026-06-01T09:00:03", "", None, {}),
|
|
272
|
+
(None, "user", "legacy english message about ranking", None, None, None, "", None, None, {}),
|
|
273
|
+
("conv-d", "user", "빌드 실패 원인 알려줘", "minseo@lattice.ai", "Minseo", "vscode", "2026-07-23T11:00:00", WS_ALPHA, "org-1", {}),
|
|
274
|
+
("conv-d", "assistant", "컴파일 오류 로그를 확인하세요.", "minseo@lattice.ai", None, "vscode", "2026-07-23T11:00:04", WS_ALPHA, "org-1", {}),
|
|
275
|
+
("conv-e", "user", "Ranking ties are broken by node id", "jiwon@lattice.ai", "지원", "web", "2026-07-24T12:00:00", WS_BETA, "org-2", {}),
|
|
276
|
+
("conv-e", "assistant", "Yes — id ascending keeps it stable.", "jiwon@lattice.ai", None, "web", "2026-07-24T12:00:05", WS_BETA, "org-2", {}),
|
|
277
|
+
("conv-f", "user", "검색 품질을 어떻게 측정하나요? 재현율과 정밀도를 모두 보고 싶고 주간 회의에서 공유할 예정입니다.", None, None, "web", "2026-07-25T13:00:00", None, None, {}),
|
|
278
|
+
("conv-f", "assistant", "재현율/정밀도 지표는 분기 리뷰 발표자료에 있습니다.", None, None, "web", "2026-07-25T13:00:05", None, None, {}),
|
|
279
|
+
("conv-g", "assistant", "assistant-first conversation", None, None, "web", "2026-07-26T14:00:00", None, None, {}),
|
|
280
|
+
("conv-g", "user", "follow up question about ranking", None, None, "web", "2026-07-26T14:00:10", None, None, {}),
|
|
281
|
+
("conv-h", "user", "Empty user and empty workspace row", "", "", "", "2026-07-27T15:00:00", "", "", {}),
|
|
282
|
+
("conv-h", "assistant", "회의 결정 사항을 정리했습니다.", "", "", "", "2026-07-27T15:00:06", "", "", {}),
|
|
229
283
|
]
|
|
230
284
|
|
|
231
285
|
# ── the query set ────────────────────────────────────────────────────────────
|
|
@@ -253,25 +307,178 @@ QUERIES: List[Dict[str, Any]] = [
|
|
|
253
307
|
{"key": "ws_alpha", "query": "hybrid retrieval ranking", "allowed": [WS_ALPHA]},
|
|
254
308
|
{"key": "ws_alpha_legacy", "query": "회의 결정 사항", "allowed": [WS_ALPHA], "legacy": True},
|
|
255
309
|
{"key": "ws_beta", "query": "온보딩 체크리스트", "allowed": [WS_BETA]},
|
|
256
|
-
# A vector floor nothing clears
|
|
257
|
-
#
|
|
258
|
-
# label without breaking the store.
|
|
310
|
+
# A vector floor nothing clears — the only way to reach the stale-embedder
|
|
311
|
+
# probe and the lexical-only fusion label without breaking the store.
|
|
259
312
|
{"key": "min_vector_floor", "query": "hybrid retrieval ranking", "min_vector": 0.95},
|
|
260
|
-
#
|
|
261
|
-
# candidate list and the cut is observable.
|
|
313
|
+
# top_k small enough that the rerank window (top_k * 2) cuts the candidates.
|
|
262
314
|
{"key": "top_k_small", "query": "hybrid retrieval ranking", "top_k": 3},
|
|
263
|
-
#
|
|
264
|
-
# query that would otherwise be recency-classed
|
|
315
|
+
# Pinned alpha: no policy, so no class, no rewrite, and no age decay on a
|
|
316
|
+
# query that would otherwise be recency-classed.
|
|
265
317
|
{"key": "alpha_pinned", "query": "지난주 회의 기록", "alpha": 0.2},
|
|
266
318
|
# A limit below the FTS hit count, so `ORDER BY rank LIMIT ?` decides which
|
|
267
|
-
# rows exist at all
|
|
268
|
-
#
|
|
269
|
-
# SQLite version difference between the two runtimes could show up.
|
|
319
|
+
# rows exist at all — the one place bm25 ordering (and therefore a SQLite
|
|
320
|
+
# version difference between the two runtimes) is observable.
|
|
270
321
|
{"key": "fts_rank_cut", "query": "Tie candidate", "limit": 2, "top_k": 2},
|
|
271
322
|
]
|
|
272
323
|
|
|
273
|
-
#:
|
|
274
|
-
#:
|
|
324
|
+
#: Every branch of the memories section: a workspace hit, one with no kind and an
|
|
325
|
+
#: empty snippet, a non-workspace row to drop, and one past the limit.
|
|
326
|
+
CONTEXT_MEMORIES: Dict[str, Any] = {
|
|
327
|
+
"results": [
|
|
328
|
+
{"id": "mem-1", "kind": "preference", "snippet": "답변은 한국어로", "score": 0.91, "source": "workspace"},
|
|
329
|
+
{"id": "mem-2", "kind": None, "snippet": "", "score": 0.0, "source": "workspace"},
|
|
330
|
+
{"id": "mem-3", "kind": "decision", "snippet": "ranking keeps alpha fusion", "score": 0.4, "source": "personal"},
|
|
331
|
+
{"id": "mem-4", "kind": "fact", "snippet": "온보딩은 다섯 걸음", "score": 0.3, "source": "workspace"},
|
|
332
|
+
]
|
|
333
|
+
}
|
|
334
|
+
|
|
335
|
+
#: A ledger with a pathless row, a non-dict row, and more than the ten-row cut.
|
|
336
|
+
CONTEXT_ARTIFACTS: List[Any] = [
|
|
337
|
+
{"path": "notes/ranking.md", "at": "2026-07-20T09:00:00", "run_id": "run-1"},
|
|
338
|
+
{"path": "notes/onboarding.md", "run_id": "run-2"},
|
|
339
|
+
{"path": "", "at": "2026-07-20T09:00:01"},
|
|
340
|
+
"not-a-dict",
|
|
341
|
+
] + [{"path": f"out/file-{index}.md", "at": None, "run_id": f"r{index}"} for index in range(10)]
|
|
342
|
+
|
|
343
|
+
#: The Phase-2/3 suites: one spec list per ported entry point.
|
|
344
|
+
SUITES: Dict[str, List[Dict[str, Any]]] = {
|
|
345
|
+
"relationship": [
|
|
346
|
+
{"key": "all"},
|
|
347
|
+
{"key": "by_type_mention", "relationship_type": "mention"},
|
|
348
|
+
{"key": "by_type_contains", "relationship_type": "CONTAINS"},
|
|
349
|
+
{"key": "by_type_unknown", "relationship_type": "relates_to"},
|
|
350
|
+
{"key": "by_node", "node_id": "dec:fusion-alpha"},
|
|
351
|
+
{"key": "by_query_ko", "query": "회의"},
|
|
352
|
+
{"key": "by_query_en", "query": "ranking"},
|
|
353
|
+
{"key": "by_query_meta", "query": "lattice"},
|
|
354
|
+
{"key": "combined", "node_id": "dec:fusion-alpha", "relationship_type": "mention", "query": "retrieval"},
|
|
355
|
+
{"key": "limit_one", "limit": 1},
|
|
356
|
+
{"key": "limit_zero", "limit": 0},
|
|
357
|
+
{"key": "limit_over", "limit": 500},
|
|
358
|
+
{"key": "scoped_alpha", "allowed": [WS_ALPHA]},
|
|
359
|
+
{"key": "scoped_alpha_legacy", "allowed": [WS_ALPHA], "legacy": True},
|
|
360
|
+
{"key": "scoped_beta", "allowed": [WS_BETA]},
|
|
361
|
+
{"key": "scoped_empty", "allowed": []},
|
|
362
|
+
{"key": "no_hit", "query": "zzqq wumpus"},
|
|
363
|
+
],
|
|
364
|
+
"traverse": [
|
|
365
|
+
# 9 clamps to 4 and -1 clamps to 0; both are on purpose.
|
|
366
|
+
*[{"key": f"hub_d{depth}", "node_id": "dec:fusion-alpha", "depth": depth}
|
|
367
|
+
for depth in (0, 1, 2, 3, 9)],
|
|
368
|
+
{"key": "hub_dneg", "node_id": "dec:fusion-alpha", "depth": -1},
|
|
369
|
+
{"key": "leaf_d2", "node_id": "code:hybrid-search", "depth": 2},
|
|
370
|
+
{"key": "isolated", "node_id": "tie:c", "depth": 2},
|
|
371
|
+
{"key": "limit_two", "node_id": "dec:fusion-alpha", "depth": 3, "limit": 2},
|
|
372
|
+
{"key": "limit_five", "node_id": "dec:fusion-alpha", "depth": 3, "limit": 5},
|
|
373
|
+
{"key": "limit_zero", "node_id": "dec:fusion-alpha", "depth": 2, "limit": 0},
|
|
374
|
+
{"key": "limit_over", "node_id": "dec:fusion-alpha", "depth": 2, "limit": 900},
|
|
375
|
+
{"key": "org_hub", "node_id": "org:lattice", "depth": 2},
|
|
376
|
+
{"key": "tie_pair", "node_id": "tie:a", "depth": 2},
|
|
377
|
+
{"key": "scoped_alpha", "node_id": "dec:fusion-alpha", "depth": 2, "allowed": [WS_ALPHA]},
|
|
378
|
+
{"key": "scoped_alpha_legacy", "node_id": "dec:fusion-alpha", "depth": 2, "allowed": [WS_ALPHA], "legacy": True},
|
|
379
|
+
{"key": "scoped_seed_hidden", "node_id": "dec:fusion-alpha", "allowed": [WS_BETA]},
|
|
380
|
+
{"key": "scoped_empty", "node_id": "dec:fusion-alpha", "allowed": []},
|
|
381
|
+
{"key": "empty_id", "node_id": ""},
|
|
382
|
+
{"key": "missing_seed", "node_id": "nope:missing"},
|
|
383
|
+
],
|
|
384
|
+
"graph_search": [
|
|
385
|
+
{"key": "en_fact", "query": "hybrid retrieval ranking"},
|
|
386
|
+
{"key": "ko_fact", "query": "회의 결정 사항"},
|
|
387
|
+
{"key": "person", "query": "who owns the onboarding checklist"},
|
|
388
|
+
{"key": "code", "query": "빌드 실패 원인"},
|
|
389
|
+
{"key": "expand0", "query": "hybrid retrieval ranking", "expand_depth": 0},
|
|
390
|
+
{"key": "expand3", "query": "hybrid retrieval ranking", "expand_depth": 3},
|
|
391
|
+
{"key": "expand_clamp", "query": "회의 결정 사항", "expand_depth": 9},
|
|
392
|
+
{"key": "limit_small", "query": "hybrid retrieval ranking", "limit": 3},
|
|
393
|
+
{"key": "limit_over", "query": "회의 결정 사항", "limit": 500},
|
|
394
|
+
{"key": "scoped_alpha", "query": "hybrid retrieval ranking", "allowed": [WS_ALPHA]},
|
|
395
|
+
{"key": "scoped_alpha_legacy", "query": "회의 결정 사항", "allowed": [WS_ALPHA], "legacy": True},
|
|
396
|
+
{"key": "scoped_empty", "query": "hybrid retrieval ranking", "allowed": []},
|
|
397
|
+
{"key": "no_hit", "query": "zzqq wumpus nonsense"},
|
|
398
|
+
{"key": "empty_query", "query": ""},
|
|
399
|
+
],
|
|
400
|
+
"service_hybrid": [
|
|
401
|
+
{"key": "en_fact", "query": "hybrid retrieval ranking"},
|
|
402
|
+
{"key": "ko_recency", "query": "지난주 회의 기록"},
|
|
403
|
+
{"key": "en_code", "query": "vector_search() returns"},
|
|
404
|
+
{"key": "ko_person", "query": "담당자 누구"},
|
|
405
|
+
{"key": "en_filler", "query": " what is the retrieval specification please "},
|
|
406
|
+
# Explicit weights disable BOTH the rewrite and the age decay, on a
|
|
407
|
+
# query that would otherwise get both. That asymmetry is the contract.
|
|
408
|
+
{"key": "pinned_recency", "query": "지난주 회의 기록", "weights": {"keyword": 0.5, "vector": 0.3, "graph": 0.2}},
|
|
409
|
+
{"key": "pinned_partial", "query": "hybrid retrieval ranking", "weights": {"graph": 1.0}},
|
|
410
|
+
{"key": "pinned_zero", "query": "회의 결정 사항", "weights": {"keyword": 0.0, "vector": 0.0, "graph": 0.0}},
|
|
411
|
+
{"key": "limit_small", "query": "hybrid retrieval ranking", "limit": 3},
|
|
412
|
+
{"key": "limit_over", "query": "회의 결정 사항", "limit": 500},
|
|
413
|
+
{"key": "channel_limits", "query": "hybrid retrieval ranking", "keyword_limit": 5, "vector_limit": 5, "graph_limit": 5},
|
|
414
|
+
{"key": "scoped_alpha", "query": "hybrid retrieval ranking", "allowed": [WS_ALPHA]},
|
|
415
|
+
{"key": "scoped_alpha_legacy", "query": "회의 결정 사항", "allowed": [WS_ALPHA], "legacy": True},
|
|
416
|
+
{"key": "scoped_empty", "query": "hybrid retrieval ranking", "allowed": []},
|
|
417
|
+
{"key": "no_hit", "query": "zzqq wumpus nonsense"},
|
|
418
|
+
{"key": "empty_query", "query": ""},
|
|
419
|
+
],
|
|
420
|
+
"history": [
|
|
421
|
+
{"key": "all"},
|
|
422
|
+
{"key": "limit_two", "limit": 2},
|
|
423
|
+
{"key": "limit_zero", "limit": 0},
|
|
424
|
+
{"key": "conv_a", "conversation_id": "conv-a"},
|
|
425
|
+
{"key": "conv_missing", "conversation_id": "nope"},
|
|
426
|
+
{"key": "conv_null", "conversation_id": ""},
|
|
427
|
+
{"key": "user_jiwon", "user_email": "jiwon@lattice.ai"},
|
|
428
|
+
{"key": "user_jiwon_strict", "user_email": "jiwon@lattice.ai", "legacy": False},
|
|
429
|
+
{"key": "user_unknown", "user_email": "ghost@lattice.ai"},
|
|
430
|
+
{"key": "ws_alpha", "allowed": [WS_ALPHA]},
|
|
431
|
+
{"key": "ws_alpha_strict", "allowed": [WS_ALPHA], "legacy": False},
|
|
432
|
+
{"key": "ws_both_strict", "allowed": [WS_ALPHA, WS_BETA], "legacy": False},
|
|
433
|
+
{"key": "ws_empty_legacy", "allowed": []},
|
|
434
|
+
{"key": "ws_empty_strict", "allowed": [], "legacy": False},
|
|
435
|
+
{"key": "user_and_ws", "user_email": "jiwon@lattice.ai", "allowed": [WS_ALPHA], "legacy": False},
|
|
436
|
+
{"key": "conv_and_user", "conversation_id": "conv-a", "user_email": "jiwon@lattice.ai", "legacy": False},
|
|
437
|
+
{"key": "ws_blank_only", "allowed": [""], "legacy": False},
|
|
438
|
+
],
|
|
439
|
+
"conversations": [
|
|
440
|
+
{"key": "all"},
|
|
441
|
+
{"key": "user_jiwon", "user_email": "jiwon@lattice.ai"},
|
|
442
|
+
{"key": "user_jiwon_strict", "user_email": "jiwon@lattice.ai", "legacy": False},
|
|
443
|
+
{"key": "ws_alpha_strict", "allowed": [WS_ALPHA], "legacy": False},
|
|
444
|
+
{"key": "ws_beta_strict", "allowed": [WS_BETA], "legacy": False},
|
|
445
|
+
{"key": "ws_empty_strict", "allowed": [], "legacy": False},
|
|
446
|
+
],
|
|
447
|
+
"conversation_messages": [
|
|
448
|
+
{"key": "conv_a", "conversation_id": "conv-a"},
|
|
449
|
+
{"key": "legacy_bucket", "conversation_id": "legacy-previous-history"},
|
|
450
|
+
{"key": "missing", "conversation_id": "nope"},
|
|
451
|
+
{"key": "scoped_alpha_strict", "conversation_id": "conv-a", "allowed": [WS_ALPHA], "legacy": False},
|
|
452
|
+
{"key": "legacy_bucket_scoped", "conversation_id": "legacy-previous-history", "allowed": [WS_ALPHA], "legacy": False},
|
|
453
|
+
],
|
|
454
|
+
"history_search": [
|
|
455
|
+
{"key": "ko_hit", "query": "회의"},
|
|
456
|
+
{"key": "ko_partial", "query": "체크리스트"},
|
|
457
|
+
{"key": "en_hit", "query": "ranking"},
|
|
458
|
+
{"key": "case_insensitive", "query": "RANKING"},
|
|
459
|
+
{"key": "blank", "query": " "},
|
|
460
|
+
{"key": "no_hit", "query": "zzqq"},
|
|
461
|
+
{"key": "limit_one", "query": "ranking", "limit": 1},
|
|
462
|
+
{"key": "scoped_strict", "query": "ranking", "allowed": [WS_BETA], "legacy": False},
|
|
463
|
+
],
|
|
464
|
+
"context_assemble": [
|
|
465
|
+
{"key": "all_seams", "query": "회의 결정 사항", "memories": CONTEXT_MEMORIES, "artifacts": CONTEXT_ARTIFACTS, "notes": "정원 노트: 랭킹은 alpha 융합을 유지한다.", "recent": {"limit": 4}},
|
|
466
|
+
{"key": "knowledge_only", "query": "hybrid retrieval ranking"},
|
|
467
|
+
{"key": "no_seams", "query": "hybrid retrieval ranking", "knowledge": False},
|
|
468
|
+
{"key": "memories_only", "query": "온보딩", "knowledge": False, "memories": CONTEXT_MEMORIES, "memory_limit": 2},
|
|
469
|
+
{"key": "artifacts_only", "query": "온보딩", "knowledge": False, "artifacts": CONTEXT_ARTIFACTS},
|
|
470
|
+
{"key": "notes_blank", "query": "온보딩", "knowledge": False, "notes": " "},
|
|
471
|
+
{"key": "recent_conversation", "query": "온보딩", "knowledge": False, "recent": {"conversation_id": "conv-a", "limit": 10}},
|
|
472
|
+
{"key": "recent_personal_workspace", "query": "온보딩", "knowledge": False, "recent": {"workspace_id": "personal", "limit": 6}},
|
|
473
|
+
{"key": "recent_user_scoped", "query": "온보딩", "knowledge": False, "recent": {"user_email": "jiwon@lattice.ai", "limit": 5}},
|
|
474
|
+
{"key": "budget_tiny", "query": "회의 결정 사항", "memories": CONTEXT_MEMORIES, "artifacts": CONTEXT_ARTIFACTS, "notes": "정원 노트: 랭킹은 alpha 융합을 유지한다.", "recent": {"limit": 4}, "budget": 20},
|
|
475
|
+
{"key": "budget_one", "query": "회의 결정 사항", "memories": CONTEXT_MEMORIES, "artifacts": CONTEXT_ARTIFACTS, "notes": "노트", "recent": {"limit": 4}, "budget": 1},
|
|
476
|
+
{"key": "budget_zero", "query": "회의 결정 사항", "memories": CONTEXT_MEMORIES, "notes": "노트", "budget": 0},
|
|
477
|
+
{"key": "knowledge_limit_one", "query": "hybrid retrieval ranking", "knowledge_limit": 1},
|
|
478
|
+
],
|
|
479
|
+
}
|
|
480
|
+
|
|
481
|
+
#: Texts whose tokenizer output, hashes and vectors pin the embedding port.
|
|
275
482
|
EMBEDDING_TEXTS: List[str] = [
|
|
276
483
|
"hybrid retrieval ranking",
|
|
277
484
|
"회의 결정 사항",
|
|
@@ -283,8 +490,7 @@ EMBEDDING_TEXTS: List[str] = [
|
|
|
283
490
|
"a",
|
|
284
491
|
]
|
|
285
492
|
|
|
286
|
-
#: Values
|
|
287
|
-
#: shapes real fusion arithmetic produces.
|
|
493
|
+
#: Values separating CPython round-half-even from naive scaling, plus real shapes.
|
|
288
494
|
ROUNDING_VALUES: List[float] = [
|
|
289
495
|
0.0, 1.0, 5e-07, 1.5e-06, 2.5e-06, 2.6535895, 1 / 3, 1 / 7,
|
|
290
496
|
0.1234565, 0.1234575, 0.9999999999, 123456.7890625, -0.0000005,
|
|
@@ -309,13 +515,14 @@ def pinned_environment() -> Iterator[None]:
|
|
|
309
515
|
|
|
310
516
|
@contextmanager
|
|
311
517
|
def frozen_clock() -> Iterator[None]:
|
|
312
|
-
"""Freeze
|
|
518
|
+
"""Freeze every ported ``datetime.now()`` at :data:`FROZEN_NOW`.
|
|
313
519
|
|
|
314
|
-
Only the recency-class age decay reads the clock,
|
|
315
|
-
|
|
316
|
-
|
|
520
|
+
Only the recency-class age decay reads the clock, through the ``datetime``
|
|
521
|
+
name its module imported — so rebinding that name is the whole patch. Two
|
|
522
|
+
modules do it: the graph-layer and the service-layer ``hybrid_search``.
|
|
317
523
|
"""
|
|
318
524
|
from lattice_brain.graph.retrieval import hybrid as hybrid_module
|
|
525
|
+
from latticeai.services import search_service as service_module
|
|
319
526
|
|
|
320
527
|
frozen = datetime.fromisoformat(FROZEN_NOW)
|
|
321
528
|
|
|
@@ -324,20 +531,23 @@ def frozen_clock() -> Iterator[None]:
|
|
|
324
531
|
def now(cls, tz=None): # noqa: ARG003 — mirrors datetime.now's signature
|
|
325
532
|
return frozen
|
|
326
533
|
|
|
327
|
-
|
|
328
|
-
|
|
534
|
+
modules = (hybrid_module, service_module)
|
|
535
|
+
originals = [module.datetime for module in modules]
|
|
536
|
+
for module in modules:
|
|
537
|
+
module.datetime = _FrozenDatetime
|
|
329
538
|
try:
|
|
330
539
|
yield
|
|
331
540
|
finally:
|
|
332
|
-
|
|
541
|
+
for module, original in zip(modules, originals, strict=True):
|
|
542
|
+
module.datetime = original
|
|
333
543
|
|
|
334
544
|
|
|
335
545
|
@contextmanager
|
|
336
546
|
def rules_only_extraction() -> Iterator[None]:
|
|
337
547
|
"""Force ``_topic_candidates`` down its rule-based path.
|
|
338
548
|
|
|
339
|
-
The LLM path needs a bound router
|
|
340
|
-
|
|
549
|
+
The LLM path needs a bound router no fixture run has, but "no router happens
|
|
550
|
+
to be bound" is an accident and this contract cannot rest on one.
|
|
341
551
|
"""
|
|
342
552
|
from lattice_brain.graph._kg_common import extraction
|
|
343
553
|
|
|
@@ -352,10 +562,27 @@ def rules_only_extraction() -> Iterator[None]:
|
|
|
352
562
|
def open_store(db_path: Path):
|
|
353
563
|
"""A ``KnowledgeGraphStore`` over ``db_path`` (blobs beside it)."""
|
|
354
564
|
from lattice_brain.graph import KnowledgeGraphStore
|
|
355
|
-
|
|
356
565
|
return KnowledgeGraphStore(Path(db_path), Path(db_path).parent / "blobs")
|
|
357
566
|
|
|
358
567
|
|
|
568
|
+
def open_conversations(db_path: Path):
|
|
569
|
+
"""A ``ConversationStore`` over ``db_path`` (the same file as the graph)."""
|
|
570
|
+
from lattice_brain.conversations import ConversationStore
|
|
571
|
+
return ConversationStore(Path(db_path))
|
|
572
|
+
|
|
573
|
+
|
|
574
|
+
def write_conversations(db_path: Path) -> None:
|
|
575
|
+
"""Append :data:`MESSAGES` through the real durable-history write path."""
|
|
576
|
+
conversations = open_conversations(db_path)
|
|
577
|
+
for conv_id, role, content, email, nick, source, stamp, workspace, org, extra in MESSAGES:
|
|
578
|
+
conversations.append({
|
|
579
|
+
"conversation_id": conv_id, "role": role, "content": content,
|
|
580
|
+
"user_email": email, "user_nickname": nick, "source": source,
|
|
581
|
+
"timestamp": stamp, "workspace_id": workspace, "organization_id": org,
|
|
582
|
+
**extra,
|
|
583
|
+
})
|
|
584
|
+
|
|
585
|
+
|
|
359
586
|
def _backdate(conn: sqlite3.Connection, node_id: str, stamp: str) -> None:
|
|
360
587
|
conn.execute("UPDATE nodes SET created_at=?, updated_at=? WHERE id=?", (stamp, stamp, node_id))
|
|
361
588
|
conn.execute(
|
|
@@ -388,15 +615,16 @@ def build_store(db_path: Path) -> None:
|
|
|
388
615
|
conn, chunk_id=chunk_id, source_node=parent, text=text, metadata=metadata
|
|
389
616
|
)
|
|
390
617
|
_backdate(conn, chunk_id, stamp)
|
|
391
|
-
for from_node, to_node, edge_type in EDGES:
|
|
392
|
-
|
|
393
|
-
|
|
394
|
-
|
|
395
|
-
|
|
396
|
-
)
|
|
397
|
-
|
|
398
|
-
|
|
399
|
-
#
|
|
618
|
+
for from_node, to_node, edge_type, weight, stamp in EDGES:
|
|
619
|
+
# ``_upsert_edge`` stamps ``created_at`` from the wall clock in BOTH
|
|
620
|
+
# tables, and the read path is the ``kgv2_edges`` view over ``edges_v2``:
|
|
621
|
+
# backdating only the legacy table (as this generator first did) left
|
|
622
|
+
# the relationship ordering moving with the clock.
|
|
623
|
+
edge_id = store._upsert_edge(conn, from_node, to_node, edge_type, weight, {})
|
|
624
|
+
conn.execute("UPDATE edges SET created_at=? WHERE id=?", (stamp, edge_id))
|
|
625
|
+
conn.execute("UPDATE edges_v2 SET created_at=? WHERE id=?", (stamp, edge_id))
|
|
626
|
+
# ``indexed_at`` decides the candidate scan order (and, when the cap bites,
|
|
627
|
+
# which candidates exist at all), so it is assigned rather than inherited.
|
|
400
628
|
item_ids = [row["item_id"] for row in conn.execute(
|
|
401
629
|
"SELECT item_id FROM vector_embeddings ORDER BY item_id ASC"
|
|
402
630
|
).fetchall()]
|
|
@@ -406,9 +634,10 @@ def build_store(db_path: Path) -> None:
|
|
|
406
634
|
"UPDATE vector_embeddings SET indexed_at=? WHERE item_id=?", (stamp, item_id)
|
|
407
635
|
)
|
|
408
636
|
store.record_embedder_fingerprint()
|
|
637
|
+
write_conversations(db_path)
|
|
409
638
|
|
|
410
639
|
# Leave one self-contained file behind: checkpoint the WAL, drop back to a
|
|
411
|
-
# rollback journal so no
|
|
640
|
+
# rollback journal so no sidecar is committed, and compact.
|
|
412
641
|
with sqlite3.connect(str(db_path)) as conn:
|
|
413
642
|
conn.execute("PRAGMA wal_checkpoint(TRUNCATE)")
|
|
414
643
|
conn.execute("PRAGMA journal_mode=DELETE")
|
|
@@ -429,17 +658,12 @@ def _allowed(spec: Dict[str, Any]):
|
|
|
429
658
|
|
|
430
659
|
ENGINES: Dict[str, Callable[[Any, Dict[str, Any]], Dict[str, Any]]] = {
|
|
431
660
|
"hybrid": lambda store, spec: store.hybrid_search(
|
|
432
|
-
spec["query"],
|
|
433
|
-
|
|
434
|
-
alpha=spec.get("alpha"),
|
|
435
|
-
allowed_workspaces=_allowed(spec),
|
|
661
|
+
spec["query"], top_k=spec.get("top_k", 20), alpha=spec.get("alpha"),
|
|
662
|
+
allowed_workspaces=_allowed(spec), min_vector_score=spec.get("min_vector", 0.0),
|
|
436
663
|
include_legacy_global=spec.get("legacy", False),
|
|
437
|
-
min_vector_score=spec.get("min_vector", 0.0),
|
|
438
664
|
),
|
|
439
665
|
"keyword": lambda store, spec: store.search(
|
|
440
|
-
spec["query"],
|
|
441
|
-
spec.get("limit", 30),
|
|
442
|
-
allowed_workspaces=_allowed(spec),
|
|
666
|
+
spec["query"], spec.get("limit", 30), allowed_workspaces=_allowed(spec),
|
|
443
667
|
include_legacy_global=spec.get("legacy", False),
|
|
444
668
|
),
|
|
445
669
|
"vector": lambda store, spec: store.vector_search(
|
|
@@ -454,6 +678,205 @@ def run_engine(store, engine: str, spec: Dict[str, Any]) -> Dict[str, Any]:
|
|
|
454
678
|
return ENGINES[engine](store, spec)
|
|
455
679
|
|
|
456
680
|
|
|
681
|
+
# ── suites (v11.5.0): KG reads, the service layer, history, context ──────────
|
|
682
|
+
#
|
|
683
|
+
# The Phase-1 engines share one query shape (a query set × an engine set); the
|
|
684
|
+
# Phase-2/3 ports do not, so each gets its own spec list and runner, both carried
|
|
685
|
+
# in the manifest so Rust and Python enumerate exactly the same work.
|
|
686
|
+
|
|
687
|
+
|
|
688
|
+
class Harness:
|
|
689
|
+
"""Every Python entry point the v11.5.0 goldens are produced from.
|
|
690
|
+
|
|
691
|
+
``require_auth=False`` is the loopback-owner configuration the native routes
|
|
692
|
+
reproduce: the history scope is whatever the caller passes explicitly.
|
|
693
|
+
"""
|
|
694
|
+
|
|
695
|
+
def __init__(self, db_path: Path):
|
|
696
|
+
from latticeai.runtime.history_runtime import build_history_query_runtime
|
|
697
|
+
from latticeai.services.chat_service import ChatService
|
|
698
|
+
from latticeai.services.search_service import SearchService
|
|
699
|
+
self.store = open_store(db_path)
|
|
700
|
+
self.service = SearchService(graph_store=self.store)
|
|
701
|
+
self.conversations = open_conversations(db_path)
|
|
702
|
+
self.history_runtime = build_history_query_runtime(
|
|
703
|
+
conversations=self.conversations,
|
|
704
|
+
workspace_service=None,
|
|
705
|
+
require_auth=False,
|
|
706
|
+
logging=logging,
|
|
707
|
+
)
|
|
708
|
+
self.chat = ChatService(store=None, get_history=self.history_runtime["get_history"])
|
|
709
|
+
|
|
710
|
+
|
|
711
|
+
def _allowed_list(spec: Dict[str, Any]):
|
|
712
|
+
"""``allowed`` as the graph layer wants it: ``None`` or a set."""
|
|
713
|
+
allowed = spec.get("allowed")
|
|
714
|
+
return None if allowed is None else set(allowed)
|
|
715
|
+
|
|
716
|
+
|
|
717
|
+
def _history_scope(spec: Dict[str, Any]) -> Dict[str, Any]:
|
|
718
|
+
"""The identity/workspace scope every history read takes.
|
|
719
|
+
|
|
720
|
+
``include_legacy_global`` defaults to ``True`` — ``ConversationStore``'s own
|
|
721
|
+
default, the opposite of the graph layer's and the kind of asymmetry a port
|
|
722
|
+
gets wrong.
|
|
723
|
+
"""
|
|
724
|
+
allowed = spec.get("allowed")
|
|
725
|
+
return {
|
|
726
|
+
"user_email": spec.get("user_email"),
|
|
727
|
+
"allowed_workspaces": None if allowed is None else list(allowed),
|
|
728
|
+
"include_legacy_global": spec.get("legacy", True),
|
|
729
|
+
}
|
|
730
|
+
|
|
731
|
+
|
|
732
|
+
def _run_relationship(h: Harness, spec: Dict[str, Any]) -> Dict[str, Any]:
|
|
733
|
+
return h.store.relationship_search(
|
|
734
|
+
query=spec.get("query", ""), node_id=spec.get("node_id", ""),
|
|
735
|
+
relationship_type=spec.get("relationship_type", ""), limit=spec.get("limit", 30),
|
|
736
|
+
allowed_workspaces=_allowed_list(spec),
|
|
737
|
+
include_legacy_global=spec.get("legacy", False),
|
|
738
|
+
)
|
|
739
|
+
|
|
740
|
+
|
|
741
|
+
def _run_traverse(h: Harness, spec: Dict[str, Any]) -> Dict[str, Any]:
|
|
742
|
+
try:
|
|
743
|
+
return h.store.traverse(
|
|
744
|
+
spec.get("node_id", ""), depth=spec.get("depth", 1),
|
|
745
|
+
limit=spec.get("limit", 100), allowed_workspaces=_allowed_list(spec),
|
|
746
|
+
include_legacy_global=spec.get("legacy", False),
|
|
747
|
+
)
|
|
748
|
+
except ValueError as exc:
|
|
749
|
+
# The two documented refusals (blank id, seed invisible to the caller's
|
|
750
|
+
# scope) are contract, so they are recorded rather than skipped; a payload
|
|
751
|
+
# never carries an "error" key, so the golden stays unambiguous.
|
|
752
|
+
return {"error": str(exc)}
|
|
753
|
+
|
|
754
|
+
|
|
755
|
+
def _run_graph_search(h: Harness, spec: Dict[str, Any]) -> Dict[str, Any]:
|
|
756
|
+
return h.service.graph_search(
|
|
757
|
+
spec["query"], limit=spec.get("limit", 30),
|
|
758
|
+
expand_depth=spec.get("expand_depth", 1), allowed_workspaces=_allowed_list(spec),
|
|
759
|
+
include_legacy_global=spec.get("legacy", False),
|
|
760
|
+
)
|
|
761
|
+
|
|
762
|
+
|
|
763
|
+
def _run_service_hybrid(h: Harness, spec: Dict[str, Any]) -> Dict[str, Any]:
|
|
764
|
+
return h.service.hybrid_search(
|
|
765
|
+
spec["query"], limit=spec.get("limit", 30),
|
|
766
|
+
keyword_limit=spec.get("keyword_limit", 30),
|
|
767
|
+
vector_limit=spec.get("vector_limit", 30),
|
|
768
|
+
graph_limit=spec.get("graph_limit", 30), weights=spec.get("weights"),
|
|
769
|
+
allowed_workspaces=_allowed_list(spec),
|
|
770
|
+
include_legacy_global=spec.get("legacy", False),
|
|
771
|
+
)
|
|
772
|
+
|
|
773
|
+
|
|
774
|
+
def _run_history(h: Harness, spec: Dict[str, Any]) -> List[Dict[str, Any]]:
|
|
775
|
+
return h.conversations.history(
|
|
776
|
+
conversation_id=spec.get("conversation_id"), limit=spec.get("limit"),
|
|
777
|
+
**_history_scope(spec),
|
|
778
|
+
)
|
|
779
|
+
|
|
780
|
+
|
|
781
|
+
def _run_conversations(h: Harness, spec: Dict[str, Any]) -> List[Dict[str, Any]]:
|
|
782
|
+
history = h.history_runtime["get_history"](**_history_scope(spec))
|
|
783
|
+
return h.history_runtime["group_history_conversations"](history)
|
|
784
|
+
|
|
785
|
+
|
|
786
|
+
def _run_conversation_messages(h: Harness, spec: Dict[str, Any]) -> List[Dict[str, Any]]:
|
|
787
|
+
return h.history_runtime["get_conversation_messages"](
|
|
788
|
+
spec["conversation_id"], **_history_scope(spec)
|
|
789
|
+
)
|
|
790
|
+
|
|
791
|
+
|
|
792
|
+
def _run_history_search(h: Harness, spec: Dict[str, Any]) -> List[Dict[str, Any]]:
|
|
793
|
+
return h.chat.search_history(
|
|
794
|
+
spec["query"], scope=_history_scope(spec), limit=spec.get("limit", 30),
|
|
795
|
+
conversation_title=h.history_runtime["conversation_title"],
|
|
796
|
+
)
|
|
797
|
+
|
|
798
|
+
|
|
799
|
+
def _context_seams(h: Harness, spec: Dict[str, Any]) -> Dict[str, Any]:
|
|
800
|
+
"""The seam set for one context spec — data seams plus the real engines.
|
|
801
|
+
|
|
802
|
+
``memories`` / ``artifacts`` / ``notes`` are *data* seams: the payload is the
|
|
803
|
+
spec, so both runtimes feed the assembler the same bytes and what is under
|
|
804
|
+
test is the assembler. ``knowledge`` and ``recent`` are real — the
|
|
805
|
+
service-layer hybrid search and the durable history reader.
|
|
806
|
+
"""
|
|
807
|
+
from latticeai.api.chat_helpers import build_recent_chat_context
|
|
808
|
+
|
|
809
|
+
# Signatures matter: the assembler inspects them to decide which context
|
|
810
|
+
# fields a seam may be handed, so each one declares exactly what it accepts.
|
|
811
|
+
seams: Dict[str, Any] = {}
|
|
812
|
+
if spec.get("memories") is not None:
|
|
813
|
+
memories = spec["memories"]
|
|
814
|
+
seams["memory_recall"] = (
|
|
815
|
+
lambda query, *, user_email=None, workspace_id=None, limit=5: memories
|
|
816
|
+
)
|
|
817
|
+
if spec.get("artifacts") is not None:
|
|
818
|
+
artifacts = spec["artifacts"]
|
|
819
|
+
seams["recent_artifacts"] = (
|
|
820
|
+
lambda *, user_email=None, conversation_id=None, workspace_id=None: artifacts
|
|
821
|
+
)
|
|
822
|
+
if spec.get("knowledge", True):
|
|
823
|
+
# Loopback trust: no workspace scoping, exactly as on the native route.
|
|
824
|
+
seams["hybrid_search"] = (
|
|
825
|
+
lambda query, *, limit=5, user_email=None, workspace_id=None:
|
|
826
|
+
h.service.hybrid_search(query, limit=limit)
|
|
827
|
+
)
|
|
828
|
+
if spec.get("notes") is not None:
|
|
829
|
+
notes = spec["notes"]
|
|
830
|
+
seams["notes_context"] = lambda query, *, user_email=None, workspace_id=None: notes
|
|
831
|
+
if spec.get("recent") is not None:
|
|
832
|
+
recent = spec["recent"]
|
|
833
|
+
seams["recent_chat"] = (
|
|
834
|
+
lambda *, user_email=None, conversation_id=None, workspace_id=None:
|
|
835
|
+
build_recent_chat_context(
|
|
836
|
+
get_history=h.history_runtime["get_history"],
|
|
837
|
+
limit=recent.get("limit", 10),
|
|
838
|
+
include_image_missing_replies=recent.get("images", True),
|
|
839
|
+
user_email=recent.get("user_email"),
|
|
840
|
+
conversation_id=recent.get("conversation_id"),
|
|
841
|
+
workspace_id=recent.get("workspace_id"),
|
|
842
|
+
)
|
|
843
|
+
)
|
|
844
|
+
return seams
|
|
845
|
+
|
|
846
|
+
|
|
847
|
+
def _run_context_assemble(h: Harness, spec: Dict[str, Any]) -> Dict[str, Any]:
|
|
848
|
+
from lattice_brain.context import ContextAssembler
|
|
849
|
+
|
|
850
|
+
assembled = ContextAssembler(**_context_seams(h, spec)).assemble(
|
|
851
|
+
spec["query"], user_email=spec.get("user_email"),
|
|
852
|
+
workspace_id=spec.get("workspace_id"),
|
|
853
|
+
conversation_id=spec.get("conversation_id"), budget=spec.get("budget", 2000),
|
|
854
|
+
memory_limit=spec.get("memory_limit", 5),
|
|
855
|
+
knowledge_limit=spec.get("knowledge_limit", 5),
|
|
856
|
+
)
|
|
857
|
+
return {"text": assembled.text, "approx_tokens": assembled.approx_tokens,
|
|
858
|
+
"trace": assembled.trace()}
|
|
859
|
+
|
|
860
|
+
|
|
861
|
+
SUITE_RUNNERS: Dict[str, Callable[[Harness, Dict[str, Any]], Any]] = {
|
|
862
|
+
"relationship": _run_relationship,
|
|
863
|
+
"traverse": _run_traverse,
|
|
864
|
+
"graph_search": _run_graph_search,
|
|
865
|
+
"service_hybrid": _run_service_hybrid,
|
|
866
|
+
"history": _run_history,
|
|
867
|
+
"conversations": _run_conversations,
|
|
868
|
+
"conversation_messages": _run_conversation_messages,
|
|
869
|
+
"history_search": _run_history_search,
|
|
870
|
+
"context_assemble": _run_context_assemble,
|
|
871
|
+
}
|
|
872
|
+
|
|
873
|
+
|
|
874
|
+
def run_suite(harness: Harness, suite: str, spec: Dict[str, Any]) -> Any:
|
|
875
|
+
"""Run one suite spec under the frozen clock and the rule-based extractor."""
|
|
876
|
+
with frozen_clock(), rules_only_extraction():
|
|
877
|
+
return SUITE_RUNNERS[suite](harness, spec)
|
|
878
|
+
|
|
879
|
+
|
|
457
880
|
def golden_path(engine: str, key: str) -> Path:
|
|
458
881
|
return GOLDEN_DIR / f"{engine}__{key}.json"
|
|
459
882
|
|
|
@@ -464,10 +887,8 @@ def golden_payload(engine: str, spec: Dict[str, Any], result: Dict[str, Any]) ->
|
|
|
464
887
|
"key": spec["key"],
|
|
465
888
|
"query": spec["query"],
|
|
466
889
|
"params": {
|
|
467
|
-
"top_k": spec.get("top_k", 20),
|
|
468
|
-
"
|
|
469
|
-
"limit": spec.get("limit", 30),
|
|
470
|
-
"min_score": spec.get("min_score", 0.0),
|
|
890
|
+
"top_k": spec.get("top_k", 20), "alpha": spec.get("alpha"),
|
|
891
|
+
"limit": spec.get("limit", 30), "min_score": spec.get("min_score", 0.0),
|
|
471
892
|
"min_vector_score": spec.get("min_vector", 0.0),
|
|
472
893
|
"allowed_workspaces": spec.get("allowed"),
|
|
473
894
|
"include_legacy_global": spec.get("legacy", False),
|
|
@@ -476,6 +897,15 @@ def golden_payload(engine: str, spec: Dict[str, Any], result: Dict[str, Any]) ->
|
|
|
476
897
|
}
|
|
477
898
|
|
|
478
899
|
|
|
900
|
+
def suite_payload(suite: str, spec: Dict[str, Any], result: Any) -> Dict[str, Any]:
|
|
901
|
+
"""One suite golden: the spec that produced it, verbatim, and the answer.
|
|
902
|
+
|
|
903
|
+
The spec rides along rather than being flattened into a fixed ``params``
|
|
904
|
+
block: these entry points share no parameter shape.
|
|
905
|
+
"""
|
|
906
|
+
return {"suite": suite, "key": spec["key"], "spec": spec, "result": result}
|
|
907
|
+
|
|
908
|
+
|
|
479
909
|
def _dump(path: Path, payload: Any) -> None:
|
|
480
910
|
path.write_text(
|
|
481
911
|
json.dumps(payload, ensure_ascii=False, sort_keys=True, indent=2) + "\n",
|
|
@@ -511,12 +941,11 @@ def embeddings_golden() -> Dict[str, Any]:
|
|
|
511
941
|
|
|
512
942
|
def rounding_golden() -> List[Dict[str, float]]:
|
|
513
943
|
"""``round(x, 6)`` for values where the tie rule is observable."""
|
|
514
|
-
return [{"input":
|
|
944
|
+
return [{"input": v, "expected": round(v, 6)} for v in ROUNDING_VALUES]
|
|
515
945
|
|
|
516
946
|
|
|
517
947
|
def manifest() -> Dict[str, Any]:
|
|
518
948
|
from lattice_brain.embeddings import LocalEmbeddingModel
|
|
519
|
-
|
|
520
949
|
model = LocalEmbeddingModel()
|
|
521
950
|
return {
|
|
522
951
|
"frozen_now": FROZEN_NOW,
|
|
@@ -525,6 +954,7 @@ def manifest() -> Dict[str, Any]:
|
|
|
525
954
|
"embedding_dim": model.dim,
|
|
526
955
|
"engines": sorted(ENGINES),
|
|
527
956
|
"queries": QUERIES,
|
|
957
|
+
"suites": {suite: SUITES[suite] for suite in sorted(SUITES)},
|
|
528
958
|
"pinned_env": PINNED_ENV,
|
|
529
959
|
}
|
|
530
960
|
|
|
@@ -543,12 +973,17 @@ def main() -> int:
|
|
|
543
973
|
result = run_engine(store, engine, spec)
|
|
544
974
|
_dump(golden_path(engine, spec["key"]), golden_payload(engine, spec, result))
|
|
545
975
|
written += 1
|
|
976
|
+
harness = Harness(STORE_PATH)
|
|
977
|
+
for suite in sorted(SUITES):
|
|
978
|
+
for spec in SUITES[suite]:
|
|
979
|
+
result = run_suite(harness, suite, spec)
|
|
980
|
+
_dump(golden_path(suite, spec["key"]), suite_payload(suite, spec, result))
|
|
981
|
+
written += 1
|
|
546
982
|
_dump(GOLDEN_DIR / "embeddings_golden.json", embeddings_golden())
|
|
547
983
|
_dump(GOLDEN_DIR / "rounding_golden.json", rounding_golden())
|
|
548
984
|
_dump(GOLDEN_DIR / "manifest.json", manifest())
|
|
549
|
-
# ``KnowledgeGraphStore.__init__`` creates its blob directory eagerly
|
|
550
|
-
# fixture has no blobs
|
|
551
|
-
# just a thing for the next reader to wonder about.
|
|
985
|
+
# ``KnowledgeGraphStore.__init__`` creates its blob directory eagerly and the
|
|
986
|
+
# fixture has no blobs; an empty directory beside an artefact is just noise.
|
|
552
987
|
blob_dir = STORE_PATH.parent / "blobs"
|
|
553
988
|
if blob_dir.is_dir() and not any(blob_dir.iterdir()):
|
|
554
989
|
blob_dir.rmdir()
|