dsh-prime-memory 0.11.0 → 0.12.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.en.md +36 -1
- package/CHANGELOG.ja.md +36 -1
- package/CHANGELOG.ko.md +36 -1
- package/CHANGELOG.md +40 -0
- package/README.en.md +28 -0
- package/README.ja.md +25 -0
- package/README.ko.md +25 -0
- package/README.md +22 -0
- package/dist/client.js +54 -30
- package/dist/config.d.ts +39 -1
- package/dist/config.js +16 -0
- package/dist/conflict-service.d.ts +38 -0
- package/dist/conflict-service.js +65 -0
- package/dist/graph/search.d.ts +15 -0
- package/dist/graph/search.js +54 -1
- package/dist/hooks/recall.js +4 -0
- package/dist/index.d.ts +28 -1
- package/dist/index.js +12 -3
- package/dist/pipeline/l1.d.ts +7 -1
- package/dist/pipeline/l1.js +165 -20
- package/dist/pipeline/runner.js +6 -1
- package/dist/prompts/l1-dedup.d.ts +22 -1
- package/dist/prompts/l1-dedup.js +61 -4
- package/dist/stats.d.ts +12 -0
- package/dist/stats.js +222 -33
- package/dist/store/conflicts.d.ts +62 -0
- package/dist/store/conflicts.js +69 -0
- package/dist/store/graph-store.d.ts +72 -1
- package/dist/store/graph-store.js +165 -1
- package/dist/store/l1.d.ts +97 -4
- package/dist/store/l1.js +163 -21
- package/dist/store/receipts.d.ts +157 -0
- package/dist/store/receipts.js +139 -0
- package/dist/store/search-utils.d.ts +15 -0
- package/dist/store/search-utils.js +20 -0
- package/dist/store/session-modes.d.ts +7 -0
- package/dist/store/session-modes.js +9 -0
- package/dist/store/sqlite.d.ts +116 -8
- package/dist/store/sqlite.js +366 -46
- package/dist/store/vec-utils.d.ts +16 -0
- package/dist/store/vec-utils.js +24 -0
- package/dist/tools/index.js +204 -31
- package/dist/types.d.ts +48 -0
- package/dist/types.js +40 -0
- package/dist/workspace.d.ts +46 -0
- package/dist/workspace.js +105 -0
- package/package.json +9 -3
package/dist/store/sqlite.js
CHANGED
|
@@ -20,7 +20,9 @@
|
|
|
20
20
|
import { createRequire } from 'node:module';
|
|
21
21
|
import { existsSync, mkdirSync } from 'node:fs';
|
|
22
22
|
import * as path from 'node:path';
|
|
23
|
-
import { familyForType, normPersistence } from '../types.js';
|
|
23
|
+
import { familyForType, isScopeVisible, normPersistence, normScope } from '../types.js';
|
|
24
|
+
import { isZeroVector, vecToBuffer } from './vec-utils.js';
|
|
25
|
+
import { normalizeWorkspacePath } from '../workspace.js';
|
|
24
26
|
import { bm25RankToScore, buildFtsQuery, tokenizeForFts } from './search-utils.js';
|
|
25
27
|
import { describeTokenizer, ensureTokenizer, tokenizerStamp } from '../util/tokenizer.js';
|
|
26
28
|
const require = createRequire(import.meta.url);
|
|
@@ -30,6 +32,7 @@ const TAG = '[memory][sqlite]';
|
|
|
30
32
|
import { CostLedger } from './cost-ledger.js';
|
|
31
33
|
// 图谱存储(graph_* 表族)同为独立职责类;init 失败仅图谱 no-op,不传染主库降级
|
|
32
34
|
import { GraphStore } from './graph-store.js';
|
|
35
|
+
import { RECEIPTS_MAX_RUNS, RECEIPTS_QUERY_LIMIT_MAX } from './receipts.js';
|
|
33
36
|
/** vec0 KNN 对遗留零向量的补偿缓冲。 */
|
|
34
37
|
const ZERO_VEC_BUFFER = 10;
|
|
35
38
|
/** IN 查询/删除的分块大小(保守避开 SQLite 变量数上限:现代构建 32766,老版 999)。 */
|
|
@@ -285,7 +288,9 @@ export class MemoryDb {
|
|
|
285
288
|
created_time TEXT DEFAULT '',
|
|
286
289
|
updated_time TEXT DEFAULT '',
|
|
287
290
|
metadata_json TEXT DEFAULT '{}',
|
|
288
|
-
family TEXT NOT NULL DEFAULT 'chat'
|
|
291
|
+
family TEXT NOT NULL DEFAULT 'chat',
|
|
292
|
+
scope TEXT NOT NULL DEFAULT 'global',
|
|
293
|
+
workspace_id TEXT NOT NULL DEFAULT ''
|
|
289
294
|
)
|
|
290
295
|
`);
|
|
291
296
|
// 旧库缺 family 列 → ALTER 补列,并按 type 前缀回填(幂等:已正确的行不再命中)
|
|
@@ -293,6 +298,14 @@ export class MemoryDb {
|
|
|
293
298
|
this.db.exec("ALTER TABLE l1_records ADD COLUMN family TEXT NOT NULL DEFAULT 'chat'");
|
|
294
299
|
this.logger?.info(`${TAG} l1_records 补 family 列(旧数据按 type 前缀回填)`);
|
|
295
300
|
}
|
|
301
|
+
// §E 可见范围(ADR-0008 条 3:向后兼容是硬要求)。**既有数据零搬运**——
|
|
302
|
+
// 补列的 DEFAULT 本身就把存量行标成 `global`,不需要 UPDATE 扫描,
|
|
303
|
+
// 也不删除任何行:"标注归属"而非"搬运/重建",免得拿事实源冒险换配置项的美观。
|
|
304
|
+
if (!this.hasColumn('l1_records', 'scope')) {
|
|
305
|
+
this.db.exec("ALTER TABLE l1_records ADD COLUMN scope TEXT NOT NULL DEFAULT 'global'");
|
|
306
|
+
this.db.exec("ALTER TABLE l1_records ADD COLUMN workspace_id TEXT NOT NULL DEFAULT ''");
|
|
307
|
+
this.logger?.info(`${TAG} l1_records 补 scope/workspace_id 列(存量数据默认归 global)`);
|
|
308
|
+
}
|
|
296
309
|
const backfilled = this.db
|
|
297
310
|
.prepare("UPDATE l1_records SET family = 'work' WHERE type LIKE 'work\\_%' ESCAPE '\\' AND family != 'work'")
|
|
298
311
|
.run().changes;
|
|
@@ -305,12 +318,52 @@ export class MemoryDb {
|
|
|
305
318
|
this.db.exec('CREATE INDEX IF NOT EXISTS idx_l1_updated ON l1_records(updated_time)');
|
|
306
319
|
this.db.exec('CREATE INDEX IF NOT EXISTS idx_l1_family ON l1_records(family)');
|
|
307
320
|
this.db.exec('CREATE INDEX IF NOT EXISTS idx_l1_valid_from ON l1_records(valid_from)');
|
|
321
|
+
// §E:workspace 过滤的唯一命中路径就是本列(scope='global' 的行走 OR 短路,不依赖索引)
|
|
322
|
+
this.db.exec('CREATE INDEX IF NOT EXISTS idx_l1_workspace ON l1_records(workspace_id)');
|
|
323
|
+
// ── §B L1 决策凭证(DDL 同为磁盘契约) ──
|
|
324
|
+
// 每条 L1 记录的 store/update/merge/skip 决策留一条凭证:决策当时看到的候选池
|
|
325
|
+
// 摘要(input_digest)+ 结论(kind)。凭证必须在事件**之前**存在——输入快照
|
|
326
|
+
// 无法事后补录,故本表先于任何消费方落地(findings.md §9)。
|
|
327
|
+
this.db.exec(`
|
|
328
|
+
CREATE TABLE IF NOT EXISTS l1_receipts (
|
|
329
|
+
receipt_id TEXT PRIMARY KEY,
|
|
330
|
+
run_id TEXT NOT NULL DEFAULT '',
|
|
331
|
+
record_id TEXT NOT NULL DEFAULT '',
|
|
332
|
+
kind TEXT NOT NULL DEFAULT '',
|
|
333
|
+
input_digest TEXT NOT NULL DEFAULT '',
|
|
334
|
+
decided_at TEXT NOT NULL DEFAULT ''
|
|
335
|
+
)
|
|
336
|
+
`);
|
|
337
|
+
// 双维回溯(task_19):按批(run_id)看一轮蒸馏的全部决策;按记录(record_id 看单条记忆的完整判定史
|
|
338
|
+
this.db.exec('CREATE INDEX IF NOT EXISTS idx_l1_receipts_run ON l1_receipts(run_id)');
|
|
339
|
+
this.db.exec('CREATE INDEX IF NOT EXISTS idx_l1_receipts_record ON l1_receipts(record_id)');
|
|
340
|
+
// ── §C 矛盾冻结:(DDL 同为磁盘契约) ──
|
|
341
|
+
// 冻结**不是"拦住写入"**:新记忆照常入 l1_records,与冲突的旧记忆作为**一对**
|
|
342
|
+
// 停放在本表,双方内容都不被改写,直到人工裁决。LLM 给的 winner_id 只表示
|
|
343
|
+
// "进入待裁决对时的排序位",**不代表最终结论**——最终结论落在 resolution。
|
|
344
|
+
// resolved_at 用 '' 而非 NULL 表示未裁决,与 l1_receipts 的约定一致,
|
|
345
|
+
// 避免 `= ''` 与 `IS NULL` 两套判据并存。
|
|
346
|
+
this.db.exec(`
|
|
347
|
+
CREATE TABLE IF NOT EXISTS conflict_pending (
|
|
348
|
+
pair_id TEXT PRIMARY KEY,
|
|
349
|
+
run_id TEXT NOT NULL DEFAULT '',
|
|
350
|
+
winner_id TEXT NOT NULL DEFAULT '',
|
|
351
|
+
loser_id TEXT NOT NULL DEFAULT '',
|
|
352
|
+
created_at TEXT NOT NULL DEFAULT '',
|
|
353
|
+
resolved_at TEXT NOT NULL DEFAULT '',
|
|
354
|
+
resolution TEXT NOT NULL DEFAULT ''
|
|
355
|
+
)
|
|
356
|
+
`);
|
|
357
|
+
// 未裁决索引:队列上限(task_24,取最旧的未裁决行)与裁决工具(task_25,取单条未裁决对)
|
|
358
|
+
// 都只关心未裁决行——"查未裁决"须走索引。偏索引同时覆盖 created_at 排序。
|
|
359
|
+
this.db.exec(`CREATE INDEX IF NOT EXISTS idx_conflict_pending_unresolved
|
|
360
|
+
ON conflict_pending(created_at) WHERE resolved_at = ''`);
|
|
308
361
|
this.stmtUpsertL1 = this.db.prepare(`
|
|
309
362
|
INSERT INTO l1_records (
|
|
310
363
|
record_id, content, type, priority, scene_name, session_id, version,
|
|
311
364
|
timestamp_str, timestamp_start, timestamp_end, created_time, updated_time, metadata_json, family,
|
|
312
|
-
valid_from, valid_to, persistence
|
|
313
|
-
) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
|
|
365
|
+
valid_from, valid_to, persistence, scope, workspace_id
|
|
366
|
+
) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
|
|
314
367
|
ON CONFLICT(record_id) DO UPDATE SET
|
|
315
368
|
content=excluded.content,
|
|
316
369
|
type=excluded.type,
|
|
@@ -325,12 +378,14 @@ export class MemoryDb {
|
|
|
325
378
|
family=excluded.family,
|
|
326
379
|
valid_from=excluded.valid_from,
|
|
327
380
|
valid_to=excluded.valid_to,
|
|
328
|
-
persistence=excluded.persistence
|
|
381
|
+
persistence=excluded.persistence,
|
|
382
|
+
scope=excluded.scope,
|
|
383
|
+
workspace_id=excluded.workspace_id
|
|
329
384
|
`);
|
|
330
385
|
this.stmtGetL1 = this.db.prepare(`
|
|
331
386
|
SELECT record_id, content, type, priority, scene_name, version, timestamp_str,
|
|
332
387
|
timestamp_start, timestamp_end, created_time, updated_time, metadata_json, family,
|
|
333
|
-
valid_from, valid_to, persistence
|
|
388
|
+
valid_from, valid_to, persistence, scope, workspace_id
|
|
334
389
|
FROM l1_records WHERE record_id = ?
|
|
335
390
|
`);
|
|
336
391
|
this.stmtL1Exists = this.db.prepare('SELECT 1 FROM l1_records WHERE record_id = ?');
|
|
@@ -366,7 +421,10 @@ export class MemoryDb {
|
|
|
366
421
|
// ── token_cost:蒸馏成本明细表(成本账本自治) ──
|
|
367
422
|
this.costLedger.init(this.db, this.logger);
|
|
368
423
|
// ── graph_*:知识图谱投影表族(GraphStore.init 自带 try/catch,失败仅图谱 no-op) ──
|
|
369
|
-
|
|
424
|
+
// §F 节点向量列的维度**复用既有探测结果**(`vecLoaded` / `dimensions` 由
|
|
425
|
+
// `prepareL1VecStatements` 之前的探测决定),不新增一套能力探测。探测未通过 →
|
|
426
|
+
// 不传 → 图谱向量路结构性不存在(不建表、不告警、不抛)。
|
|
427
|
+
this.graphStore.init(this.db, this.logger, this.vecLoaded && this.dimensions > 0 ? { dimensions: this.dimensions } : undefined);
|
|
370
428
|
// ── FTS5 全文索引(建表失败仅停用 FTS,不降级整个库) ──
|
|
371
429
|
try {
|
|
372
430
|
// 索引重建判据(FTS5 无法 ALTER,只能 drop 后从源表全量回灌):
|
|
@@ -376,10 +434,10 @@ export class MemoryDb {
|
|
|
376
434
|
const savedStamp = this.readMetaString('fts_tokenizer') ?? 'bigram-v1';
|
|
377
435
|
const tokenizerChanged = savedStamp !== wantStamp;
|
|
378
436
|
let ftsRebuilt = false;
|
|
379
|
-
if (this.tableExists('l1_fts') && (!this.hasColumn('l1_fts', 'family') || tokenizerChanged)) {
|
|
437
|
+
if (this.tableExists('l1_fts') && (!this.hasColumn('l1_fts', 'family') || !this.hasColumn('l1_fts', 'scope') || tokenizerChanged)) {
|
|
380
438
|
this.db.exec('DROP TABLE l1_fts');
|
|
381
439
|
ftsRebuilt = true;
|
|
382
|
-
this.logger?.info(`${TAG} l1_fts 缺 family 列或分词器已变更(${savedStamp} → ${wantStamp}),重建全文索引`);
|
|
440
|
+
this.logger?.info(`${TAG} l1_fts 缺 family/scope 列或分词器已变更(${savedStamp} → ${wantStamp}),重建全文索引`);
|
|
383
441
|
}
|
|
384
442
|
let l0FtsRebuilt = false;
|
|
385
443
|
if (this.tableExists('l0_fts') && tokenizerChanged) {
|
|
@@ -401,7 +459,9 @@ export class MemoryDb {
|
|
|
401
459
|
timestamp_start UNINDEXED,
|
|
402
460
|
timestamp_end UNINDEXED,
|
|
403
461
|
metadata_json UNINDEXED,
|
|
404
|
-
family UNINDEXED
|
|
462
|
+
family UNINDEXED,
|
|
463
|
+
scope UNINDEXED,
|
|
464
|
+
workspace_id UNINDEXED
|
|
405
465
|
)
|
|
406
466
|
`);
|
|
407
467
|
this.db.exec(`
|
|
@@ -417,26 +477,33 @@ export class MemoryDb {
|
|
|
417
477
|
`);
|
|
418
478
|
this.stmtL1FtsInsert = this.db.prepare(`
|
|
419
479
|
INSERT INTO l1_fts (content, content_original, record_id, type, priority, scene_name,
|
|
420
|
-
session_id, version, timestamp_str, timestamp_start, timestamp_end, metadata_json, family
|
|
421
|
-
|
|
480
|
+
session_id, version, timestamp_str, timestamp_start, timestamp_end, metadata_json, family,
|
|
481
|
+
scope, workspace_id)
|
|
482
|
+
VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
|
|
422
483
|
`);
|
|
423
484
|
this.stmtL1FtsDelete = this.db.prepare('DELETE FROM l1_fts WHERE record_id = ?');
|
|
485
|
+
// §E 可见范围过滤的**唯一缝**。设计要点:
|
|
486
|
+
// ① `? = ''` 是"不过滤"的哨兵——检索侧不传工作区标识(即 cfg.scope='global')时
|
|
487
|
+
// 整个条件短路为真,行为与改动前逐字一致(**零漂移不是比对出来的,是构造出来的**);
|
|
488
|
+
// ② 过滤表达式与 family 落在**同一条语句**里(ADR-0008 组合关系:scope 过滤必须与
|
|
489
|
+
// 族隔离同层),否则会产出"看不见但已影响决策"的记忆——去重候选召回也走这里;
|
|
490
|
+
// ③ `scope = 'global'` 分支让跨工作区可见的记忆在任何工作区都能被召回。
|
|
424
491
|
this.stmtL1FtsSearch = this.db.prepare(`
|
|
425
492
|
SELECT record_id, content_original AS content, type, priority, scene_name, version,
|
|
426
|
-
timestamp_str, timestamp_start, timestamp_end, metadata_json, family,
|
|
493
|
+
timestamp_str, timestamp_start, timestamp_end, metadata_json, family, scope, workspace_id,
|
|
427
494
|
bm25(l1_fts) AS rank
|
|
428
495
|
FROM l1_fts
|
|
429
|
-
WHERE l1_fts MATCH ?
|
|
496
|
+
WHERE l1_fts MATCH ? AND (? = '' OR scope = 'global' OR workspace_id = ?)
|
|
430
497
|
ORDER BY rank ASC
|
|
431
498
|
LIMIT ?
|
|
432
499
|
`);
|
|
433
500
|
// 族过滤版(FTS5 UNINDEXED 列可作行级过滤条件)
|
|
434
501
|
this.stmtL1FtsSearchFamily = this.db.prepare(`
|
|
435
502
|
SELECT record_id, content_original AS content, type, priority, scene_name, version,
|
|
436
|
-
timestamp_str, timestamp_start, timestamp_end, metadata_json, family,
|
|
503
|
+
timestamp_str, timestamp_start, timestamp_end, metadata_json, family, scope, workspace_id,
|
|
437
504
|
bm25(l1_fts) AS rank
|
|
438
505
|
FROM l1_fts
|
|
439
|
-
WHERE l1_fts MATCH ? AND family = ?
|
|
506
|
+
WHERE l1_fts MATCH ? AND family = ? AND (? = '' OR scope = 'global' OR workspace_id = ?)
|
|
440
507
|
ORDER BY rank ASC
|
|
441
508
|
LIMIT ?
|
|
442
509
|
`);
|
|
@@ -565,15 +632,26 @@ export class MemoryDb {
|
|
|
565
632
|
this.logger?.info(`${TAG} l1_records 补 ${column} 列(时间增强)`);
|
|
566
633
|
}
|
|
567
634
|
}
|
|
568
|
-
/**
|
|
635
|
+
/**
|
|
636
|
+
* 重建后的 l1_fts 从 l1_records 全量回灌(仅在 drop 重建时调用;iterate 流式防大库内存峰值)。
|
|
637
|
+
*
|
|
638
|
+
* ⚠️ 本函数的参数列表**必须与 `stmtL1FtsInsert` 逐位对齐**。不对齐时 node:sqlite 会在这里抛错,
|
|
639
|
+
* 而下面的 `catch` 是**逐行吞掉**的——症状是 `count` 停在 0、索引静默变空,
|
|
640
|
+
* 全库记录从此全文检索不可见却没有任何错误日志。§E 加列时正是这个位置最容易漏
|
|
641
|
+
* (三处列清单:DDL / insert 语句 / 本函数),故在此留下警示。
|
|
642
|
+
*/
|
|
569
643
|
backfillL1Fts() {
|
|
570
644
|
let count = 0;
|
|
571
645
|
const stmt = this.db
|
|
572
646
|
.prepare(`SELECT record_id, content, type, priority, scene_name, session_id, version,
|
|
573
|
-
timestamp_str, timestamp_start, timestamp_end, metadata_json, family
|
|
647
|
+
timestamp_str, timestamp_start, timestamp_end, metadata_json, family,
|
|
648
|
+
scope, workspace_id FROM l1_records`);
|
|
574
649
|
for (const r of stmt.iterate()) {
|
|
575
650
|
try {
|
|
576
|
-
this.stmtL1FtsInsert.run(tokenizeForFts(String(r.content ?? '')), String(r.content ?? ''), String(r.record_id ?? ''), String(r.type ?? ''), Number(r.priority ?? 50), String(r.scene_name ?? ''), String(r.session_id ?? 'default'), Number(r.version ?? 0), String(r.timestamp_str ?? ''), String(r.timestamp_start ?? ''), String(r.timestamp_end ?? ''), String(r.metadata_json ?? '{}'), String(r.family ?? 'chat')
|
|
651
|
+
this.stmtL1FtsInsert.run(tokenizeForFts(String(r.content ?? '')), String(r.content ?? ''), String(r.record_id ?? ''), String(r.type ?? ''), Number(r.priority ?? 50), String(r.scene_name ?? ''), String(r.session_id ?? 'default'), Number(r.version ?? 0), String(r.timestamp_str ?? ''), String(r.timestamp_start ?? ''), String(r.timestamp_end ?? ''), String(r.metadata_json ?? '{}'), String(r.family ?? 'chat'),
|
|
652
|
+
// §E:回灌必须带上可见范围——漏掉这两列等于**每次 FTS 重建都把隔离抹平**
|
|
653
|
+
// (所有行回落 'global',跨工作区记忆瞬间互相可见),且没有任何报错。
|
|
654
|
+
normScope(r.scope), String(r.workspace_id ?? ''));
|
|
577
655
|
count++;
|
|
578
656
|
}
|
|
579
657
|
catch {
|
|
@@ -712,12 +790,26 @@ export class MemoryDb {
|
|
|
712
790
|
const priority = record.priority ?? 50;
|
|
713
791
|
const sceneName = record.scene_name ?? '';
|
|
714
792
|
const family = record.family ?? familyForType(type);
|
|
793
|
+
// §E 写入侧兜底归一,并维持一条**不变量**:`scope='global'` 的记录 `workspace_id` 恒为空串。
|
|
794
|
+
// 不维持它就会出现"标着 global 却带着工作区归属"的行——语义含糊,且日后改判定时
|
|
795
|
+
// 无法区分"全局可见但顺带记了个 id"与"其实属于某工作区"。写入侧算好归属
|
|
796
|
+
// (pipeline 用 `resolveRecordScope`),这里只保证不变量,不重新决策。
|
|
797
|
+
//
|
|
798
|
+
// ⚠️ **形态归一必须在这里做**(不是"顺便"):检索侧传入的标识经 `workspaceIdOf`
|
|
799
|
+
// 归一(Windows 转小写、resolve 掉 `..`),而写入侧若原样存调用方给的字符串,
|
|
800
|
+
// 同一个工作区会以两种拼写落库 → `isScopeVisible` 的字符串相等判定必然落空
|
|
801
|
+
// → **记忆写进去却再也查不出来**。这一条曾被端到端测试抓出
|
|
802
|
+
// (`tests/scope-tool-wiring.test.ts` 的 `expected 1 to be 2`),
|
|
803
|
+
// 当时的实现只做了"清空"归一而漏了"形态"归一。收在存储层是因为调用方不止一个
|
|
804
|
+
// (pipeline / memory_add / 外部导入),逐个记得归一迟早漏一个。
|
|
805
|
+
const scope = normScope(record.scope);
|
|
806
|
+
const workspaceId = scope === 'workspace' ? (normalizeWorkspacePath(record.workspaceId) ?? '') : '';
|
|
715
807
|
// 防御性 FTS 删除的前置点查(主键索引,微秒级):record_id 在 FTS 表是 UNINDEXED,
|
|
716
808
|
// 按 id DELETE 是 O(N) 全表扫描——导入/重建/重嵌等"全新增"路径曾为每条记录白付一次
|
|
717
809
|
// 全扫(批量写整体 O(N²))。只有主表已有该行(覆盖/合并)才可能有旧 FTS 行需要删。
|
|
718
810
|
// 同批重复 id 也能正确处理:首条插入后,第二条的点查在同一事务内已见新行。
|
|
719
811
|
const ftsExisted = this.ftsAvailable ? this.stmtL1Exists.get(record.id) !== undefined : false;
|
|
720
|
-
this.stmtUpsertL1.run(record.id, record.content, type, priority, sceneName, record.sessionId ?? 'default', record.version ?? 0, ts.str, ts.start, ts.end, toIso(record.createdAt), toIso(record.updatedAt), JSON.stringify(record.metadata ?? {}), family, toIso(record.validFrom), toIso(record.validTo), normPersistence(record.persistence) ?? '');
|
|
812
|
+
this.stmtUpsertL1.run(record.id, record.content, type, priority, sceneName, record.sessionId ?? 'default', record.version ?? 0, ts.str, ts.start, ts.end, toIso(record.createdAt), toIso(record.updatedAt), JSON.stringify(record.metadata ?? {}), family, toIso(record.validFrom), toIso(record.validTo), normPersistence(record.persistence) ?? '', scope, workspaceId);
|
|
721
813
|
// vec0 不支持 ON CONFLICT → 先删后插;零向量跳过(cosine 未定义)
|
|
722
814
|
if (this.stmtDeleteL1Vec && this.stmtInsertL1Vec) {
|
|
723
815
|
this.stmtDeleteL1Vec.run(record.id);
|
|
@@ -730,7 +822,7 @@ export class MemoryDb {
|
|
|
730
822
|
if (this.ftsAvailable) {
|
|
731
823
|
if (ftsExisted)
|
|
732
824
|
this.stmtL1FtsDelete.run(record.id);
|
|
733
|
-
this.stmtL1FtsInsert.run(tokenizeForFts(record.content), record.content, record.id, type, priority, sceneName, record.sessionId ?? 'default', record.version ?? 0, ts.str, ts.start, ts.end, JSON.stringify(record.metadata ?? {}), family);
|
|
825
|
+
this.stmtL1FtsInsert.run(tokenizeForFts(record.content), record.content, record.id, type, priority, sceneName, record.sessionId ?? 'default', record.version ?? 0, ts.str, ts.start, ts.end, JSON.stringify(record.metadata ?? {}), family, scope, workspaceId);
|
|
734
826
|
}
|
|
735
827
|
}
|
|
736
828
|
/** 批量删除 L1(元数据 + 向量 + FTS),返回删除条数。IN 按 ≤900 分块(避变量数上限)。
|
|
@@ -769,8 +861,11 @@ export class MemoryDb {
|
|
|
769
861
|
const temporal = table === 'l1_records' && this.hasColumn('l1_records', 'valid_from')
|
|
770
862
|
? ', valid_from, valid_to, persistence'
|
|
771
863
|
: '';
|
|
864
|
+
// §E:同款按形状探测——未迁移的旧库(补列前)不应因缺列让"按 id 取记录"整条路径失败。
|
|
865
|
+
const scopeCols = table === 'l1_records' && this.hasColumn('l1_records', 'scope') ? ', scope, workspace_id' : '';
|
|
772
866
|
const metaCols = 'record_id, content, type, priority, scene_name, version, timestamp_str, created_time, updated_time, metadata_json, family' +
|
|
773
|
-
temporal
|
|
867
|
+
temporal +
|
|
868
|
+
scopeCols;
|
|
774
869
|
stmt =
|
|
775
870
|
action === 'delete'
|
|
776
871
|
? this.db.prepare(`DELETE FROM ${table} WHERE record_id IN (${ph})`)
|
|
@@ -830,7 +925,7 @@ export class MemoryDb {
|
|
|
830
925
|
if (this.degraded)
|
|
831
926
|
return [];
|
|
832
927
|
const rows = this.db
|
|
833
|
-
.prepare('SELECT record_id, content, type, priority, scene_name, version, timestamp_str, created_time, updated_time, metadata_json, family, valid_from, valid_to, persistence FROM l1_records')
|
|
928
|
+
.prepare('SELECT record_id, content, type, priority, scene_name, version, timestamp_str, created_time, updated_time, metadata_json, family, valid_from, valid_to, persistence, scope, workspace_id FROM l1_records')
|
|
834
929
|
.all();
|
|
835
930
|
return rows.map(rowToRecord);
|
|
836
931
|
}
|
|
@@ -843,7 +938,194 @@ export class MemoryDb {
|
|
|
843
938
|
}
|
|
844
939
|
return rows.map(rowToRecord);
|
|
845
940
|
}
|
|
846
|
-
/**
|
|
941
|
+
/**
|
|
942
|
+
* §B 决策凭证批量落盘。`INSERT OR IGNORE` + 确定性 `receipt_id`
|
|
943
|
+
* (见 `receipts.ts` 的 `receiptIdFor`)→ 同一次 run 重放不产生重复行。
|
|
944
|
+
* 返回实际新增条数(被忽略的重复不计)。
|
|
945
|
+
*
|
|
946
|
+
* 刻意**不开事务**:凭证是旁路观测数据,单条独立、重放幂等,部分写入无害;
|
|
947
|
+
* 为它引入事务只会把失败面扩大。调用方另有 `persistReceiptsSafely` 兜底不抛。
|
|
948
|
+
*
|
|
949
|
+
* 写入后**顺带执行保留策略**(task_18)。把裁剪挂在这里而不是交给调用方,
|
|
950
|
+
* 是为了让"有界"成为**结构性保证**:任何写路径都不可能忘记裁剪,
|
|
951
|
+
* 因而表容量不可能随使用时间无界增长。裁剪自身失败只 warn——
|
|
952
|
+
* 它是省空间的动作,失败了最坏是这次没省下来,绝不能因此弄丢刚落盘的凭证
|
|
953
|
+
* (故裁剪在写入**之后**,且包在 try 里)。
|
|
954
|
+
*/
|
|
955
|
+
recordReceipts(rows, opts) {
|
|
956
|
+
if (this.degraded || rows.length === 0)
|
|
957
|
+
return 0;
|
|
958
|
+
const stmt = this.db.prepare(`INSERT OR IGNORE INTO l1_receipts (receipt_id, run_id, record_id, kind, input_digest, decided_at)
|
|
959
|
+
VALUES (?, ?, ?, ?, ?, ?)`);
|
|
960
|
+
let n = 0;
|
|
961
|
+
for (const r of rows) {
|
|
962
|
+
n += Number(stmt.run(r.receiptId, r.runId, r.recordId, r.kind, r.inputDigest, r.decidedAt).changes);
|
|
963
|
+
}
|
|
964
|
+
try {
|
|
965
|
+
this.trimReceipts(opts?.maxRuns ?? RECEIPTS_MAX_RUNS);
|
|
966
|
+
}
|
|
967
|
+
catch (err) {
|
|
968
|
+
this.logger?.warn(`${TAG} L1 决策凭证裁剪失败,本次不回收空间(**凭证已正常落盘,记忆与回溯不受影响**): ${err instanceof Error ? err.message : String(err)}`);
|
|
969
|
+
}
|
|
970
|
+
return n;
|
|
971
|
+
}
|
|
972
|
+
/**
|
|
973
|
+
* §C 矛盾冻结(task_22):落盘待裁决冲突对。
|
|
974
|
+
*
|
|
975
|
+
* `INSERT OR IGNORE`——幂等来自 **pair_id 主键**而非调用方自觉:
|
|
976
|
+
* `conflictPairId(runId, winner, loser)` 对同一三元组恒等,故一轮蒸馏重复落盘
|
|
977
|
+
* 只会得到一行。与 §B 凭证同一手法(那边是 `receipt_id` 主键)。
|
|
978
|
+
*
|
|
979
|
+
* 与凭证不同,这里**不做保留裁剪**:待裁决对是**欠人的债**,不是观测数据。
|
|
980
|
+
* 裁剪它等于把用户还没看的裁决请求悄悄删掉,那是丢工作而不是省空间。
|
|
981
|
+
* 有界性交给 task_24 的队列上限(超限不再停放、回落自动裁决),语义是
|
|
982
|
+
* 「**不收新的**」而非「**偷偷删旧的**」。
|
|
983
|
+
*
|
|
984
|
+
* @returns 实际新插入的行数。
|
|
985
|
+
*/
|
|
986
|
+
recordConflictPending(rows) {
|
|
987
|
+
if (this.degraded || rows.length === 0)
|
|
988
|
+
return 0;
|
|
989
|
+
const stmt = this.db.prepare(`INSERT OR IGNORE INTO conflict_pending
|
|
990
|
+
(pair_id, run_id, winner_id, loser_id, created_at, resolved_at, resolution)
|
|
991
|
+
VALUES (?, ?, ?, ?, ?, ?, ?)`);
|
|
992
|
+
let n = 0;
|
|
993
|
+
for (const r of rows) {
|
|
994
|
+
n += Number(stmt.run(r.pairId, r.runId, r.winnerId, r.loserId, r.createdAt, r.resolvedAt, r.resolution).changes);
|
|
995
|
+
}
|
|
996
|
+
return n;
|
|
997
|
+
}
|
|
998
|
+
/** §C 冻结:把图谱 `disputed` 状态同步到给定冲突集(薄缝,便于单测替换)。 */
|
|
999
|
+
syncGraphDisputed(disputedRecordIds) {
|
|
1000
|
+
if (this.degraded)
|
|
1001
|
+
return { marked: 0, cleared: 0 };
|
|
1002
|
+
return this.graphStore.syncDisputed(disputedRecordIds);
|
|
1003
|
+
}
|
|
1004
|
+
/**
|
|
1005
|
+
* §C 冻结队列的**未裁决**条数(task_24 队列上限判据)。
|
|
1006
|
+
* 走 `idx_conflict_pending_unresolved` 偏索引,不是全表扫描。
|
|
1007
|
+
*/
|
|
1008
|
+
countConflictPendingUnresolved() {
|
|
1009
|
+
if (this.degraded)
|
|
1010
|
+
return 0;
|
|
1011
|
+
const row = this.db.prepare(`SELECT COUNT(*) AS n FROM conflict_pending WHERE resolved_at = ''`).get();
|
|
1012
|
+
return Number(row?.n ?? 0);
|
|
1013
|
+
}
|
|
1014
|
+
/**
|
|
1015
|
+
* §C 取未裁决冲突对(task_24 超时扫描 / task_25 裁决工具)。
|
|
1016
|
+
*
|
|
1017
|
+
* `createdBefore` 为**排他上界**(ISO 串):只取该时刻之前创建的,用于超时判定。
|
|
1018
|
+
* 定序 `created_at ASC, pair_id ASC`——先来先服务,且同一毫秒内仍**确定可复现**。
|
|
1019
|
+
*/
|
|
1020
|
+
listConflictPending(opts = {}) {
|
|
1021
|
+
if (this.degraded)
|
|
1022
|
+
return [];
|
|
1023
|
+
const limit = Number.isFinite(opts.limit) && (opts.limit ?? 0) > 0 ? Math.floor(opts.limit) : 500;
|
|
1024
|
+
const params = [];
|
|
1025
|
+
let where = `resolved_at = ''`;
|
|
1026
|
+
if (opts.createdBefore) {
|
|
1027
|
+
where += ` AND created_at < ?`;
|
|
1028
|
+
params.push(opts.createdBefore);
|
|
1029
|
+
}
|
|
1030
|
+
const rows = this.db
|
|
1031
|
+
.prepare(`SELECT pair_id, run_id, winner_id, loser_id, created_at, resolved_at, resolution
|
|
1032
|
+
FROM conflict_pending WHERE ${where}
|
|
1033
|
+
ORDER BY created_at ASC, pair_id ASC LIMIT ?`)
|
|
1034
|
+
.all(...params, limit);
|
|
1035
|
+
return rows.map(toConflictPair);
|
|
1036
|
+
}
|
|
1037
|
+
/**
|
|
1038
|
+
* §C 打上裁决结论。
|
|
1039
|
+
*
|
|
1040
|
+
* `WHERE resolved_at = ''` 使**已裁决的不会被覆盖**:裁决是一次性的判定行为,
|
|
1041
|
+
* 重复调用不该把第一次的结论改写掉(人工裁决与自动了结的次序因此不可逆)。
|
|
1042
|
+
*
|
|
1043
|
+
* @returns 受影响行数(0 = 该对被裁决过或不存在)。
|
|
1044
|
+
*/
|
|
1045
|
+
resolveConflictPending(pairId, resolution, resolvedAt) {
|
|
1046
|
+
if (this.degraded)
|
|
1047
|
+
return 0;
|
|
1048
|
+
const stmt = this.db.prepare(`UPDATE conflict_pending SET resolved_at = ?, resolution = ?
|
|
1049
|
+
WHERE pair_id = ? AND resolved_at = ''`);
|
|
1050
|
+
return Number(stmt.run(resolvedAt, resolution, pairId).changes);
|
|
1051
|
+
}
|
|
1052
|
+
/**
|
|
1053
|
+
* §B 凭证保留策略(task_18):只保留**最新**的 `maxRuns` 个 run,更老的整批删除。
|
|
1054
|
+
* 返回被删除的行数。
|
|
1055
|
+
*
|
|
1056
|
+
* 两条刻意的约束:
|
|
1057
|
+
* - **粒度是 run,不是行**。按行裁剪会切出"半截批次",而 task_19 的按 run 回溯
|
|
1058
|
+
* 正是要回答"这一轮蒸馏都判了什么"——一个少了尾巴的批次会给出**看似完整、
|
|
1059
|
+
* 实则遗漏**的结论,比查不到更糟。整批留、整批删,回溯的原子性才有保证。
|
|
1060
|
+
* - **只碰 `l1_receipts`,绝不碰 `l1_records`**。前者是可再生/可丢弃的观测数据,
|
|
1061
|
+
* 后者是用户的事实源。为省几 MB 而波及记忆本体,是把容量优化做成了数据丢失。
|
|
1062
|
+
*
|
|
1063
|
+
* `maxRuns <= 0` 或非有限值一律**不裁剪**——"传 0 即清空"是个太容易被误触的
|
|
1064
|
+
* 语义,宁可把它定义为无效输入。
|
|
1065
|
+
*
|
|
1066
|
+
* 定序取每 run 的 `MAX(decided_at)`(凭证的 decided_at 在一批内恒定)并以
|
|
1067
|
+
* `run_id` 兜底,使同一时刻产生的多个 run 也有**确定**的相对序,裁剪结果可复现。
|
|
1068
|
+
*/
|
|
1069
|
+
trimReceipts(maxRuns) {
|
|
1070
|
+
if (this.degraded)
|
|
1071
|
+
return 0;
|
|
1072
|
+
if (!Number.isFinite(maxRuns) || maxRuns <= 0)
|
|
1073
|
+
return 0;
|
|
1074
|
+
const stmt = this.db.prepare(`DELETE FROM l1_receipts WHERE run_id NOT IN (
|
|
1075
|
+
SELECT run_id FROM l1_receipts
|
|
1076
|
+
GROUP BY run_id
|
|
1077
|
+
ORDER BY MAX(decided_at) DESC, run_id DESC
|
|
1078
|
+
LIMIT ?
|
|
1079
|
+
)`);
|
|
1080
|
+
return Number(stmt.run(Math.floor(maxRuns)).changes);
|
|
1081
|
+
}
|
|
1082
|
+
/**
|
|
1083
|
+
* §B 双维回溯(task_19):按 `record_id` / `run_id` 查判定史,两维同给为 **AND**。
|
|
1084
|
+
*
|
|
1085
|
+
* 两条刻意的行为:
|
|
1086
|
+
* - **两维都不给返回空,而不是全表**。「查全部凭证」不是本能力的目标;把缺参
|
|
1087
|
+
* 兜成全表,会让一次误调用变成全库判定史导出。调用方本就该先拒绝这种用法
|
|
1088
|
+
* (工具层给提示、端点层直接报错),这里是第二道,方向一致。
|
|
1089
|
+
* - **定序确定**:`decided_at DESC, run_id DESC`。回溯的价值在于可复现——
|
|
1090
|
+
* 同一问题两次问出不同顺序,核对时就会怀疑是不是数据变了。`run_id` 兜底
|
|
1091
|
+
* 同一毫秒内的多批(L1 蒸馏是 LLM 调用,同刻两批罕见但非不可能)。
|
|
1092
|
+
* 新的在前,与 `listL1` 的倒序口径一致。
|
|
1093
|
+
*/
|
|
1094
|
+
listReceipts(opts) {
|
|
1095
|
+
if (this.degraded)
|
|
1096
|
+
return [];
|
|
1097
|
+
const where = receiptWhere(opts);
|
|
1098
|
+
if (where === null)
|
|
1099
|
+
return [];
|
|
1100
|
+
const limit = Math.min(Math.max(Math.floor(opts.limit) || 1, 1), RECEIPTS_QUERY_LIMIT_MAX);
|
|
1101
|
+
const rows = this.db
|
|
1102
|
+
.prepare(`SELECT receipt_id, run_id, record_id, kind, input_digest, decided_at
|
|
1103
|
+
FROM l1_receipts WHERE ${where.sql}
|
|
1104
|
+
ORDER BY decided_at DESC, run_id DESC
|
|
1105
|
+
LIMIT ?`)
|
|
1106
|
+
.all(...where.params, limit);
|
|
1107
|
+
return rows.map((r) => ({
|
|
1108
|
+
receiptId: r.receipt_id,
|
|
1109
|
+
runId: r.run_id,
|
|
1110
|
+
recordId: r.record_id,
|
|
1111
|
+
kind: r.kind,
|
|
1112
|
+
inputDigest: r.input_digest,
|
|
1113
|
+
decidedAt: r.decided_at,
|
|
1114
|
+
}));
|
|
1115
|
+
}
|
|
1116
|
+
/** 同维度命中的**总条数**(不受 limit 影响,供"还有多少条没显示"提示)。 */
|
|
1117
|
+
countReceipts(opts) {
|
|
1118
|
+
if (this.degraded)
|
|
1119
|
+
return 0;
|
|
1120
|
+
const where = receiptWhere(opts);
|
|
1121
|
+
if (where === null)
|
|
1122
|
+
return 0;
|
|
1123
|
+
const row = this.db
|
|
1124
|
+
.prepare(`SELECT COUNT(*) AS n FROM l1_receipts WHERE ${where.sql}`)
|
|
1125
|
+
.get(...where.params);
|
|
1126
|
+
return Number(row.n);
|
|
1127
|
+
}
|
|
1128
|
+
/** 浏览列表(UI 用):按更新时间倒序,支持类型/场景/族/Hall/可见范围过滤与分页。失败返回空。 */
|
|
847
1129
|
listL1(opts) {
|
|
848
1130
|
if (this.degraded)
|
|
849
1131
|
return { items: [], total: 0 };
|
|
@@ -862,6 +1144,11 @@ export class MemoryDb {
|
|
|
862
1144
|
where.push('family = ?');
|
|
863
1145
|
params.push(opts.family);
|
|
864
1146
|
}
|
|
1147
|
+
// §E 可见范围(缺省不过滤):global 记录 + 本工作区记录。与检索路径同一条判据。
|
|
1148
|
+
if (opts.workspaceId) {
|
|
1149
|
+
where.push("(scope = 'global' OR workspace_id = ?)");
|
|
1150
|
+
params.push(opts.workspaceId);
|
|
1151
|
+
}
|
|
865
1152
|
if (opts.hall) {
|
|
866
1153
|
// Hall 存于 metadata_json,用 json_extract 过滤(表小,逐行代价可接受)
|
|
867
1154
|
where.push(`json_extract(metadata_json, '$.hall') = ?`);
|
|
@@ -870,7 +1157,7 @@ export class MemoryDb {
|
|
|
870
1157
|
const whereSql = where.length > 0 ? ` WHERE ${where.join(' AND ')}` : '';
|
|
871
1158
|
const totalRow = this.db.prepare(`SELECT COUNT(*) AS n FROM l1_records${whereSql}`).get(...params);
|
|
872
1159
|
const rows = this.db
|
|
873
|
-
.prepare(`SELECT record_id, content, type, priority, scene_name, version, timestamp_str, created_time, updated_time, metadata_json, family, valid_from, valid_to, persistence FROM l1_records${whereSql} ORDER BY updated_time DESC LIMIT ? OFFSET ?`)
|
|
1160
|
+
.prepare(`SELECT record_id, content, type, priority, scene_name, version, timestamp_str, created_time, updated_time, metadata_json, family, valid_from, valid_to, persistence, scope, workspace_id FROM l1_records${whereSql} ORDER BY updated_time DESC LIMIT ? OFFSET ?`)
|
|
874
1161
|
.all(...params, opts.limit, opts.offset);
|
|
875
1162
|
return { items: rows.map(rowToRecord), total: totalRow?.n ?? 0 };
|
|
876
1163
|
}
|
|
@@ -896,17 +1183,19 @@ export class MemoryDb {
|
|
|
896
1183
|
// ============================
|
|
897
1184
|
// L1 检索
|
|
898
1185
|
// ============================
|
|
899
|
-
/** FTS5 BM25 检索(family 缺省不过滤)。失败返回空数组(调用方降级)。 */
|
|
900
|
-
searchL1Fts(query, limit, family) {
|
|
1186
|
+
/** FTS5 BM25 检索(family / workspaceId 缺省不过滤)。失败返回空数组(调用方降级)。 */
|
|
1187
|
+
searchL1Fts(query, limit, family, workspaceId) {
|
|
901
1188
|
if (this.degraded || !this.ftsAvailable || limit <= 0)
|
|
902
1189
|
return [];
|
|
903
1190
|
const ftsQuery = buildFtsQuery(query);
|
|
904
1191
|
if (!ftsQuery)
|
|
905
1192
|
return [];
|
|
1193
|
+
// 哨兵:'' = 不做可见范围过滤(与 SQL 里的 `? = ''` 分支对应)
|
|
1194
|
+
const ws = workspaceId ?? '';
|
|
906
1195
|
try {
|
|
907
1196
|
const rows = (family
|
|
908
|
-
? this.stmtL1FtsSearchFamily.all(ftsQuery, family, limit)
|
|
909
|
-
: this.stmtL1FtsSearch.all(ftsQuery, limit));
|
|
1197
|
+
? this.stmtL1FtsSearchFamily.all(ftsQuery, family, ws, ws, limit)
|
|
1198
|
+
: this.stmtL1FtsSearch.all(ftsQuery, ws, ws, limit));
|
|
910
1199
|
return rows.map((r) => ({
|
|
911
1200
|
id: r.record_id,
|
|
912
1201
|
content: r.content,
|
|
@@ -922,13 +1211,17 @@ export class MemoryDb {
|
|
|
922
1211
|
return [];
|
|
923
1212
|
}
|
|
924
1213
|
}
|
|
925
|
-
/**
|
|
926
|
-
|
|
1214
|
+
/**
|
|
1215
|
+
* vec0 余弦 KNN 检索(score = 1 - cosine distance)。失败返回空数组。
|
|
1216
|
+
* family / workspaceId 过滤走**过度召回 + 回查过滤**(vec0 无法 WHERE)。
|
|
1217
|
+
* 放大倍数对两条轴**相乘**:两轴各自丢弃行,单独放大任一条都不够。
|
|
1218
|
+
*/
|
|
1219
|
+
searchL1Vector(embedding, topK, family, workspaceId) {
|
|
927
1220
|
if (this.degraded || !this.stmtSearchL1Vec || topK <= 0)
|
|
928
1221
|
return [];
|
|
929
1222
|
try {
|
|
930
|
-
//
|
|
931
|
-
const retrieveCount = (topK + ZERO_VEC_BUFFER) * (family ? 3 : 1);
|
|
1223
|
+
// 过度召回补偿遗留零向量;带过滤时再放大(不命中过滤条件的行会被丢弃)
|
|
1224
|
+
const retrieveCount = (topK + ZERO_VEC_BUFFER) * (family ? 3 : 1) * (workspaceId ? 3 : 1);
|
|
932
1225
|
const rows = this.stmtSearchL1Vec.all(vecToBuffer(embedding), retrieveCount);
|
|
933
1226
|
const hits = [];
|
|
934
1227
|
for (const { record_id, distance } of rows) {
|
|
@@ -939,6 +1232,8 @@ export class MemoryDb {
|
|
|
939
1232
|
continue;
|
|
940
1233
|
if (family && normFamily(meta.family, meta.type) !== family)
|
|
941
1234
|
continue;
|
|
1235
|
+
if (workspaceId && !isScopeVisible(meta.scope, meta.workspace_id, workspaceId))
|
|
1236
|
+
continue;
|
|
942
1237
|
hits.push({
|
|
943
1238
|
id: record_id,
|
|
944
1239
|
content: meta.content,
|
|
@@ -1429,14 +1724,47 @@ function rowToRecord(row) {
|
|
|
1429
1724
|
version: row.version ?? 0,
|
|
1430
1725
|
metadata,
|
|
1431
1726
|
family: normFamily(row.family, row.type),
|
|
1727
|
+
// §E:读回时归一(缺列/缺值一律 global)——"读不到归属"等价于"跨工作区可见",
|
|
1728
|
+
// 与写入侧 fail-open 同向:宁可退化成全局可见,也不让记忆凭空消失。
|
|
1729
|
+
scope: normScope(row.scope),
|
|
1730
|
+
workspaceId: row.workspace_id ?? '',
|
|
1432
1731
|
// 时间轴回读:空串 → undefined(空串表示"未填",不是"时间 0")
|
|
1433
1732
|
validFrom: row.valid_from ? Date.parse(row.valid_from) || undefined : undefined,
|
|
1434
1733
|
validTo: row.valid_to ? Date.parse(row.valid_to) || undefined : undefined,
|
|
1435
1734
|
persistence: normPersistence(row.persistence),
|
|
1436
1735
|
};
|
|
1437
1736
|
}
|
|
1438
|
-
/**
|
|
1439
|
-
|
|
1737
|
+
/**
|
|
1738
|
+
* 双维回溯的 WHERE 构造。返回 `null` 表示**二维皆缺**——调用方据此返回空,
|
|
1739
|
+
* 而不是拼出无 WHERE 的全表扫描(那会把误用变成全库导出)。
|
|
1740
|
+
* 两维同给时按 `AND` 组合:问的是"这条记录在那一轮里被判成了什么"。
|
|
1741
|
+
*/
|
|
1742
|
+
function receiptWhere(q) {
|
|
1743
|
+
const sql = [];
|
|
1744
|
+
const params = [];
|
|
1745
|
+
if (q.recordId) {
|
|
1746
|
+
sql.push('record_id = ?');
|
|
1747
|
+
params.push(q.recordId);
|
|
1748
|
+
}
|
|
1749
|
+
if (q.runId) {
|
|
1750
|
+
sql.push('run_id = ?');
|
|
1751
|
+
params.push(q.runId);
|
|
1752
|
+
}
|
|
1753
|
+
return sql.length === 0 ? null : { sql: sql.join(' AND '), params };
|
|
1754
|
+
}
|
|
1755
|
+
/** `conflict_pending` 行 → {@link ConflictPair}(snake_case 只活在这一层)。 */
|
|
1756
|
+
function toConflictPair(r) {
|
|
1757
|
+
return {
|
|
1758
|
+
pairId: String(r.pair_id ?? ''),
|
|
1759
|
+
runId: String(r.run_id ?? ''),
|
|
1760
|
+
winnerId: String(r.winner_id ?? ''),
|
|
1761
|
+
loserId: String(r.loser_id ?? ''),
|
|
1762
|
+
createdAt: String(r.created_at ?? ''),
|
|
1763
|
+
resolvedAt: String(r.resolved_at ?? ''),
|
|
1764
|
+
resolution: String(r.resolution ?? ''),
|
|
1765
|
+
};
|
|
1766
|
+
}
|
|
1767
|
+
/** epoch 数组 → {逗号连接 ISO, 首尾 ISO}(磁盘格式契约:逗号连接、升序、ISO)。 */ function timestampsToDb(ts) {
|
|
1440
1768
|
if (!ts || ts.length === 0)
|
|
1441
1769
|
return { str: '', start: '', end: '' };
|
|
1442
1770
|
const sorted = [...ts].filter((t) => Number.isFinite(t)).sort((a, b) => a - b);
|
|
@@ -1458,14 +1786,6 @@ function toIso(epochMs) {
|
|
|
1458
1786
|
return '';
|
|
1459
1787
|
return new Date(epochMs).toISOString();
|
|
1460
1788
|
}
|
|
1461
|
-
/** 全零向量(cosine 未定义,不可入向量表)。reindex 侧用它区分"不可嵌入"与"写入失败"。 */
|
|
1462
|
-
export function isZeroVector(vec) {
|
|
1463
|
-
for (const v of vec) {
|
|
1464
|
-
if (v !== 0)
|
|
1465
|
-
return false;
|
|
1466
|
-
}
|
|
1467
|
-
return true;
|
|
1468
|
-
}
|
|
1469
1789
|
/** NOT IN 片段(空集 → 空串;配合 notInParams 使用)。 */
|
|
1470
1790
|
function notInClause(column, exclude) {
|
|
1471
1791
|
if (!exclude || exclude.size === 0)
|
|
@@ -1486,6 +1806,6 @@ function normFamily(raw, type) {
|
|
|
1486
1806
|
return raw;
|
|
1487
1807
|
return familyForType(type);
|
|
1488
1808
|
}
|
|
1489
|
-
|
|
1490
|
-
|
|
1491
|
-
}
|
|
1809
|
+
// 向量编码工具的实现已抽到 `vec-utils.ts`(避免 sqlite ↔ graph-store 形成模块环),
|
|
1810
|
+
// 此处 re-export 保持既有外部导入点不变;文件内部使用走上方 import。
|
|
1811
|
+
export { isZeroVector, vecToBuffer };
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* vec0 编码工具:抽取向量 ↔ Buffer 的**唯一**编码约定。
|
|
3
|
+
*
|
|
4
|
+
* 为什么单独一个文件:`l1_vec` / `l0_vec`(在 `sqlite.ts`)与 `graph_node_vec`
|
|
5
|
+
* (在 `graph-store.ts`)必须用**同一种**二进制编码,否则同一份 vec0 扩展会出现
|
|
6
|
+
* 两种字节布局。而 `sqlite.ts` 已经 import 了 `GraphStore`——若让 `graph-store.ts`
|
|
7
|
+
* 反过来 import `sqlite.ts`,就形成模块环:能跑,但生死取决于加载顺序,
|
|
8
|
+
* 属于"隐式依赖",不是可以留在代码里的东西。
|
|
9
|
+
*/
|
|
10
|
+
/** 全零向量(cosine 未定义,不可入向量表)。reindex 侧用它区分"不可嵌入"与"写入失败"。 */
|
|
11
|
+
export declare function isZeroVector(vec: Float32Array): boolean;
|
|
12
|
+
/**
|
|
13
|
+
* Float32Array → Buffer(零拷贝视图,共享底层 ArrayBuffer)。
|
|
14
|
+
* vec0 的 `float[N]` 列按 little-endian 连续 float32 读取,这正是该视图的布局。
|
|
15
|
+
*/
|
|
16
|
+
export declare function vecToBuffer(vec: Float32Array): Buffer;
|