@yandy0725/pi-memory 1.4.0 → 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/migrate.ts ADDED
@@ -0,0 +1,303 @@
1
+ import { cp, mkdir, readdir, readFile, stat, writeFile } from "node:fs/promises";
2
+ import { join } from "node:path";
3
+ import { isEntryType } from "./entry-file";
4
+ import { BACKUP_DIR, INDEX_FILE, LOCK_FILE, unlinkStrict, type MemoryStore } from "./memory-store";
5
+
6
+ /** 迁移完成标记。写在**最后一步** —— 中途失败时它不存在,下次 session_start 会重试(spec §15.4)。 */
7
+ export const MIGRATED_FILE = ".migrated";
8
+
9
+ /** spec §15.3 步骤 1:迁移的整轮锁超时是 30s(要在锁内逐条重写整个目录)。 */
10
+ export const MIGRATE_LOCK_TIMEOUT_MS = 30_000;
11
+
12
+ export interface MigrationResult {
13
+ /** 被拆开并删除的 legacy topic 文件数。 */
14
+ files: number;
15
+ /** 生成的 entry 数。 */
16
+ entries: number;
17
+ /** 回滚点目录(`.backups/migrate-<ts>/`),原 topic 文件另存于其下的 `originals/`。 */
18
+ backupDir: string;
19
+ }
20
+
21
+ interface MigrationMarker extends MigrationResult {
22
+ migratedAt: string;
23
+ }
24
+
25
+ export interface LegacyEntry {
26
+ title: string;
27
+ content: string;
28
+ type: string;
29
+ updated: string;
30
+ }
31
+
32
+ /** CRLF(以及老 Mac 的单独 CR)会让下面每一个正则都失配:`line === "---"` 不成立、
33
+ * `^## (.+)$` 里的 `.` 不匹配 `\r`。归一只在解析入口做一次。 */
34
+ function normalizeEol(raw: string): string {
35
+ return raw.includes("\r") ? raw.replace(/\r\n?/g, "\n") : raw;
36
+ }
37
+
38
+ const FIELD_RE = /^([A-Za-z_][A-Za-z0-9_]*):[ \t]*(.*)$/;
39
+
40
+ /**
41
+ * 拆分文件开头的 frontmatter:返回字段表与正文。
42
+ *
43
+ * v1 的 `parseEntries` 把每一行 `---` 都当作 frontmatter 开关:正文里只要出现一行 `---`
44
+ * (markdown 常见的分隔线),其后所有内容(包括下一个 `## ` 段)都会被当成 frontmatter 静默
45
+ * 吞掉。这里**有意收紧**:frontmatter 只在文件开头识别一次,找到第一个闭合的 `---` 行后,
46
+ * 其后整体视为正文,正文扫描期间不再切换状态。理由:迁移会 `unlink` 原文件(只留 `.backups/`
47
+ * 里的副本),宽松解析吞掉的数据用户不会再看见。
48
+ *
49
+ * 没有 frontmatter(不以 `---\n` 开头)或没有闭合行时 `fields` 为空、`body` 为原文 ——
50
+ * 无 frontmatter 的 legacy 文件同样合法(spec §15.2 的第二个触发条件是「≥2 个 `## ` 段」)。
51
+ */
52
+ function splitFrontmatter(text: string): { fields: Record<string, string>; body: string } {
53
+ if (!text.startsWith("---\n")) return { fields: {}, body: text };
54
+ const lines = text.split("\n");
55
+ let end = -1;
56
+ for (let i = 1; i < lines.length; i++) {
57
+ if (lines[i] === "---") {
58
+ end = i;
59
+ break;
60
+ }
61
+ }
62
+ if (end === -1) return { fields: {}, body: text };
63
+ const fields: Record<string, string> = {};
64
+ for (const line of lines.slice(1, end)) {
65
+ const m = line.match(FIELD_RE);
66
+ if (m) fields[m[1]] = m[2].trim();
67
+ }
68
+ return { fields, body: lines.slice(end + 1).join("\n") };
69
+ }
70
+
71
+ /**
72
+ * 按 `## ` 切段(算法迁自 v1 `topic-file.ts#parseEntries`,但按 Finding I1 收紧:调用方必须
73
+ * 先用 `splitFrontmatter` 剥掉开头的 frontmatter,正文里的 `---` 不再是分隔符)。
74
+ */
75
+ function parseLegacySections(body: string): Array<{ title: string; content: string }> {
76
+ const entries: Array<{ title: string; content: string }> = [];
77
+ let currentTitle = "";
78
+ let currentContent: string[] = [];
79
+ let inEntry = false;
80
+
81
+ for (const line of body.split("\n")) {
82
+ const h2 = line.match(/^## (.+)$/);
83
+ if (h2) {
84
+ if (inEntry) entries.push({ title: currentTitle, content: currentContent.join("\n").trim() });
85
+ currentTitle = h2[1];
86
+ currentContent = [];
87
+ inEntry = true;
88
+ continue;
89
+ }
90
+ if (inEntry) currentContent.push(line);
91
+ }
92
+ if (inEntry) entries.push({ title: currentTitle, content: currentContent.join("\n").trim() });
93
+ return entries;
94
+ }
95
+
96
+ /**
97
+ * 这个文件是不是 v1 的 topic 文件(spec §15.2)?
98
+ *
99
+ * 判据:`MEMORY.md` 之外的 `.md`,且(frontmatter 含旧字段 `updated` **或** 含 ≥2 个 `## ` 段)。
100
+ *
101
+ * **必须先排除含 `modified` 的文件**:v2 的 entry 文件正文里完全可能有多个 `## ` 小标题,
102
+ * 若不排除,一条正常的 v2 记忆会被当成 legacy topic 再拆一次 —— 正文被切碎、原文件被删。
103
+ */
104
+ export function isLegacyTopicFile(raw: string): boolean {
105
+ const { fields, body } = splitFrontmatter(normalizeEol(raw));
106
+ if (fields.modified !== undefined) return false;
107
+ if (fields.updated !== undefined) return true;
108
+ return parseLegacySections(body).length >= 2;
109
+ }
110
+
111
+ /** 拆出 legacy topic 文件里的全部 `## ` 段,并把该文件 frontmatter 的 `type` / `updated` 附到每一段上。 */
112
+ export function parseLegacyEntries(raw: string): LegacyEntry[] {
113
+ const { fields, body } = splitFrontmatter(normalizeEol(raw));
114
+ return parseLegacySections(body).map((section) => ({
115
+ title: section.title,
116
+ content: section.content,
117
+ type: fields.type ?? "",
118
+ updated: fields.updated ?? "",
119
+ }));
120
+ }
121
+
122
+ /**
123
+ * 把 legacy 标题变成合法的 entry `name`。
124
+ *
125
+ * `](` 必须中和:`MemoryStore.#validateName` 会拒绝它(它能伪造索引行的 name/file 分组),
126
+ * 而迁移**不能**因为一条标题里有个 markdown 链接就整轮失败。替换成 `] (` 保住可读性。
127
+ */
128
+ function sanitizeName(title: string): string {
129
+ return title
130
+ .replace(/[\r\n]+/g, " ")
131
+ .replaceAll("](", "] (")
132
+ .trim();
133
+ }
134
+
135
+ /**
136
+ * 给一条 legacy 段落挑一个不与其它记忆冲突的 `name`(spec §15.3 的后缀规则)。
137
+ *
138
+ * 候选按 `base`、`base (2)`、`base (3)`…… 依次探测(`name` 精确匹配,与 store 的语义一致):
139
+ * - 未被占用 → 采用。
140
+ * - 已被占用,但磁盘上同名条目的正文与这一段相同 → **复用**这个名字:`addEntry` 对同名条目
141
+ * 是幂等覆盖,且保留磁盘上的 `created`。迁移中途失败后重跑时,上一轮已经写成功的条目走的正是
142
+ * 这条路 —— 不会产出 ` (2)` 影子副本(spec §15.4 的重跑安全)。
143
+ * - 已被占用、正文不同 → 试下一个后缀:迁移前就存在的同名用户条目必须被保护。
144
+ */
145
+ function pickName(used: Set<string>, bodies: Map<string, string>, base: string, body: string): string {
146
+ const wanted = body.trim();
147
+ let candidate = base;
148
+ for (let n = 2; used.has(candidate) && bodies.get(candidate)?.trim() !== wanted; n++) {
149
+ candidate = `${base} (${n})`;
150
+ }
151
+ return candidate;
152
+ }
153
+
154
+ /** 旧 `updated` 归一为 store 接受的 `YYYY-MM-DD`;解析不出来就用今天(`created` 是必填字段)。 */
155
+ function normalizeDate(updated: string, now: Date): string {
156
+ const iso = updated.match(/(\d{4})-(\d{2})-(\d{2})/);
157
+ if (iso) return `${iso[1]}-${iso[2]}-${iso[3]}`;
158
+ const parsed = updated ? new Date(updated) : null;
159
+ if (parsed && !Number.isNaN(parsed.getTime())) return parsed.toISOString().slice(0, 10);
160
+ return now.toISOString().slice(0, 10);
161
+ }
162
+
163
+ async function isFile(path: string): Promise<boolean> {
164
+ return stat(path).then(
165
+ (info) => info.isFile(),
166
+ () => false,
167
+ );
168
+ }
169
+
170
+ /** 只把 ENOENT 当作「文件不在了」;其余读取错误(EISDIR/EACCES/EIO)一律上抛(spec §15.4)。 */
171
+ async function readIfExists(path: string): Promise<string | null> {
172
+ try {
173
+ return await readFile(path, "utf8");
174
+ } catch (e) {
175
+ if ((e as NodeJS.ErrnoException).code === "ENOENT") return null;
176
+ throw e;
177
+ }
178
+ }
179
+
180
+ /**
181
+ * 自建回滚点 `.backups/migrate-<ts>/`(spec §15.3 步骤 2)。
182
+ *
183
+ * **不能用 `createSnapshot`**:它恒产出 `<ts>-<label>`,不可能以 `migrate-` 开头,
184
+ * 而 `pruneSnapshots` 只豁免 `migrate-` 前缀 —— 用它建的回滚点会在后续任何一次写入时
185
+ * 被 `snapshotKeep`(默认 5)裁掉,用户就再也回不去了(spec §19 风险表明列了这一条)。
186
+ *
187
+ * 布局:`migrate-<ts>/` = memoryDir 下所有普通文件的副本;`migrate-<ts>/originals/` = 待迁移的
188
+ * topic 文件副本(手工回滚时从这里恢复,见 spec §15.5)。
189
+ */
190
+ async function createRollbackPoint(memoryDir: string, legacyFiles: string[], now: Date): Promise<string> {
191
+ const stamp = now.toISOString().replace(/[:.]/g, "-");
192
+ const backupDir = join(memoryDir, BACKUP_DIR, `migrate-${stamp}`);
193
+ await mkdir(join(backupDir, "originals"), { recursive: true });
194
+
195
+ // memoryDir 的存在性已在调用前检查过;列表失败(EACCES/EIO)不能吞:那会返回一个只有
196
+ // `originals/` 的不完整回滚点,而它随后就被写进 `.migrated` —— 用户再也回不去(spec §15.4)。
197
+ const entries = await readdir(memoryDir, { withFileTypes: true });
198
+ for (const entry of entries) {
199
+ // 只复制普通文件、跳过锁记录;`.backups` 与 `sessions/` 是目录,天然被排除。
200
+ // `.migrated` 此刻还不存在(它在最后一步才写),所以不需要显式排除。
201
+ if (!entry.isFile() || entry.name === LOCK_FILE) continue;
202
+ await cp(join(memoryDir, entry.name), join(backupDir, entry.name));
203
+ }
204
+ for (const file of legacyFiles) {
205
+ await cp(join(memoryDir, file), join(backupDir, "originals", file));
206
+ }
207
+ return backupDir;
208
+ }
209
+
210
+ async function writeMarker(path: string, marker: MigrationMarker): Promise<void> {
211
+ await writeFile(path, `${JSON.stringify(marker, null, 2)}\n`, "utf8");
212
+ }
213
+
214
+ /**
215
+ * 首次 `session_start` 时把 v1 的 topic 布局迁移到 v2 的 per-entry 布局(spec §15 / D10)。
216
+ *
217
+ * 返回 `null` 表示「本次没有迁移任何东西」(已迁移过、目录不存在、或没有 legacy 文件)。
218
+ * 失败时**不写** `.migrated`、保留备份、把错误原样上抛(spec §15.4):下次 session_start 重试,
219
+ * 而重跑是安全的 —— `addEntry` 对同名条目幂等,且候选名只在「磁盘上同名条目的正文与当前段落不同」
220
+ * 时才追加后缀(正文相同即复用,见 `pickName`)。
221
+ */
222
+ export async function migrateIfNeeded(store: MemoryStore, options?: { now?: Date }): Promise<MigrationResult | null> {
223
+ const memoryDir = store.cfg.memoryDir;
224
+ const markerPath = join(memoryDir, MIGRATED_FILE);
225
+ if (await isFile(markerPath)) return null;
226
+
227
+ // 目录还不存在:没有 legacy 数据,也不在这里替 store 建目录(首次写入时它自己会建)。
228
+ const names = await readdir(memoryDir).catch(() => null);
229
+ if (names === null) return null;
230
+
231
+ const now = options?.now ?? new Date();
232
+ const nothing: MigrationMarker = { migratedAt: now.toISOString(), entries: 0, files: 0, backupDir: "" };
233
+
234
+ // 触发条件之一是「MEMORY.md 存在」(spec §15.2)。不存在就没有旧索引可迁,写标记以免每次
235
+ // session_start 都全目录 readFile。
236
+ if (!names.includes(INDEX_FILE)) {
237
+ await writeMarker(markerPath, nothing);
238
+ return null;
239
+ }
240
+
241
+ const candidates = names.filter((n) => n.endsWith(".md") && n !== INDEX_FILE && !n.startsWith(".")).sort();
242
+ const legacyFiles: string[] = [];
243
+ for (const file of candidates) {
244
+ // 只容忍 ENOENT(readdir 与 read 之间文件消失 → 当作非 legacy)。EISDIR/EACCES/EIO 必须上抛:
245
+ // 把不可读的候选归为「非 legacy」会在它是唯一候选时写下 0/0 标记,重试永久不再发生(spec §15.4)。
246
+ const raw = await readIfExists(join(memoryDir, file));
247
+ if (raw !== null && isLegacyTopicFile(raw)) legacyFiles.push(file);
248
+ }
249
+ if (legacyFiles.length === 0) {
250
+ await writeMarker(markerPath, nothing);
251
+ return null;
252
+ }
253
+
254
+ return store.withLogicalLock(async () => {
255
+ const backupDir = await createRollbackPoint(memoryDir, legacyFiles, now);
256
+
257
+ // `name` 唯一性集合的初值来自磁盘:这让「迁移中途失败后重跑」不会产出 ` (2)` 影子副本 ——
258
+ // 上一轮已经写成功的条目会在 listEntries 里,正文与当前段落相同的候选会被复用(addEntry 幂等)。
259
+ const summaries = await store.listEntries();
260
+ const usedNames = new Set(summaries.map((summary) => summary.name));
261
+ // 候选探测还需要「同名的正文」。一次 `searchEntries("")` 拿走全部正文(空 needle 对
262
+ // 任何字符串都是 includes 命中),避免旧实现里在摘要循环内反复 `readEntry`(每次都重扫
263
+ // 目录 → O(N²))。同名多条目时以 listEntries 的顺序为准(后者覆盖前者),与现状一致。
264
+ const bodies = new Map<string, string>();
265
+ for (const entry of await store.searchEntries("")) bodies.set(entry.name, entry.body);
266
+ let entries = 0;
267
+
268
+ for (const file of legacyFiles) {
269
+ const raw = await readFile(join(memoryDir, file), "utf8");
270
+ for (const legacy of parseLegacyEntries(raw)) {
271
+ const name = pickName(usedNames, bodies, sanitizeName(legacy.title), legacy.content);
272
+ // 空标题或空正文的段落不生成 entry:store 会拒绝("name is required" / "body is required"),
273
+ // 而让整轮迁移因为一个空 `## ` 段失败是更差的取舍。原文件在 originals/ 里,信息没有丢。
274
+ if (!name || !legacy.content) continue;
275
+ usedNames.add(name);
276
+ await store.addEntry(
277
+ {
278
+ name,
279
+ description: name,
280
+ type: isEntryType(legacy.type) ? legacy.type : "feedback",
281
+ body: legacy.content,
282
+ created: normalizeDate(legacy.updated, now),
283
+ },
284
+ { skipLogicalLock: true, skipSnapshot: true },
285
+ );
286
+ entries++;
287
+ }
288
+ }
289
+
290
+ await store.rebuildIndex({ skipLogicalLock: true, skipSnapshot: true });
291
+
292
+ // 原 topic 文件从 memory 目录移除,否则 rebuildIndex 之后它们还会作为「无法解析的 .md」
293
+ // 留在目录里(spec §15.3 步骤 5)。它们已经保留在 migrate-<ts>/originals/。
294
+ // 删除失败必须上抛(只有 ENOENT 视为已删除):吞掉的话 `.migrated` 会在文件仍留在目录里的
295
+ // 情况下被写下,重试永久不再发生(spec §15.4)。
296
+ for (const file of legacyFiles) {
297
+ await unlinkStrict(join(memoryDir, file));
298
+ }
299
+
300
+ await writeMarker(markerPath, { migratedAt: now.toISOString(), entries, files: legacyFiles.length, backupDir });
301
+ return { files: legacyFiles.length, entries, backupDir };
302
+ }, MIGRATE_LOCK_TIMEOUT_MS);
303
+ }
package/src/paths.ts CHANGED
@@ -1,6 +1,6 @@
1
1
  import { execFile } from "node:child_process";
2
2
  import { createHash } from "node:crypto";
3
- import { join, normalize, resolve, sep } from "node:path";
3
+ import { join, resolve } from "node:path";
4
4
  import { promisify } from "node:util";
5
5
 
6
6
  const execFileP = promisify(execFile);
@@ -175,16 +175,3 @@ export async function resolveMemoryDir(config: { memoryDir: string }, cwd: strin
175
175
  const { kind, key } = await projectIdentity(cwd);
176
176
  return join(config.memoryDir, kind, projectDirName(key));
177
177
  }
178
-
179
- export function safeTopicPath(memoryDir: string, topic: string): string {
180
- const normalized = normalize(topic);
181
- if (normalized.includes("..") || normalized.startsWith(sep)) {
182
- throw new Error(`Unsafe topic path: ${topic}`);
183
- }
184
- const resolved = resolve(memoryDir, normalized);
185
- const resolvedMemoryDir = resolve(memoryDir);
186
- if (!resolved.startsWith(resolvedMemoryDir + sep) && resolved !== resolvedMemoryDir) {
187
- throw new Error(`Topic escapes memory dir: ${topic}`);
188
- }
189
- return resolved;
190
- }
@@ -0,0 +1,122 @@
1
+ /**
2
+ * 进程内互斥(**不跨进程**)。
3
+ *
4
+ * 为什么存在:`memory add`、extract、dream 都在**同一个 pi 进程**里(dream 与 extract 是
5
+ * 进程内的 SDK 会话,不是子进程),而它们的互斥作用域长度相差三个数量级 —— 单次写入是毫秒,
6
+ * dream 整轮是分钟。让一把**跨进程**的 `.lock` 同时承担这两段时长会引出两串麻烦:
7
+ * 1. dream 整轮持 `.lock` 时,它自己的每次原语调用都会撞上自己的锁(`.lock` 不可重入);
8
+ * 2. 另一个 pi 进程(同仓库的多个 worktree 共享 memory 目录)的写入要被堵住整轮;
9
+ * 3. 「持锁数分钟」逼出了 TTL 续约心跳与 stale 接管,而后者在 POSIX 上没有
10
+ * compare-and-replace,无法做到可证明安全(见 fs-lock.ts 的说明)。
11
+ *
12
+ * 所以分层:**本模块承担「逻辑作用域」(毫秒的单次调用 或 分钟的整轮),
13
+ * `.lock` 只承担毫秒级的物理写入。** 进程内互斥是一个 Promise 队列 —— 天然可嵌套地表达
14
+ * 「谁在等谁」,零文件系统、零 staleness、零平台语义。
15
+ */
16
+
17
+ /** 等待超过 `timeoutMs` 仍未拿到锁。调用方应把它当作可向用户交代的明确失败,而不是可重试的抖动。 */
18
+ export class ProcessLockTimeoutError extends Error {
19
+ constructor(
20
+ readonly key: string,
21
+ readonly timeoutMs: number,
22
+ ) {
23
+ super(`Memory operations for ${key} are already running in this process (waited ${timeoutMs}ms)`);
24
+ this.name = "ProcessLockTimeoutError";
25
+ }
26
+ }
27
+
28
+ /** 每个 key 的状态。**永不删除** —— 删了会出现「同一 key 两套状态」的脑裂互斥(一方用旧状态、一方用新状态)。条目数 = 进程见过的 memoryDir 数,可忽略。 */
29
+ interface LockState {
30
+ /** 队尾:最后一个等待者放行后 resolve 的 promise。 */
31
+ tail: Promise<void>;
32
+ /** 当前是否有人持有。仅用于诊断/断言 —— 互斥本身依赖 `tail`,不依赖它。 */
33
+ held: boolean;
34
+ /** 正在排队(尚未拿到)的调用数。 */
35
+ waiters: number;
36
+ }
37
+
38
+ const states = new Map<string, LockState>();
39
+
40
+ function stateFor(key: string): LockState {
41
+ let state = states.get(key);
42
+ if (!state) {
43
+ state = { tail: Promise.resolve(), held: false, waiters: 0 };
44
+ states.set(key, state);
45
+ }
46
+ return state;
47
+ }
48
+
49
+ /**
50
+ * 排队取得某个 key 的进程内锁,返回释放函数。
51
+ *
52
+ * 超时时**必须把自己的 gate 放掉**:我们从未持有过锁,若不放行,排在后面的调用会被我们的 gate
53
+ * 永久挂住(队尾链是 `previous.then(() => gate)`,我们的 gate 不放,后面所有人都拿不到)。
54
+ *
55
+ * 注意:**不能用「队尾是否为空」当作「有没有被持有」** —— 等待者超时会把自己从队尾摘掉,
56
+ * 而持有者仍在工作,于是队尾变空而锁定仍被持有。所以 `held` / `waiters` 单独计数。
57
+ */
58
+ async function acquire(key: string, timeoutMs: number): Promise<() => void> {
59
+ const state = stateFor(key);
60
+ const previous = state.tail;
61
+ let release!: () => void;
62
+ const gate = new Promise<void>((resolve) => {
63
+ release = resolve;
64
+ });
65
+ state.tail = previous.then(() => gate, () => gate);
66
+ state.waiters += 1;
67
+
68
+ let timer: ReturnType<typeof setTimeout> | undefined;
69
+ const timedOut = Symbol("timeout");
70
+ const outcome = await Promise.race([
71
+ previous.then(() => "acquired" as const),
72
+ new Promise<typeof timedOut>((resolve) => {
73
+ timer = setTimeout(() => resolve(timedOut), timeoutMs);
74
+ }),
75
+ ]);
76
+ if (timer) clearTimeout(timer);
77
+ state.waiters -= 1;
78
+
79
+ if (outcome === timedOut) {
80
+ release();
81
+ throw new ProcessLockTimeoutError(key, timeoutMs);
82
+ }
83
+
84
+ state.held = true;
85
+ return () => {
86
+ state.held = false;
87
+ release();
88
+ };
89
+ }
90
+
91
+ /** 在进程内串行执行 `fn`(按 `key` 分键)。超时抛 `ProcessLockTimeoutError`。 */
92
+ export async function withProcessLock<T>(key: string, timeoutMs: number, fn: () => Promise<T>): Promise<T> {
93
+ const release = await acquire(key, timeoutMs);
94
+ try {
95
+ return await fn();
96
+ } finally {
97
+ release();
98
+ }
99
+ }
100
+
101
+ /** 只在 `key` 空闲时执行 `fn`;有人在持有或排队则立刻返回 `null`,绝不等待(extract 的「锁忙则跳过本轮」)。 */
102
+ export async function tryWithProcessLock<T>(key: string, fn: () => Promise<T>): Promise<T | null> {
103
+ try {
104
+ return await withProcessLock(key, 0, fn);
105
+ } catch (e) {
106
+ if (e instanceof ProcessLockTimeoutError) return null;
107
+ throw e;
108
+ }
109
+ }
110
+
111
+ /**
112
+ * 该 key 上是否有人在持有或排队。
113
+ * 用途:整轮持有者(dream / 迁移)启动前断言自己确实已经包住了锁 —— 「忘了包」会静默失去整轮互斥,
114
+ * 这条断言把它变成一个立刻可见的错误。
115
+ *
116
+ * 注意它只是**诊断用**的:互斥由 `tail` 保证。释放与下一个持有者接上之间存在一个微任务窗口,
117
+ * 此窗口内 `isProcessLockActive` 可能瞬时返回 false,但那时新调用者仍然会正确地排在 `tail` 之后。
118
+ */
119
+ export function isProcessLockActive(key: string): boolean {
120
+ const state = states.get(key);
121
+ return state !== undefined && (state.held || state.waiters > 0);
122
+ }
@@ -0,0 +1,36 @@
1
+ /**
2
+ * 注入净化(spec §13)。**只在注入时**使用,绝不写回磁盘(D11)—— 磁盘上的 entry 必须
3
+ * 保持用户可读、可手工编辑的原始形态。
4
+ *
5
+ * 本模块是纯函数模块:不 import `node:fs`、不读配置、不碰 store。
6
+ */
7
+
8
+ /**
9
+ * 全部 Unicode `Cf`(format)类别字符:零宽(U+200B–U+200D、U+FEFF)、bidi 控制符
10
+ * (U+202A–U+202E、U+2066–U+2069)、软连字符、LRM/RLM、word joiner 等。
11
+ *
12
+ * 用 `\p{Cf}` 而不是逐个码点枚举:新加入的格式字符自动被覆盖。`\n` / `\t` 是 Cc、
13
+ * 空格是 Zs,都不在此列 —— 注入文本的排版必须原样保留。
14
+ */
15
+ const INVISIBLE_RE = /\p{Cf}/gu;
16
+
17
+ /** 剥离全部不可见格式字符(bidi 覆写可以让磁盘上的记忆在注入后"读出"别的意思)。 */
18
+ export function stripInvisibleChars(text: string): string {
19
+ return text.replace(INVISIBLE_RE, "");
20
+ }
21
+
22
+ /**
23
+ * 净化一段将要进入 system prompt / 注入消息的文本:剥离不可见字符 + 中和尖括号。
24
+ *
25
+ * **幂等是硬要求,所以这里刻意不转义 `&`。** 若把 `&` → `&amp;`,第一次产出的 `&lt;`
26
+ * 在第二次就会变成 `&amp;lt;`;而 `memory_index` 的值在 resume / fork / reload 时会被
27
+ * 逐轮重放(D13 / D14),每重放一次就漂移一次 —— system prompt 头部再也稳定不下来,
28
+ * prefix cache 全废。`sanitizeForInjection(sanitizeForInjection(x)) === sanitizeForInjection(x)`
29
+ * 由测试钉住。
30
+ *
31
+ * 代价:正文里字面的 `&lt;` 在注入后无法与「原本是 `<`」区分。这是可接受的取舍 ——
32
+ * 两者在模型眼里都读作 `<`,而我们要防的正是「用 `<` 伪造系统标签」。
33
+ */
34
+ export function sanitizeForInjection(text: string): string {
35
+ return stripInvisibleChars(text).replaceAll("<", "&lt;").replaceAll(">", "&gt;");
36
+ }
@@ -0,0 +1,68 @@
1
+ import { cp, mkdir, readdir, rm, stat } from "node:fs/promises";
2
+ import { basename, join } from "node:path";
3
+
4
+ export interface SnapshotOptions {
5
+ keep: number;
6
+ now?: () => Date;
7
+ }
8
+
9
+ /** 可字典序排序的快照时间戳(ISO 8601,`:` 与 `.` 换为 `-`)。 */
10
+ export function snapshotStamp(now: () => Date): string {
11
+ return now().toISOString().replace(/[:.]/g, "-");
12
+ }
13
+
14
+ async function uniqueDir(backupRoot: string, base: string): Promise<string> {
15
+ await mkdir(backupRoot, { recursive: true });
16
+ for (let n = 1; ; n++) {
17
+ const candidate = n === 1 ? join(backupRoot, base) : join(backupRoot, `${base}-${n}`);
18
+ try {
19
+ await mkdir(candidate);
20
+ return candidate;
21
+ } catch (e) {
22
+ if ((e as NodeJS.ErrnoException).code !== "EEXIST") throw e;
23
+ }
24
+ }
25
+ }
26
+
27
+ /** 将给定的 memory 目录相对路径复制到新快照目录;不存在的文件被跳过。 */
28
+ export async function createSnapshot(
29
+ backupRoot: string,
30
+ label: string,
31
+ files: string[],
32
+ memoryDir: string,
33
+ options: SnapshotOptions,
34
+ ): Promise<string> {
35
+ const dir = await uniqueDir(backupRoot, `${snapshotStamp(options.now ?? (() => new Date()))}-${label}`);
36
+ for (const file of files) {
37
+ try {
38
+ await cp(join(memoryDir, file), join(dir, basename(file)));
39
+ } catch (e) {
40
+ if ((e as NodeJS.ErrnoException).code === "ENOENT") continue;
41
+ throw e;
42
+ }
43
+ }
44
+ await pruneSnapshots(backupRoot, options.keep);
45
+ return dir;
46
+ }
47
+
48
+ /** 保留名字序最新的 keep 个快照;`migrate-` 开头的目录永不参与裁剪。 */
49
+ export async function pruneSnapshots(backupRoot: string, keep: number): Promise<void> {
50
+ let names: string[];
51
+ try {
52
+ names = await readdir(backupRoot);
53
+ } catch {
54
+ return;
55
+ }
56
+
57
+ const dirs: string[] = [];
58
+ for (const name of names) {
59
+ if (name.startsWith("migrate-")) continue;
60
+ const info = await stat(join(backupRoot, name)).catch(() => null);
61
+ if (info?.isDirectory()) dirs.push(name);
62
+ }
63
+
64
+ dirs.sort();
65
+ for (const name of dirs.slice(0, Math.max(0, dirs.length - Math.max(0, keep)))) {
66
+ await rm(join(backupRoot, name), { recursive: true, force: true });
67
+ }
68
+ }
package/src/index-file.ts DELETED
@@ -1,91 +0,0 @@
1
- export interface IndexEntry {
2
- name: string; // 原名 title,取自 frontmatter name
3
- topic: string; // 文件名
4
- hook: string; // 一行描述
5
- raw: string; // 原始行文本
6
- }
7
-
8
- export interface IndexFile {
9
- entries: IndexEntry[];
10
- raw: string;
11
- }
12
-
13
- // Matches: - [Name](topic.md) — hook
14
- const LINE_RE = /^-\s+\[([^\]]+)\]\(([^)]+)\)\s*—\s*(.*)$/;
15
-
16
- export function parseIndex(content: string): IndexFile {
17
- const entries: IndexEntry[] = [];
18
- for (const line of content.split("\n")) {
19
- const m = line.match(LINE_RE);
20
- if (m) {
21
- entries.push({
22
- name: m[1].trim(),
23
- topic: m[2].trim(),
24
- hook: m[3].trim(),
25
- raw: line,
26
- });
27
- }
28
- }
29
- return { entries, raw: content };
30
- }
31
-
32
- export function serializeIndex(entries: IndexEntry[]): string {
33
- return entries.map((e) => `- [${e.name}](${e.topic}) — ${e.hook}`).join("\n");
34
- }
35
-
36
- export function upsertEntryByTopic(entries: IndexEntry[], entry: IndexEntry): IndexEntry[] {
37
- const idx = entries.findIndex((e) => e.topic === entry.topic);
38
- if (idx === -1) return [...entries, entry];
39
- const next = [...entries];
40
- next[idx] = entry;
41
- return next;
42
- }
43
-
44
- export function removeEntryByTopic(entries: IndexEntry[], topic: string): IndexEntry[] {
45
- const idx = entries.findIndex((e) => e.topic === topic);
46
- if (idx === -1) throw new Error(`Topic "${topic}" not found in index`);
47
- const next = [...entries];
48
- next.splice(idx, 1);
49
- return next;
50
- }
51
-
52
- export function findEntryByTopic(entries: IndexEntry[], topic: string): IndexEntry | null {
53
- return entries.find((e) => e.topic === topic) ?? null;
54
- }
55
-
56
- export function updateHook(entries: IndexEntry[], topic: string, hook: string): IndexEntry[] {
57
- const idx = entries.findIndex((e) => e.topic === topic);
58
- if (idx === -1) throw new Error(`Topic "${topic}" not found in index`);
59
- const next = [...entries];
60
- next[idx] = { ...next[idx], hook, raw: "" };
61
- return next;
62
- }
63
-
64
- export function truncateForInjection(
65
- content: string,
66
- maxLines: number,
67
- maxBytes: number,
68
- ): { ok: boolean; content: string; truncated: boolean } {
69
- const lines = content.split("\n");
70
- let out = content;
71
- let truncated = false;
72
- if (lines.length > maxLines) {
73
- out = lines.slice(0, maxLines).join("\n");
74
- truncated = true;
75
- }
76
- if (Buffer.byteLength(out, "utf8") > maxBytes) {
77
- let cut = out;
78
- while (Buffer.byteLength(cut, "utf8") > maxBytes && cut.length > 0) cut = cut.slice(0, -1);
79
- out = cut;
80
- truncated = true;
81
- }
82
- if (truncated) out += `\n[truncated: memory index exceeds injection limit]`;
83
- return { ok: !truncated, content: out, truncated };
84
- }
85
-
86
- export function checkCapacity(entries: IndexEntry[], maxLines: number, maxBytes: number): boolean {
87
- const serialized = serializeIndex(entries);
88
- if (entries.length > maxLines) return false;
89
- if (Buffer.byteLength(serialized, "utf8") > maxBytes) return false;
90
- return true;
91
- }