@yandy0725/pi-memory 1.4.0 → 2.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +248 -140
- package/README.zh.md +266 -156
- package/index.ts +393 -95
- package/package.json +1 -1
- package/src/agent-runner.ts +17 -2
- package/src/config.ts +31 -6
- package/src/dream.ts +98 -56
- package/src/entry-file.ts +81 -0
- package/src/entry-index.ts +112 -0
- package/src/extract.ts +338 -57
- package/src/filename.ts +48 -0
- package/src/fs-lock.ts +251 -0
- package/src/index-source.ts +140 -0
- package/src/inject.ts +121 -73
- package/src/memory-store.ts +498 -0
- package/src/memory-tool.ts +244 -326
- package/src/migrate.ts +303 -0
- package/src/paths.ts +1 -14
- package/src/process-lock.ts +122 -0
- package/src/sanitize.ts +36 -0
- package/src/snapshot.ts +68 -0
- package/src/index-file.ts +0 -91
- package/src/topic-file.ts +0 -119
package/src/config.ts
CHANGED
|
@@ -24,7 +24,12 @@ export interface AutoSurfacingConfig {
|
|
|
24
24
|
model?: string;
|
|
25
25
|
thinkLevel: ThinkLevel;
|
|
26
26
|
maxFiles: number;
|
|
27
|
-
|
|
27
|
+
/**
|
|
28
|
+
* 单条 entry 注入正文的字节上限。
|
|
29
|
+
* v1 叫 `maxTopicBytes`(一个 topic 文件含多个 `##` 条目);v2 一个 entry 一个文件,故改名。
|
|
30
|
+
* **旧键不再生效**(`deepMerge` 会把它挂到对象上,但没有任何代码读它)。
|
|
31
|
+
*/
|
|
32
|
+
maxEntryBytes: number;
|
|
28
33
|
maxInjectionBytes: number;
|
|
29
34
|
}
|
|
30
35
|
|
|
@@ -33,6 +38,10 @@ export interface ExtractMemoriesConfig {
|
|
|
33
38
|
model?: string;
|
|
34
39
|
thinkLevel: ThinkLevel;
|
|
35
40
|
maxContextTokens: number;
|
|
41
|
+
/** 渲染进 extract prompt 时,单条 tool_result 的字符上限(spec §11.2)。 */
|
|
42
|
+
maxToolResultChars: number;
|
|
43
|
+
/** 渲染进 extract prompt 时,单条 assistant 文本的字符上限(spec §11.2)。 */
|
|
44
|
+
maxAssistantChars: number;
|
|
36
45
|
}
|
|
37
46
|
|
|
38
47
|
export interface MemoryConfig {
|
|
@@ -44,10 +53,20 @@ export interface MemoryConfig {
|
|
|
44
53
|
memIndexMaxLines: number;
|
|
45
54
|
/** Write capacity: max bytes of serialized MEMORY.md index. */
|
|
46
55
|
memIndexMaxBytes: number;
|
|
47
|
-
/**
|
|
56
|
+
/**
|
|
57
|
+
* 注入截断:`memory_index` section 最多带多少**行**索引。
|
|
58
|
+
* 与 `memIndexMaxLines` **读写同口径**(D3):v2 一行 = 一条 entry,注入预算若小于写入
|
|
59
|
+
* 上限,写满的记忆就有一部分永远看不见。
|
|
60
|
+
*/
|
|
48
61
|
memIndexInjectMaxLines: number;
|
|
49
|
-
/**
|
|
62
|
+
/** 注入截断:`memory_index` section 最多带多少**字节**(与 `memIndexMaxBytes` 同口径,D3)。 */
|
|
50
63
|
memIndexInjectMaxBytes: number;
|
|
64
|
+
/**
|
|
65
|
+
* 两级锁的参数(spec §5.2)。结构与 `StoreConfig["lock"]` 逐字一致,因此可以原样传给
|
|
66
|
+
* `new MemoryStore({ ..., lock: config.lock })`。
|
|
67
|
+
* **没有** ttl / 心跳 / 接管字段:跨进程 `.lock` 永远只持毫秒且永不自动回收。
|
|
68
|
+
*/
|
|
69
|
+
lock: { timeoutMs: number; snapshotKeep: number };
|
|
51
70
|
dream: {
|
|
52
71
|
nudgeAfterSessions: number;
|
|
53
72
|
nudgeAfterHours: number;
|
|
@@ -66,24 +85,30 @@ export interface MemoryConfig {
|
|
|
66
85
|
|
|
67
86
|
export const DEFAULT_CONFIG: MemoryConfig = {
|
|
68
87
|
enabled: true,
|
|
88
|
+
// headless 子会话默认只在内存里跑:extract / dream / 侧查询都不该往用户的 sessions 目录里落盘。
|
|
89
|
+
defaults: { sessionPersistence: { enabled: false } },
|
|
69
90
|
memoryDir: join(homedir(), CONFIG_DIR_NAME, "memory"),
|
|
70
91
|
memIndexMaxLines: 200,
|
|
71
92
|
memIndexMaxBytes: 25600,
|
|
72
|
-
|
|
73
|
-
|
|
93
|
+
// 读写同口径(D3):200 行 = 200 条记忆,写满时注入也看得到全部。
|
|
94
|
+
memIndexInjectMaxLines: 200,
|
|
95
|
+
memIndexInjectMaxBytes: 25600,
|
|
96
|
+
lock: { timeoutMs: 5000, snapshotKeep: 5 },
|
|
74
97
|
dream: { nudgeAfterSessions: 5, nudgeAfterHours: 24, thinkLevel: "high" },
|
|
75
98
|
sessionSearch: { maxSessions: 10, maxMatches: 5 },
|
|
76
99
|
autoSurfacing: {
|
|
77
100
|
enabled: true,
|
|
78
101
|
thinkLevel: "off",
|
|
79
102
|
maxFiles: 3,
|
|
80
|
-
|
|
103
|
+
maxEntryBytes: 3072,
|
|
81
104
|
maxInjectionBytes: 10240,
|
|
82
105
|
},
|
|
83
106
|
extractMemories: {
|
|
84
107
|
enabled: true,
|
|
85
108
|
thinkLevel: "high",
|
|
86
109
|
maxContextTokens: 2000,
|
|
110
|
+
maxToolResultChars: 500,
|
|
111
|
+
maxAssistantChars: 2000,
|
|
87
112
|
},
|
|
88
113
|
};
|
|
89
114
|
|
package/src/dream.ts
CHANGED
|
@@ -1,84 +1,126 @@
|
|
|
1
|
+
import { readdir } from "node:fs/promises";
|
|
2
|
+
import { join } from "node:path";
|
|
1
3
|
import type { Model } from "@earendil-works/pi-ai";
|
|
2
|
-
import type { ModelRegistry } from "@earendil-works/pi-coding-agent";
|
|
4
|
+
import type { ModelRegistry, ToolDefinition } from "@earendil-works/pi-coding-agent";
|
|
3
5
|
import { runHeadlessAgent } from "./agent-runner";
|
|
4
6
|
import type { SessionPersistenceConfig, ThinkLevel } from "./config";
|
|
7
|
+
import { BACKUP_DIR, type MemoryStore } from "./memory-store";
|
|
8
|
+
import { createSnapshot } from "./snapshot";
|
|
5
9
|
|
|
6
|
-
/** Build dream consolidation task. */
|
|
7
|
-
export function buildDreamTask(
|
|
8
|
-
return `You are a memory consolidation agent.
|
|
9
|
-
|
|
10
|
+
/** Build the dream consolidation task. dream 只有 `memory` 工具的 7 个 action,没有任何文件工具。 */
|
|
11
|
+
export function buildDreamTask(maxLines: number): string {
|
|
12
|
+
return `You are a memory consolidation agent. The memory store is a flat directory: one memory per file, and MEMORY.md holds exactly one index line per memory, formatted as - [name](file.md) — description.
|
|
13
|
+
|
|
14
|
+
Your ONLY tool is \`memory\`. You have no read, write, edit, ls or bash access — you cannot touch files directly, and you do not need to.
|
|
15
|
+
|
|
16
|
+
memory(action="list") — every memory: name, type, modified, description, file.
|
|
17
|
+
memory(action="search", query="…") — full-text search; returns the whole body of each match.
|
|
18
|
+
memory(action="add", name, description, type, content) — create a memory, or overwrite the one whose name matches exactly.
|
|
19
|
+
memory(action="replace", name, description, type, content) — rewrite an existing memory.
|
|
20
|
+
memory(action="rename", name, new_name) — retitle a memory; its file name and index line follow.
|
|
21
|
+
memory(action="remove", name) — delete a memory and its index line.
|
|
22
|
+
memory(action="rebuild_index") — regenerate MEMORY.md from the directory.
|
|
10
23
|
|
|
11
24
|
Phase 1 — Orient:
|
|
12
|
-
-
|
|
13
|
-
-
|
|
14
|
-
- Skim each topic file to understand its contents
|
|
25
|
+
- Call list to see every memory.
|
|
26
|
+
- Call search (and list) to read the bodies of the memories you intend to touch. Never rewrite a memory you have not read.
|
|
15
27
|
|
|
16
28
|
Phase 2 — Gather Signal:
|
|
17
|
-
-
|
|
18
|
-
-
|
|
19
|
-
-
|
|
20
|
-
-
|
|
29
|
+
- Duplicates: several memories stating the same fact.
|
|
30
|
+
- Contradictions: two memories that cannot both be true. Pick the accurate one.
|
|
31
|
+
- Outdated: memories superseded by later ones, or relative dates ("today", "last week") that should be absolute.
|
|
32
|
+
- Merge candidates: small fragments that only make sense together.
|
|
33
|
+
- Weak descriptions: vague text that gives a future session no way to decide relevance.
|
|
21
34
|
|
|
22
35
|
Phase 3 — Consolidate:
|
|
23
|
-
- Merge
|
|
24
|
-
-
|
|
25
|
-
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
-
|
|
36
|
+
- Merge = replace(target name, the merged body) + remove(the other name). Keep the surviving name readable.
|
|
37
|
+
- Retitle with rename when the name no longer matches the content, or when two names are confusable.
|
|
38
|
+
- Rewrite any description that is not self-contained. It is the ONLY text a future session sees when deciding whether to recall this memory, so it must state what, where and which value.
|
|
39
|
+
Bad: "Debugging tips"
|
|
40
|
+
Good: "staging SSH listens on 2222, not 22; MySQL pool times out after 30s"
|
|
41
|
+
- Remove memories that are no longer true or no longer useful.
|
|
29
42
|
|
|
30
43
|
Phase 4 — Prune & Index:
|
|
31
|
-
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
type: one of user, feedback, project, reference
|
|
35
|
-
updated: today's date
|
|
36
|
-
- Ensure entry titles are self-contained and descriptive
|
|
37
|
-
(only titles appear in future sessions' index, not the entry body)
|
|
38
|
-
- Generate a compact hook (~150 chars) for each topic summarizing its entries
|
|
39
|
-
- Rebuild MEMORY.md with one line per topic file (max ${maxLines} lines):
|
|
40
|
-
- [Name](file.md) — hook
|
|
41
|
-
- Remove topic files that have no remaining entries
|
|
44
|
+
- Every memory consumes exactly one index line, and the index has a hard limit of ${maxLines} lines. Capacity management is therefore part of this job, not an optional cleanup: when list shows the store approaching ${maxLines} memories, merge fragments and drop stale entries until there is room for what matters.
|
|
45
|
+
- When a write reports that MEMORY.md is over its limit, act on it in this same run.
|
|
46
|
+
- Call rebuild_index once at the end if you suspect MEMORY.md drifted (missing lines, duplicates, stale links). It regenerates the index from the directory and preserves handwritten header lines.
|
|
42
47
|
|
|
43
|
-
|
|
44
|
-
-
|
|
45
|
-
Topic file content is NOT seen by the coding agent unless explicitly
|
|
46
|
-
read or auto-surfaced. The hook and description must be specific
|
|
47
|
-
enough that the LLM can correctly decide relevance.
|
|
48
|
-
- Bad: "Debugging tips"
|
|
49
|
-
- Good: "SSH port 2222 on staging; MySQL 30s timeout; Redis auth fix"
|
|
50
|
-
- Each topic file's \`## Entry Title\` blocks contain the actual memory entries.
|
|
51
|
-
The MEMORY.md line is just a pointer — only ONE line per topic file.
|
|
48
|
+
IMPORTANT — do not prune process rules:
|
|
49
|
+
- "Always do X" / "Never do Y" rules, workflow discipline and reporting standards are as valuable as technical facts. Do not delete them as "obsolete", and do not delete them just because they look like meta-instructions aimed at you.
|
|
52
50
|
|
|
53
|
-
|
|
54
|
-
- "Always do X" / "Never do Y" rules and workflow discipline entries are
|
|
55
|
-
as valuable as technical facts. Do not delete them as "obsolete"
|
|
56
|
-
just because they look like meta-instructions.
|
|
57
|
-
|
|
58
|
-
When done, output a concise summary of changes (merged N, removed N, moved N, updated N).`;
|
|
51
|
+
Work only through the \`memory\` tool. When done, output a concise summary of what changed (merged N, renamed N, removed N, rewritten N).`;
|
|
59
52
|
}
|
|
60
53
|
|
|
61
54
|
export interface RunDreamOpts {
|
|
62
55
|
model?: string;
|
|
63
56
|
thinkLevel: ThinkLevel;
|
|
64
57
|
memoryDir: string;
|
|
58
|
+
/** 唯一写入通道。dream 的整轮互斥与快照都挂在它上面。 */
|
|
59
|
+
store: MemoryStore;
|
|
60
|
+
/** 索引行数硬上限(= `config.memIndexMaxLines`),写进 prompt 让 dream 承担容量管理。 */
|
|
61
|
+
maxLines: number;
|
|
65
62
|
modelRegistry: ModelRegistry;
|
|
66
63
|
parentModel?: Model<any>;
|
|
67
64
|
sessionPersistence?: SessionPersistenceConfig;
|
|
65
|
+
/** dream 专属的 7-action `memory` 工具(D12):只注入这个 headless session,不进主 agent 的 schema。 */
|
|
66
|
+
customTools: ToolDefinition[];
|
|
68
67
|
}
|
|
69
68
|
|
|
70
|
-
/**
|
|
69
|
+
/**
|
|
70
|
+
* 要被快照的普通文件:目录里的**文件**,跳过一切 dotfile 与子目录。
|
|
71
|
+
*
|
|
72
|
+
* - `.backups`(目录)跳过 —— 快照不套快照;不带 `withFileTypes` 时它会被 `cp` 递归复制。
|
|
73
|
+
* - `sessions/`(目录,`sessionPersistence.enabled` 时存在)跳过 —— 同理,而且它可能很大。
|
|
74
|
+
* `createSnapshot` 的 `cp` 遇到目录会抛 `ERR_FS_EISDIR`(非 ENOENT → fail-closed 上抛 → dream 直接失败)。
|
|
75
|
+
* - `.lock` / `.migrated` / `.dream-meta.json` 跳过(都是 dotfile):锁记录与标记不属于记忆内容。
|
|
76
|
+
* - 其余全部 `*.md`(entry 文件 + `MEMORY.md`)都会被快照。
|
|
77
|
+
*/
|
|
78
|
+
async function snapshotFiles(memoryDir: string): Promise<string[]> {
|
|
79
|
+
const entries = await readdir(memoryDir, { withFileTypes: true }).catch(() => []);
|
|
80
|
+
return entries
|
|
81
|
+
.filter((entry) => entry.isFile() && !entry.name.startsWith("."))
|
|
82
|
+
.map((entry) => entry.name)
|
|
83
|
+
.sort();
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
/**
|
|
87
|
+
* 跑一轮 dream:整轮持有进程内逻辑锁 → 对整目录拍一次快照 → 起一个只有 `memory` 工具的
|
|
88
|
+
* headless agent(spec §5.2 / §6 / §12)。
|
|
89
|
+
*
|
|
90
|
+
* 整轮持锁是刻意的:dream 的多次 `replace` / `remove` 必须对同进程的 `memory add` 与 extract
|
|
91
|
+
* 呈现为**一个**原子区间,否则用户会在 dream 中途读到半合并的状态。锁在进程内(不是 `.lock`),
|
|
92
|
+
* 所以跨进程的其它 worktree 仍能正常写入 —— 它们的每次物理写入只被挡毫秒级。
|
|
93
|
+
*
|
|
94
|
+
* 其内部的每个原语调用必须传 `{ skipLogicalLock: true, skipSnapshot: true }`(由调用方构造
|
|
95
|
+
* `customTools` 时设置),否则它们会去抢自己已持有的锁而自锁到超时,并且会为同一批变更各拍一次快照。
|
|
96
|
+
*/
|
|
71
97
|
export async function runDream(opts: RunDreamOpts): Promise<string> {
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
98
|
+
return opts.store.withLogicalLock(async () => {
|
|
99
|
+
// 启动自检:忘了包 withLogicalLock 会静默失去整轮互斥(spec §5.2 末段)。
|
|
100
|
+
if (!opts.store.logicalLockActive()) {
|
|
101
|
+
throw new Error("dream must run under the memory logical lock");
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
// 进入时对整目录拍一次快照;这一轮里各原语的逐文件快照被 skipSnapshot 跳过(spec §6)。
|
|
105
|
+
const files = await snapshotFiles(opts.memoryDir);
|
|
106
|
+
await createSnapshot(join(opts.memoryDir, BACKUP_DIR), "dream", files, opts.memoryDir, {
|
|
107
|
+
keep: opts.store.cfg.lock.snapshotKeep,
|
|
108
|
+
});
|
|
109
|
+
|
|
110
|
+
return runHeadlessAgent({
|
|
111
|
+
task: buildDreamTask(opts.maxLines),
|
|
112
|
+
cwd: opts.memoryDir,
|
|
113
|
+
modelRegistry: opts.modelRegistry,
|
|
114
|
+
model: opts.model,
|
|
115
|
+
parentModel: opts.parentModel,
|
|
116
|
+
thinkLevel: opts.thinkLevel,
|
|
117
|
+
maxTurns: undefined,
|
|
118
|
+
timeoutMs: 600_000,
|
|
119
|
+
// 没有裸写权限(spec §12.1):dream 的重构能力边界由 7 个原语定义,因此首次可被单测覆盖。
|
|
120
|
+
// 必须用 noTools 而不是 tools: [] —— 后者是白名单,会把 customTools(memory 工具)一起滤掉。
|
|
121
|
+
noTools: "builtin",
|
|
122
|
+
customTools: opts.customTools,
|
|
123
|
+
sessionPersistence: opts.sessionPersistence,
|
|
124
|
+
});
|
|
83
125
|
});
|
|
84
126
|
}
|
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
export const ENTRY_TYPES = ["user", "feedback", "project", "reference"] as const;
|
|
2
|
+
export type EntryType = (typeof ENTRY_TYPES)[number];
|
|
3
|
+
|
|
4
|
+
export interface EntryMeta {
|
|
5
|
+
name: string;
|
|
6
|
+
description: string;
|
|
7
|
+
type: EntryType;
|
|
8
|
+
/** YYYY-MM-DD */
|
|
9
|
+
created: string;
|
|
10
|
+
/** ISO 8601 时间戳 */
|
|
11
|
+
modified: string;
|
|
12
|
+
}
|
|
13
|
+
|
|
14
|
+
export interface ParsedEntryFile {
|
|
15
|
+
meta: EntryMeta;
|
|
16
|
+
body: string;
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
const FIELD_RE = /^([A-Za-z_][A-Za-z0-9_]*):[ \t]*(.*)$/;
|
|
20
|
+
const MD_MARKER_RE = /^\s*(#{1,6}\s+|[-*+]\s+|\d+[.)]\s+)/;
|
|
21
|
+
const SENTENCE_END_RE = /[。!?!?.]/;
|
|
22
|
+
|
|
23
|
+
const DESCRIPTION_MAX = 200;
|
|
24
|
+
|
|
25
|
+
export function isEntryType(value: string): value is EntryType {
|
|
26
|
+
return (ENTRY_TYPES as readonly string[]).includes(value);
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
/** 序列化为「frontmatter + 空行 + 正文」的 markdown。 */
|
|
30
|
+
export function serializeEntryFile(meta: EntryMeta, body: string): string {
|
|
31
|
+
const header = [
|
|
32
|
+
"---",
|
|
33
|
+
`name: ${meta.name}`,
|
|
34
|
+
`description: ${meta.description}`,
|
|
35
|
+
`type: ${meta.type}`,
|
|
36
|
+
`created: ${meta.created}`,
|
|
37
|
+
`modified: ${meta.modified}`,
|
|
38
|
+
"---",
|
|
39
|
+
].join("\n");
|
|
40
|
+
const trimmed = body.trim();
|
|
41
|
+
return trimmed.length === 0 ? `${header}\n` : `${header}\n\n${trimmed}\n`;
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
/** 解析 entry 文件。frontmatter 缺失或字段不全时返回 null。 */
|
|
45
|
+
export function parseEntryFile(raw: string): ParsedEntryFile | null {
|
|
46
|
+
// CRLF 会让整份文件解析不出来(首行分隔匹配不上、字段行的值也匹配不到 \r),
|
|
47
|
+
// 该 entry 就会从清单、搜索、remove/replace 定位里一起消失 —— 静默丢记忆。
|
|
48
|
+
// 归一只在解析入口做一次,serializeEntryFile 仍然只输出 \n。
|
|
49
|
+
const text = raw.includes("\r") ? raw.replace(/\r\n?/g, "\n") : raw;
|
|
50
|
+
if (!text.startsWith("---\n")) return null;
|
|
51
|
+
const end = text.indexOf("\n---", 4);
|
|
52
|
+
if (end === -1) return null;
|
|
53
|
+
|
|
54
|
+
const fields: Record<string, string> = {};
|
|
55
|
+
for (const line of text.slice(4, end).split("\n")) {
|
|
56
|
+
const m = line.match(FIELD_RE);
|
|
57
|
+
if (m) fields[m[1]] = m[2].trim();
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
const name = fields.name ?? "";
|
|
61
|
+
const description = fields.description ?? "";
|
|
62
|
+
const type = fields.type ?? "";
|
|
63
|
+
const created = fields.created ?? "";
|
|
64
|
+
const modified = fields.modified ?? "";
|
|
65
|
+
// description 必须「存在」但允许为空。deriveDescription 对「首行只有 markdown 标记」的正文会返回 "",
|
|
66
|
+
// 若把空值当成缺字段,写出的文件将永远解析不了(对 store 静默不可见)。
|
|
67
|
+
if (!name || !created || !modified || !isEntryType(type) || !("description" in fields)) return null;
|
|
68
|
+
|
|
69
|
+
const bodyStart = text.indexOf("\n", end + 1);
|
|
70
|
+
const body = bodyStart === -1 ? "" : text.slice(bodyStart + 1).trim();
|
|
71
|
+
return { meta: { name, description, type, created, modified }, body };
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
/** 从正文派生一行摘要:取首个非空行的首句,剥掉 markdown 标记。 */
|
|
75
|
+
export function deriveDescription(body: string, max = DESCRIPTION_MAX): string {
|
|
76
|
+
const firstLine = body.split("\n").find((line) => line.trim().length > 0) ?? "";
|
|
77
|
+
const cleaned = firstLine.replace(MD_MARKER_RE, "").trim();
|
|
78
|
+
const stop = cleaned.search(SENTENCE_END_RE);
|
|
79
|
+
const sentence = (stop === -1 ? cleaned : cleaned.slice(0, stop + 1)).trim();
|
|
80
|
+
return sentence.length <= max ? sentence : `${sentence.slice(0, max - 1)}…`;
|
|
81
|
+
}
|
|
@@ -0,0 +1,112 @@
|
|
|
1
|
+
// name 与 file 两组都必须是**非贪婪**的:(1) file 可能含 `)`(entryFileName 不剥括号);
|
|
2
|
+
// (2) name 可能含 `]`。贪婪组会把分割点吃掉,[^\]]+ / [^)]+ 则表示这两种行根本匹配不上。
|
|
3
|
+
const LINE_RE = /^-\s+\[(.+?)\]\((.+?)\)\s*—\s*(.*)$/;
|
|
4
|
+
|
|
5
|
+
export interface IndexLineEntry {
|
|
6
|
+
name: string;
|
|
7
|
+
file: string;
|
|
8
|
+
description: string;
|
|
9
|
+
lineNo: number;
|
|
10
|
+
}
|
|
11
|
+
|
|
12
|
+
export interface ParsedEntryIndex {
|
|
13
|
+
lines: string[];
|
|
14
|
+
entries: IndexLineEntry[];
|
|
15
|
+
/** 非空但无法识别的行数(标题、分组、注释等)。 */
|
|
16
|
+
unrecognized: number;
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
/**
|
|
20
|
+
* 索引唯一的按行拆分入口(解析与写入必须共用它,否则同一个 `lineNo` 在两处含义不同)。
|
|
21
|
+
*
|
|
22
|
+
* CRLF(以及老 Mac 的单独 CR)必须在这里归一:JS 的 `.` 不匹配 `\r`,且无 `m` 标志的 `$`
|
|
23
|
+
* 也无法在一个 `\r` 之前成立 —— `LINE_RE` 对 CRLF 行会**完全匹配不上**,于是每一行都被计成
|
|
24
|
+
* `unrecognized`:删除变成空操作(留下死链)、追加变成重复行、`rebuildIndex` 把整块当成手写头部。
|
|
25
|
+
* 归一之后写回仍只输出 `\n`(`joinLines`),因此对 CRLF 文件的首次写入会把它转成 LF ——
|
|
26
|
+
* 这是有意的,它消除的是「混合行尾」这个更糟的中间态。
|
|
27
|
+
*/
|
|
28
|
+
function splitLines(raw: string): string[] {
|
|
29
|
+
const text = raw.includes("\r") ? raw.replace(/\r\n?/g, "\n") : raw;
|
|
30
|
+
if (text.length === 0) return [];
|
|
31
|
+
const lines = text.split("\n");
|
|
32
|
+
if (lines[lines.length - 1] === "") lines.pop();
|
|
33
|
+
return lines;
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
function joinLines(lines: string[]): string {
|
|
37
|
+
return lines.length === 0 ? "" : `${lines.join("\n")}\n`;
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
export function parseEntryIndex(raw: string): ParsedEntryIndex {
|
|
41
|
+
const lines = splitLines(raw);
|
|
42
|
+
const entries: IndexLineEntry[] = [];
|
|
43
|
+
let unrecognized = 0;
|
|
44
|
+
for (let i = 0; i < lines.length; i++) {
|
|
45
|
+
const line = lines[i];
|
|
46
|
+
if (line.trim().length === 0) continue;
|
|
47
|
+
const m = line.match(LINE_RE);
|
|
48
|
+
if (m) entries.push({ name: m[1].trim(), file: m[2].trim(), description: m[3].trim(), lineNo: i });
|
|
49
|
+
else unrecognized++;
|
|
50
|
+
}
|
|
51
|
+
return { lines, entries, unrecognized };
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
export function formatIndexLine(name: string, file: string, description: string): string {
|
|
55
|
+
return `- [${name}](${file}) — ${description}`;
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
/**
|
|
59
|
+
* 外科式写入:只改目标行,其余行(标题、分组、注释、空行)逐字保留。
|
|
60
|
+
* 新增时插到最后一条已识别行之后;`atLineNo` 用于改名时保持行位置。
|
|
61
|
+
*/
|
|
62
|
+
export function upsertIndexLine(
|
|
63
|
+
raw: string,
|
|
64
|
+
entry: { name: string; file: string; description: string },
|
|
65
|
+
options?: { atLineNo?: number },
|
|
66
|
+
): string {
|
|
67
|
+
const parsed = parseEntryIndex(raw);
|
|
68
|
+
const line = formatIndexLine(entry.name, entry.file, entry.description);
|
|
69
|
+
const lines = splitLines(raw);
|
|
70
|
+
|
|
71
|
+
if (options?.atLineNo !== undefined) {
|
|
72
|
+
lines[options.atLineNo] = line;
|
|
73
|
+
return joinLines(lines);
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
const existing = parsed.entries.find((e) => e.file === entry.file);
|
|
77
|
+
if (existing) {
|
|
78
|
+
lines[existing.lineNo] = line;
|
|
79
|
+
return joinLines(lines);
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
const last = parsed.entries[parsed.entries.length - 1];
|
|
83
|
+
if (last) lines.splice(last.lineNo + 1, 0, line);
|
|
84
|
+
else lines.push(line);
|
|
85
|
+
return joinLines(lines);
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
/**
|
|
89
|
+
* 只删匹配到的那一行,其余行(含手写标题、分组、注释)逐字保留。
|
|
90
|
+
* 没匹配上时**原样返回输入**,不做任何重写(否则一次「无操作的删除」也会补上一个尾换行、
|
|
91
|
+
* 或把 CRLF 文件转成 LF,让调用方在无意间改动用户的文件)。
|
|
92
|
+
*/
|
|
93
|
+
export function removeIndexLine(raw: string, file: string): string {
|
|
94
|
+
const target = parseEntryIndex(raw).entries.find((e) => e.file === file);
|
|
95
|
+
if (!target) return raw;
|
|
96
|
+
const lines = splitLines(raw);
|
|
97
|
+
lines.splice(target.lineNo, 1);
|
|
98
|
+
return joinLines(lines);
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
export interface IndexCapacity {
|
|
102
|
+
lineCount: number;
|
|
103
|
+
byteLength: number;
|
|
104
|
+
ok: boolean;
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
export function indexCapacity(raw: string, maxLines: number, maxBytes: number): IndexCapacity {
|
|
108
|
+
// 与解析共用同一切分口径,避开「同一规则在三处各自实现」的漂移
|
|
109
|
+
const lineCount = splitLines(raw).filter((line) => line.trim().length > 0).length;
|
|
110
|
+
const byteLength = Buffer.byteLength(raw, "utf8");
|
|
111
|
+
return { lineCount, byteLength, ok: lineCount <= maxLines && byteLength <= maxBytes };
|
|
112
|
+
}
|