@yandy0725/pi-memory 1.4.0 → 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/extract.ts CHANGED
@@ -2,56 +2,326 @@ import type { Model } from "@earendil-works/pi-ai";
2
2
  import type { ModelRegistry, ToolDefinition } from "@earendil-works/pi-coding-agent";
3
3
  import { runHeadlessAgent } from "./agent-runner";
4
4
  import type { SessionPersistenceConfig, ThinkLevel } from "./config";
5
+ import type { MemoryStore } from "./memory-store";
6
+
7
+ /**
8
+ * 渲染进 extract prompt 的一条消息。
9
+ *
10
+ * v1 只把「本轮第一条 user 消息 + 最后一条 assistant 消息」压成 `{role, content}` 字符串交给
11
+ * extract,中间的纠正、工具调用与工具结果全部丢失(spec §11.1)—— 这是提取失真的根因。
12
+ * v2 保留角色、tool_call 的名称与参数摘要、tool_result 的成功/失败标记。
13
+ */
14
+ export interface ExtractMessage {
15
+ role: "user" | "assistant" | "toolResult";
16
+ text?: string;
17
+ toolCalls?: Array<{ name: string; args: string }>;
18
+ /** 产生该结果的工具名。保留供诊断(Plan C 的通知会用到),渲染时不出现。 */
19
+ toolName?: string;
20
+ isError?: boolean;
21
+ }
22
+
23
+ export interface ConversationLimits {
24
+ maxToolResultChars: number;
25
+ maxAssistantChars: number;
26
+ maxContextTokens: number;
27
+ }
5
28
 
6
29
  export interface RunExtractOpts {
7
30
  model?: string;
8
31
  thinkLevel: ThinkLevel;
9
32
  memoryDir: string;
10
- messages: Array<{ role: string; content: string }>;
33
+ /** 唯一写入通道:extract 的整轮互斥挂在它的 `tryWithLogicalLock` 上。 */
34
+ store: MemoryStore;
35
+ /** pi 的 `agent_end` 消息原样传入(`AgentMessage[]`)。形状不认识的消息按「尽量保留文本」处理。 */
36
+ messages: unknown[];
11
37
  maxContextTokens: number;
38
+ maxToolResultChars: number;
39
+ maxAssistantChars: number;
12
40
  modelRegistry: ModelRegistry;
13
41
  parentModel?: Model<any>;
14
42
  sessionPersistence?: SessionPersistenceConfig;
15
- customTools?: ToolDefinition[];
16
- agentsMdBlocks?: string[]; // 新增
43
+ /** extract 子会话专属的工具集(5 个 action,D12)。 */
44
+ customTools: ToolDefinition[];
45
+ agentsMdBlocks?: string[];
46
+ }
47
+
48
+ /** 单条文本的尾部截断。user 消息**不**走这里(spec §11.2:优先保留全部 user 消息)。 */
49
+ function clip(text: string, max: number): string {
50
+ if (max <= 0 || text.length <= max) return text;
51
+ return `${text.slice(0, max)}\n[truncated: ${text.length - max} chars omitted]`;
52
+ }
53
+
54
+ /** 中段裁减的标记行本体(与字符串回退共用,保证两种路径逐字相同)。 */
55
+ const middleMarkerLabel = (omitted: number): string => `[truncated: ${omitted} chars omitted from the middle]`;
56
+
57
+ /**
58
+ * 总预算超限时的**中段**裁减:首尾优先保留(spec §11.2)。
59
+ * 开头是用户的原始诉求、结尾是最终的结论与纠正,中段大多是可以牺牲的工具输出。
60
+ */
61
+ const middleMarker = (omitted: number): string => `\n${middleMarkerLabel(omitted)}\n`;
62
+
63
+ /**
64
+ * 字符串级的中段裁减:只剩 user 块仍超预算时的回退。
65
+ *
66
+ * `alreadyOmitted` 是块级裁减已经丢掉的字符数,它并入标记里的 N,并且**不再插入第二个标记**
67
+ * —— 输出里 `[truncated: …]` 恒为一个。旧实现直接对「已含块级标记的文本」再裁一次,
68
+ * 于是极端预算下会出现两个标记,而且回退标记的 N 把旧标记自身的长度也算成「省略的正文」
69
+ * (Plan B ledger 的 Minor)。
70
+ */
71
+ function clipMiddle(text: string, maxChars: number, alreadyOmitted = 0): string {
72
+ if (maxChars <= 0 || text.length <= maxChars) return text;
73
+ // 块级标记自己占的字符既不是被省略的正文,也不该在输出里出现第二次。
74
+ // 它可能是独立一行(后面跟着 `\n`),也可能被 `assemble` 追加在末尾(后面没有 `\n`)。
75
+ const label = alreadyOmitted > 0 ? middleMarkerLabel(alreadyOmitted) : "";
76
+ const body = label === "" ? text : text.replace(`${label}\n`, "").replace(label, "");
77
+ // 给标记文本预留位置(按一个六位数省略量估算),避免「裁减之后反而更长」。
78
+ const budget = Math.max(0, maxChars - middleMarker(999999).length);
79
+ const head = Math.min(Math.ceil(budget / 2), body.length);
80
+ const tail = Math.min(budget - head, Math.max(0, body.length - head));
81
+ const omitted = alreadyOmitted + (body.length - head - tail);
82
+ return `${body.slice(0, head)}${middleMarker(omitted)}${body.slice(body.length - tail)}`;
17
83
  }
18
84
 
19
- /** Build extraction task prompt using memory tools instead of raw file I/O. */
85
+ /** 从字符串或内容块数组里抠出文本。`joiner`/`images` 供 custom 消息(空格连接、不要图片占位)使用。 */
86
+ function textOf(content: unknown, opts: { joiner?: string; images?: boolean } = {}): string {
87
+ if (typeof content === "string") return content;
88
+ if (!Array.isArray(content)) return "";
89
+ const joiner = opts.joiner ?? "\n";
90
+ const keepImages = opts.images ?? true;
91
+ const parts: string[] = [];
92
+ for (const block of content) {
93
+ if (!block || typeof block !== "object") continue;
94
+ const part = block as Record<string, unknown>;
95
+ if (part.type === "text" && typeof part.text === "string") parts.push(part.text);
96
+ else if (keepImages && part.type === "image") parts.push("[image]");
97
+ }
98
+ return parts.join(joiner);
99
+ }
100
+
101
+ /** 未知 role 的兜底:按常见字段顺序尽力取文本(`content` → `output` → `text` → `summary`)。 */
102
+ function bestEffortText(m: Record<string, unknown>): string {
103
+ for (const field of [m.content, m.output, m.text, m.summary]) {
104
+ const text = textOf(field);
105
+ if (text) return text;
106
+ }
107
+ return "";
108
+ }
109
+
110
+ function toolCallsOf(content: unknown): Array<{ name: string; args: string }> {
111
+ if (!Array.isArray(content)) return [];
112
+ const out: Array<{ name: string; args: string }> = [];
113
+ for (const block of content) {
114
+ if (!block || typeof block !== "object") continue;
115
+ const part = block as Record<string, unknown>;
116
+ if (part.type !== "toolCall" || typeof part.name !== "string") continue;
117
+ let args = "";
118
+ try {
119
+ args = JSON.stringify(part.arguments ?? {});
120
+ } catch {
121
+ args = "";
122
+ }
123
+ out.push({ name: part.name, args });
124
+ }
125
+ return out;
126
+ }
127
+
128
+ /**
129
+ * 把 pi 的消息序列(`AgentMessage` 联合类型)转成渲染用的结构。
130
+ *
131
+ * 除了 user / assistant / toolResult,还要认领 pi 的机器消息:
132
+ * - `bashExecution` → `toolResult`(`!!` 前缀即 `excludeFromContext` 的则整条跳过),
133
+ * 这样 `command`/`exitCode` 不再丢失,且自动受 `maxToolResultChars` 与 `[error] ` 前缀约束;
134
+ * - `branchSummary` / `compactionSummary` → `assistant`(带标签),受 `maxAssistantChars` 约束;
135
+ * - `custom` → `assistant`;唯一例外是 `customType === "memory-auto-surfacing"`(pi-memory
136
+ * 自己注入的 `<relevant_memories>`)—— 把记忆正文当 assistant 文本再喂给 extract 等于让
137
+ * extract 从自己的记忆里反复提取,因此整条跳过。
138
+ *
139
+ * 未知 role **绝不**当作 user(user 文本不截断,且 prompt 会把它当规则/纠正):按 assistant
140
+ * 尽力保留文本(`content` → `output` → `text` → `summary`),抠不出来才跳过。
141
+ */
142
+ export function toExtractMessages(messages: unknown[]): ExtractMessage[] {
143
+ const out: ExtractMessage[] = [];
144
+ for (const raw of messages) {
145
+ if (!raw || typeof raw !== "object") continue;
146
+ const m = raw as Record<string, unknown>;
147
+ const role = typeof m.role === "string" ? m.role : "";
148
+
149
+ if (role === "user") {
150
+ const text = textOf(m.content);
151
+ if (text) out.push({ role: "user", text });
152
+ continue;
153
+ }
154
+ if (role === "assistant") {
155
+ const text = textOf(m.content);
156
+ const toolCalls = toolCallsOf(m.content);
157
+ if (!text && toolCalls.length === 0) continue;
158
+ out.push({
159
+ role: "assistant",
160
+ ...(text ? { text } : {}),
161
+ ...(toolCalls.length > 0 ? { toolCalls } : {}),
162
+ });
163
+ continue;
164
+ }
165
+ if (role === "toolResult") {
166
+ out.push({
167
+ role: "toolResult",
168
+ text: textOf(m.content),
169
+ toolName: typeof m.toolName === "string" ? m.toolName : undefined,
170
+ isError: m.isError === true,
171
+ });
172
+ continue;
173
+ }
174
+ if (role === "bashExecution") {
175
+ // `!!` 前缀(excludeFromContext)明确不进 LLM 上下文,extract 也不该看到。
176
+ if (m.excludeFromContext === true) continue;
177
+ const command = typeof m.command === "string" ? m.command : "";
178
+ const output = typeof m.output === "string" ? m.output : "";
179
+ const exitCode = typeof m.exitCode === "number" ? m.exitCode : undefined;
180
+ out.push({
181
+ role: "toolResult",
182
+ toolName: "bash",
183
+ text: `$ ${command}\n${output}`,
184
+ isError: exitCode !== undefined && exitCode !== 0,
185
+ });
186
+ continue;
187
+ }
188
+ if (role === "branchSummary" || role === "compactionSummary") {
189
+ const summary = typeof m.summary === "string" ? m.summary : "";
190
+ if (!summary) continue;
191
+ const label = role === "branchSummary" ? "[branch summary]" : "[compaction summary]";
192
+ out.push({ role: "assistant", text: `${label} ${summary}` });
193
+ continue;
194
+ }
195
+ if (role === "custom") {
196
+ // pi-memory 自己注入的 <relevant_memories> 是「记忆正文的渲染」,不是本轮对话内容。
197
+ // 其他 customType(插件通知等)保持渲染为 assistant。
198
+ if (m.customType === "memory-auto-surfacing") continue;
199
+ const text = textOf(m.content, { joiner: " ", images: false });
200
+ if (text) out.push({ role: "assistant", text });
201
+ continue;
202
+ }
203
+
204
+ const text = bestEffortText(m);
205
+ if (text) out.push({ role: "assistant", text });
206
+ }
207
+ return out;
208
+ }
209
+
210
+ /** 单条消息的渲染块(块之间以 `\n` 连接,块内不含分隔换行)。 */
211
+ function renderBlock(m: ExtractMessage, index: number, limits: ConversationLimits): string {
212
+ const n = index + 1;
213
+ if (m.role === "user") return `[${n}] user: ${m.text ?? ""}`;
214
+ if (m.role === "assistant") {
215
+ const parts: string[] = [];
216
+ if (m.text) parts.push(clip(m.text, limits.maxAssistantChars));
217
+ for (const call of m.toolCalls ?? []) parts.push(`tool_call: ${call.name}(${clip(call.args, 120)})`);
218
+ return `[${n}] assistant: ${parts.join(" | ")}`;
219
+ }
220
+ const body = m.isError ? `[error] ${m.text ?? ""}` : (m.text ?? "");
221
+ return `[${n}] tool_result: ${clip(body, limits.maxToolResultChars)}`;
222
+ }
223
+
224
+ /**
225
+ * 结构化渲染整轮对话(spec §11.2),每条消息一行:
226
+ *
227
+ * [1] user: <全文>
228
+ * [2] assistant: <文本> | tool_call: memory({"action":"list"})
229
+ * [3] tool_result: <摘要> // isError 时以 [error] 开头
230
+ *
231
+ * user 文本不截断;assistant 文本按 `maxAssistantChars`;tool_result 按 `maxToolResultChars`。
232
+ * 总长超过 `maxContextTokens * 4` 字符时做**块级中段裁减**:先按单条上限把每条消息渲染成块,
233
+ * 再从中段向外逐块丢弃**非 user** 块(每次丢离中心最近的那块,tie 取靠后的),直到回到预算内,
234
+ * 或只剩 user 块(此时回退到字符串中段裁减)。user 块绝不因丢非 user 块而消失。
235
+ */
236
+ export function renderConversation(messages: ExtractMessage[], limits: ConversationLimits): string {
237
+ const blocks = messages.map((m, i) => renderBlock(m, i, limits));
238
+ const maxChars = limits.maxContextTokens * 4;
239
+ if (maxChars <= 0) return blocks.join("\n");
240
+
241
+ const remaining = blocks.map((block, index) => ({ block, index, user: messages[index].role === "user" }));
242
+ let omitted = 0;
243
+ let firstDropped = -1;
244
+
245
+ // 标记插在首个被丢块的位置(块之间仍以 `\n` 连接,序号沿用原始下标,允许跳号)。
246
+ const assemble = (): string => {
247
+ const lines: string[] = [];
248
+ let markerInserted = false;
249
+ for (const entry of remaining) {
250
+ if (!markerInserted && firstDropped >= 0 && entry.index > firstDropped) {
251
+ lines.push(middleMarkerLabel(omitted));
252
+ markerInserted = true;
253
+ }
254
+ lines.push(entry.block);
255
+ }
256
+ if (firstDropped >= 0 && !markerInserted) lines.push(middleMarkerLabel(omitted));
257
+ return lines.join("\n");
258
+ };
259
+
260
+ let text = assemble();
261
+ while (text.length > maxChars) {
262
+ const center = (remaining.length - 1) / 2;
263
+ let target = -1;
264
+ let best = Number.POSITIVE_INFINITY;
265
+ for (let i = 0; i < remaining.length; i++) {
266
+ if (remaining[i].user) continue;
267
+ const distance = Math.abs(i - center);
268
+ // `<=` + 升序扫描:距离相同时取靠后的那一块。
269
+ if (distance <= best) {
270
+ best = distance;
271
+ target = i;
272
+ }
273
+ }
274
+ // 只剩 user 块:回退到字符串中段裁减(首尾各约一半并预留标记长度)。
275
+ // 把块级已经省略的量交下去,输出里只会留一个标记。
276
+ if (target === -1) return clipMiddle(text, maxChars, omitted);
277
+
278
+ const [dropped] = remaining.splice(target, 1);
279
+ // N 含被丢块的换行(spec §11.2 的逐字格式)。
280
+ omitted += dropped.block.length + 1;
281
+ if (firstDropped === -1) firstDropped = dropped.index;
282
+ text = assemble();
283
+ }
284
+ return text;
285
+ }
286
+
287
+ /** Build the extraction task prompt. */
20
288
  export function buildExtractTask(
21
- memoryDir: string,
22
- messages: Array<{ role: string; content: string }>,
289
+ messages: ExtractMessage[],
23
290
  maxTokens: number,
24
- agentsMdBlocks: string[] = [], // 新增
291
+ agentsMdBlocks: string[],
292
+ opts: { maxToolResultChars: number; maxAssistantChars: number },
25
293
  ): string {
26
- const fromUser = messages.find((m) => m.role === "user");
27
- const fromAssistant = messages.findLast((m) => m.role === "assistant");
28
- const userText = fromUser?.content ?? "";
29
- const assistantText = fromAssistant?.content ?? "";
30
-
31
- const maxChars = maxTokens * 4;
32
- const truncatedUser = userText.slice(0, maxChars / 2);
33
- const truncatedAssistant = assistantText.slice(0, maxChars / 2);
294
+ const conversation = renderConversation(messages, {
295
+ maxToolResultChars: opts.maxToolResultChars,
296
+ maxAssistantChars: opts.maxAssistantChars,
297
+ maxContextTokens: maxTokens,
298
+ });
34
299
 
35
300
  return [
36
- `You are a memory extraction agent. Your working directory is the memory directory at ${memoryDir}.`,
37
- "",
38
- "Analyze the conversation snippet below. If you find valuable learnings, persist them using the memory tools.",
301
+ "You are a memory extraction agent. Below is one full turn of a coding session: every user message, every assistant reply, every tool call and every tool result. Decide what is worth remembering for FUTURE sessions and persist it.",
39
302
  "",
40
303
  "## Tools",
41
- "- ls — list files in the memory directory",
42
- "- read — read MEMORY.md and topic files to check for existing topics",
43
- "- memory_search — full-text search across all memory files for related entries",
44
- "- memory_add — persist a new memory entry to a topic file (creates the topic if new)",
304
+ "Your ONLY tool is `memory`. You have no read, write, edit, ls or bash access.",
305
+ 'memory(action="list") — every stored memory: name, type, modified, description, file.',
306
+ 'memory(action="search", query="…") — full-text search; returns the whole body of each match.',
307
+ 'memory(action="add", name, description, type, content) — create a memory, or overwrite the one whose name matches exactly.',
308
+ 'memory(action="replace", name, description, type, content) — rewrite an existing memory.',
45
309
  "",
46
- "Do NOT use 'bash', 'write', 'edit', or any other tools.",
310
+ "## Storage model",
311
+ "- One memory = one file, and MEMORY.md holds exactly one index line per memory.",
312
+ "- `content` IS the memory: write the fact itself, not a pointer to it.",
313
+ "- `name` must be unique and human-readable. Adding a name that already exists overwrites that memory instead of duplicating it.",
314
+ '- `description` is required in practice: it is the only text a future session sees when deciding relevance, so it must be self-contained (what, where, which value). Bad: "Debugging tips". Good: "staging SSH listens on 2222, not 22".',
315
+ "- `type` is one of user / feedback / project / reference (default feedback).",
47
316
  "",
48
317
  "## Workflow",
49
- "1. Use ls + read to survey existing topic files and MEMORY.md index.",
50
- "2. Use memory_search to check for overlapping or related entries.",
51
- "3. Use memory_add to write new memories — only if the information is novel and valuable.",
318
+ "1. Read the conversation below. Note the corrections the user made, the rules they stated, and the facts that were hard to discover.",
319
+ '2. Call memory(action="list") or memory(action="search") to check whether the fact is already stored.',
320
+ "3. Already stored but wrong or incomplete → replace that exact name. New → add. Already stored correctly → do nothing.",
321
+ "4. Persist at most a handful of memories per turn. When in doubt, skip it.",
52
322
  "",
53
323
  "## What to Remember",
54
- "- Process rules: \"Always do X\" / \"Never do Y\" directives, workflow discipline, reporting standards, self-check habits — treat these as seriously as technical facts",
324
+ '- Process rules: "Always do X" / "Never do Y" directives, workflow discipline, reporting standards, self-check habits — treat these as seriously as technical facts',
55
325
  "- User preferences: coding style, tool choices, naming conventions, workflow habits",
56
326
  "- Project conventions: architecture decisions, file organization, tech stack choices",
57
327
  "- Discoveries: debugging workarounds, gotchas, configuration quirks, undocumented behavior",
@@ -65,40 +335,51 @@ export function buildExtractTask(
65
335
  "- Git history or recent changes",
66
336
  "- Obvious or trivial observations",
67
337
  "",
68
- "## Memory Entry Guidelines",
69
- "- Entry titles must be self-contained and descriptive (only titles appear in future sessions' index)",
70
- '- Choose the appropriate type: user, feedback, project, or reference (default "feedback")',
71
- "- Be concise but complete — one clear point per entry",
72
- "- When in doubt, skip it",
338
+ "## Output",
339
+ 'Persist the memories with the `memory` tool, then reply with one short line saying how many you added or replaced (or "nothing worth saving").',
73
340
  "",
74
- ...(agentsMdBlocks.length > 0
75
- ? ["## AGENTS.md Rules", ...agentsMdBlocks, ""]
76
- : []),
341
+ ...(agentsMdBlocks.length > 0 ? ["## AGENTS.md Rules", ...agentsMdBlocks, ""] : []),
77
342
  "=== Conversation ===",
78
- `User: ${truncatedUser}`,
79
- `Assistant: ${truncatedAssistant}`,
343
+ conversation,
80
344
  ].join("\n");
81
345
  }
82
346
 
83
- /** Fire-and-forget memory extraction. Does not await the headless agent. */
84
- export async function runExtract(opts: RunExtractOpts): Promise<void> {
85
- if (opts.messages.length === 0) return;
86
- const task = buildExtractTask(opts.memoryDir, opts.messages, opts.maxContextTokens, opts.agentsMdBlocks ?? []);
87
- // fire-and-forget: runner disposes internally via finally
88
- runHeadlessAgent({
89
- task,
90
- cwd: opts.memoryDir,
91
- modelRegistry: opts.modelRegistry,
92
- model: opts.model,
93
- parentModel: opts.parentModel,
94
- thinkLevel: opts.thinkLevel,
95
- maxTurns: 5,
96
- timeoutMs: 120_000,
97
-
98
- tools: ["read", "ls"],
99
- customTools: opts.customTools ?? [],
100
- sessionPersistence: opts.sessionPersistence,
101
- }).catch(() => {
102
- /* silently ignore extract errors */
347
+ /**
348
+ * 跑一轮 extract(spec §5.2 / §11)。
349
+ *
350
+ * **不等待逻辑锁**:拿不到就跳过本轮并返回 `{ skipped: true }`。extract 在每次 `agent_end` 都触发,
351
+ * 让它排队等 dream(分钟级)只会把一批批早已过时的对话堆在队列里,等 dream 结束后依次重放。
352
+ *
353
+ * 错误**不吞**:v1 内部的 `.catch(() => {})` 让 extract 的失败对用户完全不可见(spec §14)。
354
+ * 由调用方决定如何呈现(Plan C 会换成限流通知)。
355
+ */
356
+ export async function runExtract(opts: RunExtractOpts): Promise<{ skipped: boolean; result?: string }> {
357
+ const messages = toExtractMessages(opts.messages);
358
+ if (messages.length === 0) return { skipped: true };
359
+
360
+ const task = buildExtractTask(messages, opts.maxContextTokens, opts.agentsMdBlocks ?? [], {
361
+ maxToolResultChars: opts.maxToolResultChars,
362
+ maxAssistantChars: opts.maxAssistantChars,
103
363
  });
364
+
365
+ const result = await opts.store.tryWithLogicalLock(() =>
366
+ runHeadlessAgent({
367
+ task,
368
+ cwd: opts.memoryDir,
369
+ modelRegistry: opts.modelRegistry,
370
+ model: opts.model,
371
+ parentModel: opts.parentModel,
372
+ thinkLevel: opts.thinkLevel,
373
+ maxTurns: 5,
374
+ timeoutMs: 120_000,
375
+ // 没有文件工具:extract 只能通过 memory 原语写(spec §11.2)。
376
+ // 必须用 noTools 而不是 tools: [] —— 后者是白名单,会把 customTools(memory 工具)一起滤掉。
377
+ noTools: "builtin",
378
+ customTools: opts.customTools,
379
+ sessionPersistence: opts.sessionPersistence,
380
+ }),
381
+ );
382
+
383
+ if (result === null) return { skipped: true };
384
+ return { skipped: false, result };
104
385
  }
@@ -0,0 +1,48 @@
1
+ import { createHash } from "node:crypto";
2
+
3
+ /** 文件系统不安全字符与控制字符。 */
4
+ // biome-ignore lint/suspicious/noControlCharactersInRegex: 控制字符是故意列入的(Global Constraints 要求剥离 U+0000–U+001F 与 U+007F)
5
+ const UNSAFE = /[/\\:*?"<>|\u0000-\u001f\u007f]/g;
6
+ /** 首尾空白与句点(句点会让 "." / ".." 变成非法文件名)。 */
7
+ const EDGE = /^[\s.]+|[\s.]+$/g;
8
+ const WHITESPACE_RUN = /\s+/g;
9
+
10
+ const STEM_MAX_BYTES = 100;
11
+
12
+ export const ENTRY_EXT = ".md";
13
+
14
+ function shortHash(input: string): string {
15
+ return createHash("sha256").update(input, "utf8").digest("hex").slice(0, 8);
16
+ }
17
+
18
+ function truncateBytes(input: string, maxBytes: number): string {
19
+ let out = "";
20
+ let bytes = 0;
21
+ for (const ch of input) {
22
+ const size = Buffer.byteLength(ch, "utf8");
23
+ if (bytes + size > maxBytes) break;
24
+ out += ch;
25
+ bytes += size;
26
+ }
27
+ return out;
28
+ }
29
+
30
+ /** 由 entry 的 name 派生确定性的文件名(含扩展名)。同一 name 在同一输入下总得到同一结果。 */
31
+ export function entryFileName(name: string): string {
32
+ const cleaned = name.replace(UNSAFE, "_").replace(EDGE, "");
33
+ const stem = cleaned.length === 0 ? "" : truncateBytes(cleaned.replace(WHITESPACE_RUN, "-"), STEM_MAX_BYTES);
34
+ return `${stem.length === 0 ? `entry-${shortHash(name)}` : stem}${ENTRY_EXT}`;
35
+ }
36
+
37
+ /** 在已占用的文件名集合中为 base 找一个空闲名字(追加 -2、-3……)。 */
38
+ export function resolveUniqueFileName(used: Iterable<string>, base: string): string {
39
+ const taken = used instanceof Set ? used : new Set(used);
40
+ if (!taken.has(base)) return base;
41
+ const hasExt = base.endsWith(ENTRY_EXT);
42
+ const stem = hasExt ? base.slice(0, -ENTRY_EXT.length) : base;
43
+ const ext = hasExt ? ENTRY_EXT : "";
44
+ for (let n = 2; ; n++) {
45
+ const candidate = `${stem}-${n}${ext}`;
46
+ if (!taken.has(candidate)) return candidate;
47
+ }
48
+ }