@flotiarenor/dsh-tool-text-editor 1.1.2 → 1.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +191 -117
- package/README.zh.md +67 -137
- package/lib/core.mjs +341 -375
- package/lib/editor.mjs +131 -154
- package/lib/mask.mjs +440 -0
- package/package.json +19 -13
- package/preset/preset.yml +1 -1
- package/scripts/install-preset.mjs +127 -74
package/lib/core.mjs
CHANGED
|
@@ -3,30 +3,19 @@
|
|
|
3
3
|
/**
|
|
4
4
|
* core.mjs —— 文本编辑核心。纯 Node,只用 `node:` 内置模块,不启动任何子进程。
|
|
5
5
|
*
|
|
6
|
-
* 它承担 dsh 原生 `write
|
|
7
|
-
* * **保留 UTF-8 BOM
|
|
8
|
-
* *
|
|
9
|
-
* *
|
|
10
|
-
*
|
|
11
|
-
* * 锚点可以从目标取(`grep` 正则 / `lines` 行号),不必手抄旧文本;
|
|
12
|
-
* * 匹配失败时给"最接近的候选",歧义时拒绝写盘而不是猜。
|
|
6
|
+
* 它承担 dsh 原生 `write` / `edit` 做不到的事:
|
|
7
|
+
* * **保留 UTF-8 BOM**(原生 `edit` / `write` 都会丢);
|
|
8
|
+
* * **行尾跟随文件**(原生不还原原文件风格:给 LF 写 LF,CRLF 文件就此变成 LF);
|
|
9
|
+
* * **宽松匹配**:精确 → 宽松 → 最接近候选,命中不确定时拒绝写盘而不是猜(原生只做精确匹配);
|
|
10
|
+
* * **锚点可以从目标取**(`grep` 正则 / `lines` 行号),不必手抄旧文本。
|
|
13
11
|
*
|
|
14
|
-
*
|
|
15
|
-
*
|
|
16
|
-
* 截断写造成的空文件;
|
|
17
|
-
* * **同目标串行**:进程内按目标路径排队,并行工具调用不会互相覆盖(跨进程不串行,
|
|
18
|
-
* 那是另一件事,README 里写明了)。
|
|
12
|
+
* 两条落盘保证:**原子写**(同目录临时文件 + fsync + rename,中途被杀不会留下半个文件),
|
|
13
|
+
* **同目标串行**(进程内按目标路径排队;跨进程不串行,见 README)。
|
|
19
14
|
*
|
|
20
|
-
*
|
|
21
|
-
*
|
|
22
|
-
*
|
|
23
|
-
* `diff` / `maxDiffLines` 的由来:工具结果按追加方式进入会话历史,不参与前缀缓存,因此单次调用
|
|
24
|
-
* 回吐的字节随调用次数累积。"新建 / 整体重写" 的 unified diff 每一行都带 `+`,等同于整文件回显
|
|
25
|
-
* ——实测回吐量与输入内容同量级(放大率 ≈ 1.0x)。故模型可见正文默认仅在改动较小时给出
|
|
26
|
-
* (`auto`),其余情况只给统计行,由调用方决定是否另行读取文件;`none` 不给出正文,`full` 始终
|
|
27
|
-
* 给出正文,但同样受行数上限约束。
|
|
15
|
+
* 模型可见文本只有 `brief`:一行统计(`replace@17 +1/-1`)加必要的警告,**既不回显改动内容,也不回显
|
|
16
|
+
* 路径**——工具结果按追加方式进入会话历史,回显会随调用次数累积,而调用方刚发过 `new_text` 与
|
|
17
|
+
* `file_path`。改动本身不落任何旁路记录:本包不写备份也不写台账。
|
|
28
18
|
*/
|
|
29
|
-
|
|
30
19
|
import {
|
|
31
20
|
closeSync,
|
|
32
21
|
existsSync,
|
|
@@ -38,18 +27,13 @@ import {
|
|
|
38
27
|
renameSync,
|
|
39
28
|
statSync,
|
|
40
29
|
unlinkSync,
|
|
41
|
-
writeFileSync,
|
|
42
30
|
writeSync,
|
|
43
31
|
} from 'node:fs'
|
|
44
32
|
import { basename, dirname, isAbsolute, join, relative, resolve, sep } from 'node:path'
|
|
45
33
|
|
|
46
34
|
export const UTF8_BOM_BYTES = Buffer.from([0xef, 0xbb, 0xbf])
|
|
47
|
-
|
|
35
|
+
/** 生成 unified diff 时的默认上下文行数(只喂 `diffStat` 的 `+N/-M`)。 */
|
|
48
36
|
export const DEFAULT_CONTEXT = 3
|
|
49
|
-
/** 模型可见 diff 正文的默认行数预算:`auto` 超了就不给正文,`full` 也按它截断。 */
|
|
50
|
-
export const DEFAULT_MAX_DIFF_LINES = 30
|
|
51
|
-
/** `diff` 的取值:`auto` 小改动才给正文 / `full` 总给(仍封顶)/ `none` 只给统计行。 */
|
|
52
|
-
export const DIFF_MODES = ['auto', 'full', 'none']
|
|
53
37
|
/** LCS 动态规划的格子上限:超过就退化成"整块替换",避免大文件吃光内存。 */
|
|
54
38
|
const MAX_DIFF_CELLS = 4_000_000
|
|
55
39
|
|
|
@@ -67,9 +51,12 @@ export function toLf(text) {
|
|
|
67
51
|
}
|
|
68
52
|
|
|
69
53
|
/**
|
|
70
|
-
* 只看字节,判断 BOM / 行尾风格 /
|
|
54
|
+
* 只看字节,判断 BOM / 行尾风格 / 能否按文本处理。
|
|
55
|
+
*
|
|
56
|
+
* 非法 UTF-8 不在这里判:它只在真正解码时才知道(`decodeText` 用 `fatal: true` 解),所以本函数不返回
|
|
57
|
+
* 一个永远为假的标志位。
|
|
71
58
|
* @param bytes - 文件原始字节。
|
|
72
|
-
* @returns { bom, eol, crlf, lf, mixed, binary
|
|
59
|
+
* @returns { bom, eol, crlf, lf, mixed, binary }
|
|
73
60
|
*/
|
|
74
61
|
export function analyzeBytes(bytes) {
|
|
75
62
|
const bom = bytes.length >= 3 && bytes[0] === 0xef && bytes[1] === 0xbb && bytes[2] === 0xbf
|
|
@@ -87,7 +74,7 @@ export function analyzeBytes(bytes) {
|
|
|
87
74
|
}
|
|
88
75
|
// 判定规则:有 CRLF 且 CRLF >= 独立 LF ⇒ CRLF(跟随多数派,平局算 CRLF)
|
|
89
76
|
const eol = crlf > 0 && crlf >= lf ? '\r\n' : '\n'
|
|
90
|
-
return { bom, eol, crlf, lf, mixed: crlf > 0 && lf > 0, binary
|
|
77
|
+
return { bom, eol, crlf, lf, mixed: crlf > 0 && lf > 0, binary }
|
|
91
78
|
}
|
|
92
79
|
|
|
93
80
|
/**
|
|
@@ -100,14 +87,14 @@ export function analyzeBytes(bytes) {
|
|
|
100
87
|
export function decodeText(bytes, label) {
|
|
101
88
|
const info = analyzeBytes(bytes)
|
|
102
89
|
if (info.binary) {
|
|
103
|
-
throw new UsageError(`${label}:
|
|
90
|
+
throw new UsageError(`${label}: contains NUL bytes, so this looks like a binary file; refusing to edit (these tools handle UTF-8 text only)`)
|
|
104
91
|
}
|
|
105
92
|
const body = info.bom ? bytes.subarray(3) : bytes
|
|
106
93
|
let text
|
|
107
94
|
try {
|
|
108
95
|
text = new TextDecoder('utf-8', { fatal: true }).decode(body)
|
|
109
96
|
} catch (error) {
|
|
110
|
-
throw new UsageError(`${label}:
|
|
97
|
+
throw new UsageError(`${label}: not valid UTF-8 (${error && error.message ? error.message : 'decode error'}); refusing to write so the file is not corrupted`)
|
|
111
98
|
}
|
|
112
99
|
return { text: toLf(text), info }
|
|
113
100
|
}
|
|
@@ -128,12 +115,6 @@ export function splitLines(text) {
|
|
|
128
115
|
return { lines, endsWithNl }
|
|
129
116
|
}
|
|
130
117
|
|
|
131
|
-
/** splitLines 的逆运算。 */
|
|
132
|
-
export function joinLines(lines, endsWithNl) {
|
|
133
|
-
if (lines.length === 0) return ''
|
|
134
|
-
return lines.join('\n') + (endsWithNl ? '\n' : '')
|
|
135
|
-
}
|
|
136
|
-
|
|
137
118
|
/** 逻辑文本 → 字符下标 → 行号(1-based)的查找表。 */
|
|
138
119
|
function lineStarts(text) {
|
|
139
120
|
const starts = [0]
|
|
@@ -153,35 +134,16 @@ function offsetToLine(starts, offset) {
|
|
|
153
134
|
}
|
|
154
135
|
|
|
155
136
|
// ─────────────────────────────────────────────────────────────────────────────
|
|
156
|
-
//
|
|
137
|
+
// 二、路径与显示
|
|
157
138
|
// ─────────────────────────────────────────────────────────────────────────────
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
|
|
165
|
-
|
|
166
|
-
* @param root - 工作区根。
|
|
167
|
-
* @returns 错误说明,`null` 表示放行。
|
|
168
|
-
*/
|
|
169
|
-
export function guardTarget(absPath, root) {
|
|
170
|
-
const norm = normalizeKey(absPath)
|
|
171
|
-
const parts = norm.split('/')
|
|
172
|
-
for (const dir of GUARD_DIRS) {
|
|
173
|
-
if (parts.includes(dir)) {
|
|
174
|
-
return `拒绝写入 ${dir}/ 内部(${absPath})——那是 git / dsh 自己的地盘`
|
|
175
|
-
}
|
|
176
|
-
}
|
|
177
|
-
const rootKey = normalizeKey(root)
|
|
178
|
-
if (rootKey !== '' && norm !== rootKey && !norm.startsWith(rootKey + '/')) {
|
|
179
|
-
return `目标在工作区之外:${absPath}(工作区根=${root})`
|
|
180
|
-
}
|
|
181
|
-
return null
|
|
182
|
-
}
|
|
183
|
-
|
|
184
|
-
/** 工作区内的相对路径(用 / 分隔),用于 diff 头与台账。 */
|
|
139
|
+
//
|
|
140
|
+
// 这里**没有**路径护栏:工作区之外、`.git/`、`.dsh/` 内部都能写。理由是能力取舍——本插件不经
|
|
141
|
+
// `ctx.fs`,宿主沙箱、fs 观察策略与 `sandbox_permissions` 升权拦不到它;而自带的"字符串前缀"护栏既拦不住
|
|
142
|
+
// 符号链接(工作区里的 junction 照样能写到外面),又会在 full-access 会话里禁掉模型本来有权限写的位置。
|
|
143
|
+
// 护栏留给宿主策略与调用方,本包不假装自己是安全边界。唯一保留的是**会话策略**的镜像:
|
|
144
|
+
// `read-only` 会话下两个工具在任何 I/O 之前拒写(见 lib/editor.mjs)。
|
|
145
|
+
|
|
146
|
+
/** 工作区内的相对路径(用 / 分隔),用于 diff 头与台账;工作区之外回落到绝对路径。 */
|
|
185
147
|
export function relativeLabel(root, absPath) {
|
|
186
148
|
const rel = relative(root, absPath)
|
|
187
149
|
if (rel === '') return basename(absPath)
|
|
@@ -195,9 +157,11 @@ export function relativeLabel(root, absPath) {
|
|
|
195
157
|
|
|
196
158
|
/**
|
|
197
159
|
* `263` | `263:270` | `263-270` | 负数从末尾算 → 1-based 闭区间。
|
|
198
|
-
*
|
|
160
|
+
* `count` 给定时按"声明的命中行数"校验:`lines` 锚点是一次点名一段行号,声明数与实际行数不符就拒绝
|
|
161
|
+
* ——与 `grep` / `old_text` 上的 `count` 语义一致(三处都是"声明期望命中数,不符即拒绝")。
|
|
162
|
+
* @throws {UsageError} 格式错、越界,或与声明的 count 不符。
|
|
199
163
|
*/
|
|
200
|
-
export function parseLinespec(spec, text) {
|
|
164
|
+
export function parseLinespec(spec, text, count = undefined) {
|
|
201
165
|
const cleaned = String(spec).trim()
|
|
202
166
|
const range = /^(-?\d+)\s*[:,-]\s*(-?\d+)$/.exec(cleaned)
|
|
203
167
|
const single = /^-?\d+$/.exec(cleaned)
|
|
@@ -210,15 +174,19 @@ export function parseLinespec(spec, text) {
|
|
|
210
174
|
a = Number(single[0])
|
|
211
175
|
b = a
|
|
212
176
|
} else {
|
|
213
|
-
throw new UsageError(`lines
|
|
177
|
+
throw new UsageError(`lines must look like 263 or 263:270, got ${JSON.stringify(spec)}`)
|
|
214
178
|
}
|
|
215
179
|
const total = splitLines(text).lines.length
|
|
216
180
|
if (a < 0) a = total + 1 + a
|
|
217
181
|
if (b < 0) b = total + 1 + b
|
|
218
182
|
if (a < 1 || b < 1 || a > total || b > total) {
|
|
219
|
-
throw new UsageError(`lines ${spec}
|
|
183
|
+
throw new UsageError(`lines ${spec} is out of range (the file has ${total} lines)`)
|
|
184
|
+
}
|
|
185
|
+
const span = a > b ? { start: b, end: a } : { start: a, end: b }
|
|
186
|
+
if (count !== undefined && span.end - span.start + 1 !== count) {
|
|
187
|
+
throw new UsageError(`lines ${spec} covers ${span.end - span.start + 1} lines, which does not match the declared count=${count} — refusing to write`)
|
|
220
188
|
}
|
|
221
|
-
return
|
|
189
|
+
return span
|
|
222
190
|
}
|
|
223
191
|
|
|
224
192
|
/** 取第 startLine..endLine 行的**原文**(含各行的换行符,因此替换/删除不会留下空行)。 */
|
|
@@ -229,15 +197,21 @@ function rawRange(text, starts, startLine, endLine) {
|
|
|
229
197
|
}
|
|
230
198
|
|
|
231
199
|
/**
|
|
232
|
-
*
|
|
233
|
-
* @throws {UsageError}
|
|
200
|
+
* 正则锚点:命中的那一行(多行正则取整块)。`count` 是声明的命中数:不符即拒绝,配对的多处命中逐处替换。
|
|
201
|
+
* @throws {UsageError} 未命中、命中数不符,或命中多处而未声明 count。
|
|
234
202
|
*/
|
|
235
203
|
export function grepSpan(pattern, text, ctx = 0, count = undefined) {
|
|
204
|
+
// 本函数已经按 `gm` 编译,所以调用方再写一个内联 `(?m)` 只会让 `new RegExp` 抛"Invalid group"——
|
|
205
|
+
// 而那正是从别处抄来的正则最常见的写法。剥掉它即可(`m` 已生效;`i` / `s` 一并识别)。
|
|
206
|
+
const source = String(pattern)
|
|
207
|
+
const inline = /^\(\?([ims]+)\)/.exec(source)
|
|
208
|
+
const cleaned = inline === null ? source : source.slice(inline[0].length)
|
|
209
|
+
const flags = `gm${inline !== null && inline[1].includes('i') ? 'i' : ''}${inline !== null && inline[1].includes('s') ? 's' : ''}`
|
|
236
210
|
let re
|
|
237
211
|
try {
|
|
238
|
-
re = new RegExp(
|
|
212
|
+
re = new RegExp(cleaned, flags)
|
|
239
213
|
} catch (error) {
|
|
240
|
-
throw new UsageError(`grep
|
|
214
|
+
throw new UsageError(`grep is not a valid regex: ${error.message}`)
|
|
241
215
|
}
|
|
242
216
|
const starts = lineStarts(text)
|
|
243
217
|
const { lines } = splitLines(text)
|
|
@@ -248,17 +222,21 @@ export function grepSpan(pattern, text, ctx = 0, count = undefined) {
|
|
|
248
222
|
if (match[0].length === 0) re.lastIndex += 1
|
|
249
223
|
if (hits.length > 5000) break
|
|
250
224
|
}
|
|
251
|
-
if (hits.length === 0) throw new UsageError(`grep ${JSON.stringify(pattern)}
|
|
252
|
-
|
|
253
|
-
|
|
254
|
-
|
|
225
|
+
if (hits.length === 0) throw new UsageError(`grep ${JSON.stringify(pattern)} matched nothing in the target file`)
|
|
226
|
+
// `count` 就是"声明期望命中数"(与 `old_text` 一致):不匹配即拒绝,配对的多处命中则逐处替换。
|
|
227
|
+
// 没有这一句,`grep` 上的 `count` 会被**静默忽略**——命中一处却声明五处照样写盘,与契约相反。
|
|
228
|
+
if (count !== undefined ? hits.length !== count : hits.length > 1) {
|
|
229
|
+
const where = hits.slice(0, 8).map((h) => offsetToLine(starts, h.start)).join(', ')
|
|
230
|
+
throw new UsageError(count === undefined
|
|
231
|
+
? `grep ${JSON.stringify(pattern)} matched ${hits.length} times (lines ${where}) — write a tighter pattern, use lines/old_text, or declare the hit count with count`
|
|
232
|
+
: `grep ${JSON.stringify(pattern)} matched ${hits.length} times (lines ${where}), which does not match the declared count=${count} — refusing to write`)
|
|
255
233
|
}
|
|
256
234
|
const spans = hits.map((h) => ({
|
|
257
235
|
start: Math.max(1, offsetToLine(starts, h.start) - ctx),
|
|
258
236
|
end: Math.min(lines.length, offsetToLine(starts, Math.max(h.start, h.end - 1)) + ctx),
|
|
259
237
|
}))
|
|
260
238
|
if (spans.length === 1) {
|
|
261
|
-
return { ...spans[0], raw: rawRange(text, starts, spans[0].start, spans[0].end) }
|
|
239
|
+
return { ...spans[0], blocks: [spans[0]], raw: rawRange(text, starts, spans[0].start, spans[0].end) }
|
|
262
240
|
}
|
|
263
241
|
spans.sort((x, y) => x.start - y.start)
|
|
264
242
|
const merged = []
|
|
@@ -267,17 +245,24 @@ export function grepSpan(pattern, text, ctx = 0, count = undefined) {
|
|
|
267
245
|
if (last && span.start <= last.end) last.end = Math.max(last.end, span.end)
|
|
268
246
|
else merged.push({ ...span })
|
|
269
247
|
}
|
|
248
|
+
// `raw` 是给"整块一次替换"用的合并文本;`blocks` 是**每一次命中自己的行块**,供声明了 count 的多处命中
|
|
249
|
+
// 逐处替换——把非相邻的行块拼成一段文本,那段文本在全文里并不存在,拿它去匹配必然失败。
|
|
270
250
|
const raw = merged.map((span) => rawRange(text, starts, span.start, span.end)).join('')
|
|
271
|
-
return { start: merged[0].start, end: merged[merged.length - 1].end, raw }
|
|
251
|
+
return { start: merged[0].start, end: merged[merged.length - 1].end, blocks: merged, raw }
|
|
272
252
|
}
|
|
273
253
|
|
|
274
|
-
/** `lines`(纯数字/区间)与 `grep
|
|
254
|
+
/** `lines`(纯数字/区间)与 `grep`(正则)统一入口;`count` 是"声明的命中数",两条路径都校验。 */
|
|
275
255
|
export function resolveAnchor(spec, text, ctx = 0, count = undefined) {
|
|
276
256
|
const cleaned = String(spec).trim()
|
|
277
|
-
const numeric = /^-?\d+(
|
|
278
|
-
const span = numeric ? parseLinespec(cleaned, text) : grepSpan(cleaned, text, ctx, count)
|
|
257
|
+
const numeric = /^-?\d+(?:\s*[:,-]\s*-?\d+)?$/.test(cleaned)
|
|
258
|
+
const span = numeric ? parseLinespec(cleaned, text, count) : grepSpan(cleaned, text, ctx, count)
|
|
279
259
|
const starts = lineStarts(text)
|
|
280
|
-
return {
|
|
260
|
+
return {
|
|
261
|
+
start: span.start,
|
|
262
|
+
end: span.end,
|
|
263
|
+
blocks: span.blocks ?? [{ start: span.start, end: span.end }],
|
|
264
|
+
raw: span.raw ?? rawRange(text, starts, span.start, span.end),
|
|
265
|
+
}
|
|
281
266
|
}
|
|
282
267
|
|
|
283
268
|
// ─────────────────────────────────────────────────────────────────────────────
|
|
@@ -324,36 +309,81 @@ function findAllExact(text, old) {
|
|
|
324
309
|
return spans
|
|
325
310
|
}
|
|
326
311
|
|
|
327
|
-
/**
|
|
328
|
-
|
|
312
|
+
/**
|
|
313
|
+
* 宽松命中的 span 是**整行块**(含行尾空白与换行符),而 `old` 可能没有换行结尾——抄自 `read` 的锚点
|
|
314
|
+
* 看不到行尾空白,`old_text: 'gamma'` 在文件里其实是 `'gamma '`。原样用它换掉整块会连带吃掉那些看不见
|
|
315
|
+
* 的字节:行尾空白留下(`GAMMA `),换行符也没了,下一行被并进替换结果——改一行变成删一行,而 `+1/-2`
|
|
316
|
+
* 是模型唯一的线索。
|
|
317
|
+
*
|
|
318
|
+
* 判据只取两条,都是匹配层已知的事实:锚点**没有**以换行结尾(调用方换的是行内容,不是行边界),却命中
|
|
319
|
+
* 了一个尾部含空白的 span。此时收掉 span 的尾部空白,`old` / `new` 的语义就对称了:换行内容,行边界不动,
|
|
320
|
+
* 行数不变。例外是收完就空(`old` 本身全是空白)——那种锚点没有内容可言,不该被"修"成删掉整行。
|
|
321
|
+
*
|
|
322
|
+
* 收的只是**尾部空白**:按行边界寻址的调用方(`lines` / `grep` 锚点,以及抄全整行块的 `old`)不受影响,
|
|
323
|
+
* README 里"`new_text` 也要以换行结尾"的约定对那两条路径照旧。
|
|
324
|
+
*
|
|
325
|
+
* @param text - 逻辑文本(\n 行尾)。
|
|
326
|
+
* @param old - 旧文本;以换行结尾表示调用方在按行边界寻址。
|
|
327
|
+
* @param span - 宽松匹配给出的 `[start, end]` 字符区间。
|
|
328
|
+
* @returns 该收就收过的区间。
|
|
329
|
+
*/
|
|
330
|
+
function clipFuzzySpan(text, old, span) {
|
|
331
|
+
if (old.endsWith('\n')) return span
|
|
332
|
+
let end = span[1]
|
|
333
|
+
// 先收掉行尾空白;span 恰好落在换行符之后时,那个换行符也是这次宽松匹配顺手带上的。
|
|
334
|
+
while (end > span[0] && (text[end - 1] === ' ' || text[end - 1] === '\t')) end -= 1
|
|
335
|
+
if (end === span[0]) return span
|
|
336
|
+
if (text[end - 1] === '\n') end -= 1
|
|
337
|
+
return end > span[0] ? [span[0], end] : span
|
|
338
|
+
}
|
|
339
|
+
|
|
340
|
+
/**
|
|
341
|
+
* 宽松匹配:滑窗比较归一化后的行块,相似度 ≥ 0.9 视为命中。
|
|
342
|
+
*
|
|
343
|
+
* 返回**全部**达标候选(相似度降序,同分按行号升序),不只挑最好的那一个:宽松命中的置信度本就低于精确
|
|
344
|
+
* 命中,多处都能命中时"悄悄挑第一处"等于把决定权交给文件顺序——精确命中多于一处时本模块会拒绝写盘,
|
|
345
|
+
* 宽松匹配没有理由更宽松。命中数由调用方用 `count` 确认。
|
|
346
|
+
*
|
|
347
|
+
* @param text - 逻辑文本。
|
|
348
|
+
* @param old - 旧片段。
|
|
349
|
+
* @param kind - 归一化方式(`trail` / `loose` / `ws`)。
|
|
350
|
+
* @returns `[{ ratio, start, end }]`,无命中时为 `[]`。
|
|
351
|
+
*/
|
|
352
|
+
function relaxedMatches(text, old, kind) {
|
|
329
353
|
const { lines } = splitLines(text)
|
|
330
354
|
const oldLines = splitLines(old).lines
|
|
331
|
-
if (oldLines.length === 0) return
|
|
355
|
+
if (oldLines.length === 0) return []
|
|
332
356
|
const keys = lines.map((line) => normalizeLine(kind, line))
|
|
333
357
|
const oldKeys = oldLines.map((line) => normalizeLine(kind, line))
|
|
334
358
|
const n = oldKeys.length
|
|
335
359
|
const starts = lineStarts(text)
|
|
336
|
-
|
|
360
|
+
const found = []
|
|
337
361
|
for (let i = 0; i + n <= lines.length; i += 1) {
|
|
338
362
|
if (keys[i] !== oldKeys[0] || keys[i + n - 1] !== oldKeys[n - 1]) continue
|
|
339
363
|
const ratio = similarity(keys.slice(i, i + n).join('\n'), oldKeys.join('\n'))
|
|
340
|
-
if (ratio
|
|
364
|
+
if (ratio < 0.9) continue
|
|
365
|
+
found.push({
|
|
366
|
+
ratio,
|
|
367
|
+
index: i,
|
|
368
|
+
start: starts[i],
|
|
369
|
+
end: i + n < starts.length ? starts[i + n] : text.length,
|
|
370
|
+
})
|
|
341
371
|
}
|
|
342
|
-
|
|
343
|
-
|
|
344
|
-
const endOffset = best.index + n < starts.length ? starts[best.index + n] : text.length
|
|
345
|
-
return [startOffset, endOffset]
|
|
372
|
+
found.sort((a, b) => b.ratio - a.ratio || a.start - b.start)
|
|
373
|
+
return found
|
|
346
374
|
}
|
|
347
375
|
|
|
348
376
|
/** 失败时给最接近的几处(行号 + 相似度 + 期望 vs 实际)。 */
|
|
349
377
|
export function nearestCandidates(text, old, limit = 3) {
|
|
350
378
|
const { lines } = splitLines(text)
|
|
351
|
-
|
|
352
|
-
const
|
|
353
|
-
const
|
|
379
|
+
// `old` 为空时 `splitLines` 给不出候选行,退化成拿 `old` 自己当一行(否则下面的窗口宽度会是 0)。
|
|
380
|
+
const oldLines = splitLines(old).lines
|
|
381
|
+
const wanted = oldLines.length > 0 ? oldLines : [old]
|
|
382
|
+
const n = wanted.length
|
|
383
|
+
const firstKey = normalizeLine('loose', wanted[0])
|
|
354
384
|
const scored = []
|
|
355
385
|
for (let i = 0; i + n <= lines.length; i += 1) {
|
|
356
|
-
const ratio = similarity(lines.slice(i, i + n).join('\n'),
|
|
386
|
+
const ratio = similarity(lines.slice(i, i + n).join('\n'), wanted.join('\n'))
|
|
357
387
|
const anchored = normalizeLine('loose', lines[i]) === firstKey ? 0.05 : 0
|
|
358
388
|
scored.push({ index: i, ratio: ratio + anchored })
|
|
359
389
|
}
|
|
@@ -365,60 +395,75 @@ export function nearestCandidates(text, old, limit = 3) {
|
|
|
365
395
|
line: entry.index + 1,
|
|
366
396
|
endLine: Math.min(entry.index + n, lines.length),
|
|
367
397
|
ratio: Math.round(entry.ratio * 1000) / 1000,
|
|
368
|
-
expected:
|
|
398
|
+
expected: wanted.slice(0, 12).join('\n'),
|
|
369
399
|
actual: lines.slice(entry.index, Math.min(entry.index + n, lines.length)).slice(0, 12).join('\n'),
|
|
370
400
|
}))
|
|
371
401
|
}
|
|
372
402
|
|
|
373
403
|
/**
|
|
374
|
-
* 在 text 里定位 old
|
|
404
|
+
* 在 text 里定位 old:精确 → 宽松(`trail` / `loose` / `ws`)→ 失败时给最接近候选。
|
|
405
|
+
*
|
|
406
|
+
* 只有一个消歧旋钮:`expect`(声明的命中数)。没有"只取第 k 次出现"这类入口——那是**选择**而非**确认**,
|
|
407
|
+
* 与"命中多处即拒绝、由调用方把锚点写准"的契约相反(此前的 `nth` 正因绕过它,才让错锚点变成静默的错编辑)。
|
|
408
|
+
*
|
|
375
409
|
* @param text - 逻辑文本。
|
|
376
410
|
* @param old - 旧片段。
|
|
377
|
-
* @param options - `
|
|
411
|
+
* @param options - `expect`(要求恰好 N 处;不符即拒绝)。
|
|
378
412
|
* @returns { ok, spans, mode, note, hits, candidates }
|
|
379
413
|
*/
|
|
380
414
|
export function matchLiteral(text, old, options = {}) {
|
|
381
|
-
const {
|
|
415
|
+
const { expect } = options
|
|
416
|
+
// 空锚点必须**先**判:`findAllExact` 的 `from = index + max(1, 0)` 会在原地打转并撑爆内存,
|
|
417
|
+
// 下面几条宽松路径也要先 `splitLines(old)`。空锚点没有内容可寻址。
|
|
418
|
+
if (old === '') return { ok: false, spans: [], mode: 'miss', hits: [], note: 'old must not be empty', candidates: [] }
|
|
382
419
|
const hits = findAllExact(text, old)
|
|
383
|
-
if (old === '') return { ok: false, spans: [], mode: 'miss', hits, note: 'old 不能为空', candidates: [] }
|
|
384
|
-
if (nth > 0) {
|
|
385
|
-
if (hits.length < nth) {
|
|
386
|
-
return { ok: false, spans: [], mode: 'miss', hits, note: `old 只出现 ${hits.length} 次,取不到第 ${nth} 次`, candidates: nearestCandidates(text, old) }
|
|
387
|
-
}
|
|
388
|
-
return { ok: true, spans: [hits[nth - 1]], mode: nth === 1 ? 'exact' : `exact:nth(${nth})`, hits, note: nth === 1 ? '' : `取第 ${nth} 次出现`, candidates: [] }
|
|
389
|
-
}
|
|
390
420
|
if (expect !== undefined && hits.length === expect) {
|
|
391
|
-
return { ok: true, spans: hits, mode: expect === 1 ? 'exact' : `exact:count(${expect})`, hits, note: expect === 1 ? '' :
|
|
421
|
+
return { ok: true, spans: hits, mode: expect === 1 ? 'exact' : `exact:count(${expect})`, hits, note: expect === 1 ? '' : `${expect} hits, all replaced`, candidates: [] }
|
|
392
422
|
}
|
|
393
423
|
if (hits.length === 1) return { ok: true, spans: hits, mode: 'exact', hits, note: '', candidates: [] }
|
|
394
424
|
if (hits.length > 1) {
|
|
395
425
|
const starts = lineStarts(text)
|
|
396
|
-
const where = hits.slice(0, 10).map(([s]) => offsetToLine(starts, s)).join('
|
|
426
|
+
const where = hits.slice(0, 10).map(([s]) => offsetToLine(starts, s)).join(', ')
|
|
397
427
|
const reason = expect !== undefined
|
|
398
|
-
? `old
|
|
399
|
-
: `old
|
|
428
|
+
? `old occurs ${hits.length} times (lines ${where}), which does not match the expected ${expect} — refusing to write`
|
|
429
|
+
: `old occurs ${hits.length} times (lines ${where}) — declare the hit count with count, or quote a longer old to narrow it down`
|
|
400
430
|
return { ok: false, spans: [], mode: 'ambiguous', hits, note: reason, candidates: [] }
|
|
401
431
|
}
|
|
402
432
|
if (expect !== undefined) {
|
|
403
|
-
return { ok: false, spans: [], mode: 'miss', hits, note: `old
|
|
433
|
+
return { ok: false, spans: [], mode: 'miss', hits, note: `old occurs exactly 0 times, expected ${expect}`, candidates: nearestCandidates(text, old) }
|
|
404
434
|
}
|
|
405
|
-
|
|
406
|
-
|
|
407
|
-
|
|
408
|
-
|
|
409
|
-
|
|
410
|
-
|
|
435
|
+
for (const [kind, label] of [['trail', 'ignoring trailing whitespace'], ['loose', 'ignoring leading/trailing whitespace'], ['ws', 'ignoring all whitespace differences']]) {
|
|
436
|
+
const found = relaxedMatches(text, old, kind)
|
|
437
|
+
if (found.length === 0) continue
|
|
438
|
+
if (found.length > 1) {
|
|
439
|
+
// 宽松命中多于一处:与精确命中同样拒绝写盘,绝不按文件顺序悄悄挑一处。
|
|
440
|
+
const starts = lineStarts(text)
|
|
441
|
+
const where = found.slice(0, 8).map((entry) => offsetToLine(starts, entry.start)).join(', ')
|
|
411
442
|
return {
|
|
412
|
-
ok:
|
|
413
|
-
spans: [
|
|
414
|
-
mode:
|
|
415
|
-
hits,
|
|
416
|
-
note:
|
|
417
|
-
|
|
443
|
+
ok: false,
|
|
444
|
+
spans: [],
|
|
445
|
+
mode: 'ambiguous',
|
|
446
|
+
hits: found.map((entry) => [entry.start, entry.end]),
|
|
447
|
+
note: `old does not match exactly; the relaxed mode (${label}) hit ${found.length} places (lines ${where}) —`
|
|
448
|
+
+ ' quote a longer old to narrow it down, or use a lines / grep anchor (a relaxed hit is never a choice: there is no way to know it picked the same place)',
|
|
449
|
+
candidates: nearestCandidates(text, old),
|
|
418
450
|
}
|
|
419
451
|
}
|
|
452
|
+
// 锚点没写换行结尾、宽松命中却落在整行块上时,把行尾空白留给文件(见 clipFuzzySpan)。
|
|
453
|
+
const best = found[0]
|
|
454
|
+
const span = clipFuzzySpan(text, old, [best.start, best.end])
|
|
455
|
+
const trimmed = span[1] !== best.end
|
|
456
|
+
return {
|
|
457
|
+
ok: true,
|
|
458
|
+
spans: [span],
|
|
459
|
+
mode: kind,
|
|
460
|
+
hits,
|
|
461
|
+
note: `exact match failed; the relaxed mode (${label}) matched instead — check that the change landed where you meant`
|
|
462
|
+
+ (trimmed ? '; the anchor had no trailing newline, so trailing whitespace and the newline stayed in place (otherwise the next line would have been joined)' : ''),
|
|
463
|
+
candidates: [],
|
|
464
|
+
}
|
|
420
465
|
}
|
|
421
|
-
return { ok: false, spans: [], mode: 'miss', hits, note: 'old
|
|
466
|
+
return { ok: false, spans: [], mode: 'miss', hits, note: 'old does not exist in the target file (both exact and relaxed matching failed)', candidates: nearestCandidates(text, old) }
|
|
422
467
|
}
|
|
423
468
|
|
|
424
469
|
// ─────────────────────────────────────────────────────────────────────────────
|
|
@@ -491,8 +536,28 @@ function diffOpsFor(a, b) {
|
|
|
491
536
|
}
|
|
492
537
|
|
|
493
538
|
/**
|
|
494
|
-
*
|
|
495
|
-
*
|
|
539
|
+
* 把"改动 op 的下标"按上下文并成 hunk(相邻改动之间不超过 2×context+1 就并进同一块),供
|
|
540
|
+
* `unifiedDiff` 切片用。
|
|
541
|
+
*
|
|
542
|
+
* @param changeIndexes - `ops` 中非 `=` 的下标(升序)。
|
|
543
|
+
* @param context - 每个 hunk 前后保留的上下文行数。
|
|
544
|
+
* @returns 每个 hunk 的 op 下标区间(闭区间)。
|
|
545
|
+
*/
|
|
546
|
+
function changeRanges(changeIndexes, context) {
|
|
547
|
+
const blocks = []
|
|
548
|
+
for (const index of changeIndexes) {
|
|
549
|
+
const last = blocks[blocks.length - 1]
|
|
550
|
+
if (last && index - last[last.length - 1] <= context * 2 + 1) last.push(index)
|
|
551
|
+
else blocks.push([index])
|
|
552
|
+
}
|
|
553
|
+
return blocks
|
|
554
|
+
}
|
|
555
|
+
|
|
556
|
+
/**
|
|
557
|
+
* unified diff 文本;`label` 只用于 `a/` `b/` 头。与 GNU diff 一样标出"末尾没有换行"的那一侧。
|
|
558
|
+
*
|
|
559
|
+
* 输出不进入模型上下文,也不落盘:它是 `diffStat` 计算 `+N/-M` 的依据(行级 LCS 保证计数与真实改动一致)。
|
|
560
|
+
* 保留它而不换成粗略的行数比较,是为了让统计行在多 hunk、重复行等情形下仍然准确。
|
|
496
561
|
*/
|
|
497
562
|
export function unifiedDiff(label, before, after, context = DEFAULT_CONTEXT) {
|
|
498
563
|
if (before === after) return ''
|
|
@@ -519,12 +584,7 @@ export function unifiedDiff(label, before, after, context = DEFAULT_CONTEXT) {
|
|
|
519
584
|
return out.join('\n') + '\n'
|
|
520
585
|
}
|
|
521
586
|
|
|
522
|
-
const blocks =
|
|
523
|
-
for (const index of changeIndexes) {
|
|
524
|
-
const last = blocks[blocks.length - 1]
|
|
525
|
-
if (last && index - last[last.length - 1] <= context * 2 + 1) last.push(index)
|
|
526
|
-
else blocks.push([index])
|
|
527
|
-
}
|
|
587
|
+
const blocks = changeRanges(changeIndexes, context)
|
|
528
588
|
|
|
529
589
|
const out = [`--- a/${label}`, `+++ b/${label}`]
|
|
530
590
|
for (const block of blocks) {
|
|
@@ -576,130 +636,60 @@ export function diffStat(before, after) {
|
|
|
576
636
|
return { added, removed }
|
|
577
637
|
}
|
|
578
638
|
|
|
639
|
+
// ─────────────────────────────────────────────────────────────────────────────
|
|
640
|
+
// 六、原子写、同目标串行
|
|
641
|
+
// ─────────────────────────────────────────────────────────────────────────────
|
|
642
|
+
|
|
579
643
|
/**
|
|
580
|
-
*
|
|
644
|
+
* 系统调用失败的**模型可读**表述。
|
|
581
645
|
*
|
|
582
|
-
*
|
|
583
|
-
*
|
|
584
|
-
*
|
|
585
|
-
* @param text - 已生成的 unified diff 文本(调用方传 0 上下文行的那份,只保留改动行)。
|
|
586
|
-
* @param mode - `auto` | `full` | `none`(见 `DIFF_MODES`)。
|
|
587
|
-
* @param maxLines - 正文行数上限,默认 `DEFAULT_MAX_DIFF_LINES`。
|
|
588
|
-
* @returns 模型可见的正文;裁剪时附一行说明(省略或截断),可能为空字符串。
|
|
646
|
+
* Node 原始 `error.message` 同时携带内部临时文件名(`.<名字>.<pid><ts>.tmp`)与绝对路径,二者都不该进入
|
|
647
|
+
* 模型上下文,而且它不提供可操作信息。这里只保留 errno 与一句原因。
|
|
589
648
|
*/
|
|
590
|
-
|
|
591
|
-
|
|
592
|
-
|
|
593
|
-
|
|
594
|
-
|
|
595
|
-
|
|
596
|
-
|
|
597
|
-
|
|
598
|
-
|
|
599
|
-
|
|
600
|
-
|
|
649
|
+
const FS_ERROR_REASONS = {
|
|
650
|
+
EACCES: 'no write permission',
|
|
651
|
+
EBUSY: 'the file is in use by another process',
|
|
652
|
+
EISDIR: 'the target is a directory',
|
|
653
|
+
EMFILE: 'out of file handles',
|
|
654
|
+
ENAMETOOLONG: 'the path is too long',
|
|
655
|
+
ENOENT: 'no such directory',
|
|
656
|
+
ENOSPC: 'no space left on the device',
|
|
657
|
+
ENOTDIR: 'a segment of the path is not a directory',
|
|
658
|
+
EPERM: 'no write permission',
|
|
659
|
+
EROFS: 'the file system is read-only',
|
|
601
660
|
}
|
|
602
661
|
|
|
603
|
-
/**
|
|
604
|
-
|
|
605
|
-
|
|
606
|
-
|
|
607
|
-
|
|
608
|
-
|
|
609
|
-
|
|
610
|
-
|
|
611
|
-
|
|
612
|
-
|
|
613
|
-
|
|
614
|
-
let from = 0
|
|
615
|
-
while (from < lines.length && (lines[from].startsWith('---') || lines[from].startsWith('+++'))) {
|
|
616
|
-
from += 1
|
|
617
|
-
}
|
|
618
|
-
return lines.slice(from).join('\n')
|
|
662
|
+
/** 兜底:把任何仍带内部临时文件名的文本抹掉(未知 errno 时才用得到)。 */
|
|
663
|
+
function scrubTempNames(text) {
|
|
664
|
+
return String(text ?? '').replace(/\S*\.tmp\b/g, '<temp file>')
|
|
665
|
+
}
|
|
666
|
+
|
|
667
|
+
/** 一次落盘相关的系统调用失败 → 一句完整、可操作、不含内部细节的失败原因。 */
|
|
668
|
+
function ioFailure(what, label, error) {
|
|
669
|
+
const code = error !== null && typeof error === 'object' && typeof error.code === 'string' ? error.code : ''
|
|
670
|
+
if (code === '') return `${label}: could not ${what} (${scrubTempNames(error && error.message ? error.message : error)})`
|
|
671
|
+
const reason = FS_ERROR_REASONS[code]
|
|
672
|
+
return `${label}: could not ${what} (${code}${reason === undefined ? '' : ': ' + reason})`
|
|
619
673
|
}
|
|
620
674
|
|
|
621
675
|
/**
|
|
622
|
-
*
|
|
623
|
-
* 与原生 `edit` / `write` 的 `presentationMeta` 同形(纯新增的 hunk 用 `oldText: null`)。
|
|
676
|
+
* 新建文件时补齐缺失的父目录(`write_text` 的"新建无需任何开关"包括目录)。
|
|
624
677
|
*
|
|
625
|
-
*
|
|
626
|
-
*
|
|
678
|
+
* 父目录存在但不是目录(路径中间夹着一个文件)时抛 `ENOTDIR`:不做这项检查时 Windows 报 `ENOENT`
|
|
679
|
+
* ("目录不存在"),而该位置实际存在一份同名文件,错误码对调用方具有误导性。
|
|
627
680
|
*
|
|
628
|
-
* @
|
|
629
|
-
* @param path - 卡片上标注的路径。
|
|
630
|
-
* @returns 每个 hunk 一个条目;没有可解析的 hunk 时返回空数组。
|
|
681
|
+
* @returns 真正创建过目录时返回该目录的工作区相对路径,否则 `null`。
|
|
631
682
|
*/
|
|
632
|
-
|
|
633
|
-
const
|
|
634
|
-
|
|
635
|
-
|
|
636
|
-
|
|
637
|
-
current = { old: [], next: [] }
|
|
638
|
-
hunks.push(current)
|
|
639
|
-
continue
|
|
640
|
-
}
|
|
641
|
-
if (current === null || line.startsWith('\\')) continue
|
|
642
|
-
if (line.startsWith('-')) current.old.push(line.slice(1))
|
|
643
|
-
else if (line.startsWith('+')) current.next.push(line.slice(1))
|
|
644
|
-
else if (line.startsWith(' ')) {
|
|
645
|
-
current.old.push(line.slice(1))
|
|
646
|
-
current.next.push(line.slice(1))
|
|
683
|
+
function ensureParentDir(absPath, root) {
|
|
684
|
+
const dir = dirname(absPath)
|
|
685
|
+
if (existsSync(dir)) {
|
|
686
|
+
if (!statSync(dir).isDirectory()) {
|
|
687
|
+
throw Object.assign(new Error('parent exists but is not a directory'), { code: 'ENOTDIR' })
|
|
647
688
|
}
|
|
689
|
+
return null
|
|
648
690
|
}
|
|
649
|
-
return hunks.map((hunk) => ({
|
|
650
|
-
path,
|
|
651
|
-
oldText: hunk.old.length > 0 ? hunk.old.join('\n') : null,
|
|
652
|
-
newText: hunk.next.join('\n'),
|
|
653
|
-
}))
|
|
654
|
-
}
|
|
655
|
-
|
|
656
|
-
// ─────────────────────────────────────────────────────────────────────────────
|
|
657
|
-
// 六、备份、台账、原子写、同目标串行
|
|
658
|
-
// ─────────────────────────────────────────────────────────────────────────────
|
|
659
|
-
|
|
660
|
-
function two(n, width = 2) {
|
|
661
|
-
return String(n).padStart(width, '0')
|
|
662
|
-
}
|
|
663
|
-
|
|
664
|
-
function stamp(date) {
|
|
665
|
-
return `${date.getFullYear()}${two(date.getMonth() + 1)}${two(date.getDate())}-`
|
|
666
|
-
+ `${two(date.getHours())}${two(date.getMinutes())}${two(date.getSeconds())}-`
|
|
667
|
-
+ `${two(date.getMilliseconds(), 3)}`
|
|
668
|
-
}
|
|
669
|
-
|
|
670
|
-
/** 备份文件名:<绝对路径扁平化>@<时间戳>。 */
|
|
671
|
-
export function backupFileNameFor(absPath, date = new Date()) {
|
|
672
|
-
const flat = resolve(absPath).replace(/:/g, '').replace(/[\\/]/g, '_')
|
|
673
|
-
return `${flat}@${stamp(date)}`
|
|
674
|
-
}
|
|
675
|
-
|
|
676
|
-
/** 落盘前的原件备份。 */
|
|
677
|
-
export function makeBackup(artifactsDir, absPath, bytes, date = new Date()) {
|
|
678
|
-
const dir = join(artifactsDir, 'backups')
|
|
679
691
|
mkdirSync(dir, { recursive: true })
|
|
680
|
-
|
|
681
|
-
writeFileSync(join(dir, name), bytes)
|
|
682
|
-
return name
|
|
683
|
-
}
|
|
684
|
-
|
|
685
|
-
/** 追加一条 JSONL 台账:每行一个对象,字段见 record。 */
|
|
686
|
-
export function appendLedger(artifactsDir, entry) {
|
|
687
|
-
const file = join(artifactsDir, 'edits.log')
|
|
688
|
-
mkdirSync(artifactsDir, { recursive: true })
|
|
689
|
-
const time = new Date()
|
|
690
|
-
const record = {
|
|
691
|
-
time: `${time.getFullYear()}-${two(time.getMonth() + 1)}-${two(time.getDate())} `
|
|
692
|
-
+ `${two(time.getHours())}:${two(time.getMinutes())}:${two(time.getSeconds())}`,
|
|
693
|
-
...entry,
|
|
694
|
-
}
|
|
695
|
-
let id = 1
|
|
696
|
-
if (existsSync(file)) {
|
|
697
|
-
const raw = readFileSync(file, 'utf8')
|
|
698
|
-
id = raw.split('\n').filter((line) => line.trim() !== '').length + 1
|
|
699
|
-
}
|
|
700
|
-
record.id = id
|
|
701
|
-
writeFileSync(file, JSON.stringify(record) + '\n', { encoding: 'utf8', flag: 'a' })
|
|
702
|
-
return record
|
|
692
|
+
return relativeLabel(root, dir)
|
|
703
693
|
}
|
|
704
694
|
|
|
705
695
|
/**
|
|
@@ -751,7 +741,8 @@ export function inferNewline(absPath) {
|
|
|
751
741
|
if (configured === 'lf') return '\n'
|
|
752
742
|
if (configured === 'crlf' || configured === 'cr') return '\r\n'
|
|
753
743
|
const dir = dirname(absPath)
|
|
754
|
-
const
|
|
744
|
+
const dot = basename(absPath).lastIndexOf('.')
|
|
745
|
+
const ext = dot > 0 ? basename(absPath).slice(dot) : ''
|
|
755
746
|
let sameExtCrlf = 0
|
|
756
747
|
let sameExtLf = 0
|
|
757
748
|
let otherCrlf = 0
|
|
@@ -783,7 +774,11 @@ export function inferNewline(absPath) {
|
|
|
783
774
|
const lf = countLf(sample) - crlf
|
|
784
775
|
if (crlf === 0 && lf === 0) continue
|
|
785
776
|
const isCrlf = crlf > 0 && crlf >= lf
|
|
786
|
-
|
|
777
|
+
// 目标没有扩展名时(`ext === ''`)同扩展名那一档退化为"同样没有扩展名的兄弟文件"。
|
|
778
|
+
if (ext !== '' && name.endsWith(ext)) {
|
|
779
|
+
if (isCrlf) sameExtCrlf += 1
|
|
780
|
+
else sameExtLf += 1
|
|
781
|
+
} else if (ext === '' && !name.includes('.')) {
|
|
787
782
|
if (isCrlf) sameExtCrlf += 1
|
|
788
783
|
else sameExtLf += 1
|
|
789
784
|
} else if (isCrlf) otherCrlf += 1
|
|
@@ -811,67 +806,58 @@ function countLf(buffer) {
|
|
|
811
806
|
// ─────────────────────────────────────────────────────────────────────────────
|
|
812
807
|
|
|
813
808
|
function fail(path, message) {
|
|
814
|
-
return { path, ok: false,
|
|
809
|
+
return { path, ok: false, brief: '', stderr: message }
|
|
815
810
|
}
|
|
816
811
|
|
|
817
812
|
function hintText(match) {
|
|
818
813
|
if (match.mode === 'ambiguous') {
|
|
819
|
-
|
|
820
|
-
return ` 候选:old 命中 ${match.hits.length} 处。用 nth 指定第几次,或 count 声明命中数。`
|
|
814
|
+
return ` Candidates: old occurs ${match.hits.length} times. Declare the hit count with count (all of them are replaced), or quote a longer old to make it unique.`
|
|
821
815
|
}
|
|
822
816
|
if (!match.candidates || match.candidates.length === 0) {
|
|
823
|
-
return '
|
|
817
|
+
return ' No near candidates: check that this is the right file, or take the anchor straight from the target with grep/lines (no need to copy it by hand).'
|
|
824
818
|
}
|
|
825
|
-
const lines = ['
|
|
819
|
+
const lines = [' Closest candidates (lines | similarity):']
|
|
826
820
|
for (const candidate of match.candidates) {
|
|
827
|
-
lines.push(` -
|
|
828
|
-
lines.push(`
|
|
829
|
-
lines.push(`
|
|
821
|
+
lines.push(` - lines ${candidate.line}-${candidate.endLine} similarity ${candidate.ratio.toFixed(2)}`)
|
|
822
|
+
lines.push(` in the file: ${JSON.stringify(candidate.actual.slice(0, 200))}`)
|
|
823
|
+
lines.push(` your old: ${JSON.stringify(candidate.expected.slice(0, 200))}`)
|
|
830
824
|
}
|
|
831
825
|
return lines.join('\n')
|
|
832
826
|
}
|
|
833
827
|
|
|
834
828
|
/**
|
|
835
|
-
*
|
|
829
|
+
* 执行一次编辑 / 写入。后端无关:调用方只需给出归一后的 plan。
|
|
836
830
|
*
|
|
837
|
-
* plan(edit):`{ kind:'edit', filePath, mode, newText, oldText, anchor:{value}|null, count
|
|
838
|
-
* plan(write):`{ kind:'write', filePath, content
|
|
831
|
+
* plan(edit):`{ kind:'edit', filePath, mode, newText, oldText, anchor:{value}|null, count }`;
|
|
832
|
+
* plan(write):`{ kind:'write', filePath, content }`。
|
|
839
833
|
*
|
|
840
|
-
* @param context - `{ root,
|
|
841
|
-
* @returns 规范结果 `{ path, ok,
|
|
842
|
-
* `
|
|
843
|
-
* `brief` / `diff` = **模型可见**的那一份(统计行与受预算约束的正文)。
|
|
834
|
+
* @param context - `{ root, newFileBom }`
|
|
835
|
+
* @returns 规范结果 `{ path, ok, brief, stderr }`:`brief` 是模型可见的全部内容(一行统计加警告),
|
|
836
|
+
* `stderr` 是失败原因(失败时非空,同样模型可见)。
|
|
844
837
|
*/
|
|
845
838
|
export async function applyPlan(plan, context) {
|
|
846
|
-
const {
|
|
847
|
-
root,
|
|
848
|
-
artifactsDir = join(root, '.dsh'),
|
|
849
|
-
backup = true,
|
|
850
|
-
log = true,
|
|
851
|
-
tool = 'edit_text',
|
|
852
|
-
context: diffContext = DEFAULT_CONTEXT,
|
|
853
|
-
newFileBom = false,
|
|
854
|
-
diff: diffMode = 'auto',
|
|
855
|
-
maxDiffLines = DEFAULT_MAX_DIFF_LINES,
|
|
856
|
-
} = context
|
|
839
|
+
const { root, newFileBom = false } = context
|
|
857
840
|
const absPath = isAbsolute(plan.filePath) ? resolve(plan.filePath) : resolve(root, plan.filePath)
|
|
858
841
|
const label = relativeLabel(root, absPath)
|
|
859
842
|
|
|
860
|
-
const guarded = guardTarget(absPath, root)
|
|
861
|
-
if (guarded) return fail(plan.filePath, guarded)
|
|
862
|
-
|
|
863
843
|
return await withTargetLock(absPath, async () => {
|
|
864
844
|
const exists = existsSync(absPath)
|
|
865
845
|
const kind = plan.kind === 'write' ? 'write' : 'edit'
|
|
866
846
|
if (kind === 'edit' && !exists) {
|
|
867
|
-
return fail(plan.filePath,
|
|
847
|
+
return fail(plan.filePath, `target does not exist: ${plan.filePath} (use write_text to create it)`)
|
|
868
848
|
}
|
|
869
849
|
|
|
870
850
|
let original = ''
|
|
871
851
|
let info
|
|
872
852
|
let bytesBefore = Buffer.alloc(0)
|
|
873
853
|
if (exists) {
|
|
874
|
-
|
|
854
|
+
// 目标可能是目录(EISDIR)或不可读(EACCES):这类失败只回一句 errno 说法,
|
|
855
|
+
// 不把 Node 的原始 message(含绝对路径)塞进模型上下文。
|
|
856
|
+
try {
|
|
857
|
+
bytesBefore = readFileSync(absPath)
|
|
858
|
+
} catch (error) {
|
|
859
|
+
return fail(plan.filePath, ioFailure('read the target', label, error))
|
|
860
|
+
}
|
|
875
861
|
try {
|
|
876
862
|
const decoded = decodeText(bytesBefore, label)
|
|
877
863
|
original = decoded.text
|
|
@@ -886,7 +872,7 @@ export async function applyPlan(plan, context) {
|
|
|
886
872
|
|
|
887
873
|
const warnings = []
|
|
888
874
|
if (info.mixed) {
|
|
889
|
-
warnings.push(`[warn]
|
|
875
|
+
warnings.push(`[warn] the target mixes line endings; writing it back as ${info.eol === '\r\n' ? 'CRLF' : 'LF'}`)
|
|
890
876
|
}
|
|
891
877
|
|
|
892
878
|
let output
|
|
@@ -895,74 +881,96 @@ export async function applyPlan(plan, context) {
|
|
|
895
881
|
if (kind === 'write') {
|
|
896
882
|
output = toLf(plan.content ?? '')
|
|
897
883
|
if (output === original && exists) {
|
|
898
|
-
return fail(plan.filePath, '
|
|
884
|
+
return fail(plan.filePath, 'no change (the new content is identical to the current content)')
|
|
899
885
|
}
|
|
900
886
|
if (!output.endsWith('\n') && output !== '') {
|
|
901
|
-
warnings.push('[warn]
|
|
887
|
+
warnings.push('[warn] the new content does not end with a newline, so the file will not either')
|
|
902
888
|
}
|
|
903
|
-
applied.push(['write', 1, Math.max(1, splitLines(output).lines.length), 'write',
|
|
889
|
+
applied.push(['write', 1, Math.max(1, splitLines(output).lines.length), 'write', ''])
|
|
904
890
|
} else {
|
|
905
|
-
const usage = []
|
|
906
|
-
const { lines } = splitLines(original)
|
|
907
891
|
if (plan.mode === 'append' || plan.mode === 'prepend') {
|
|
908
892
|
output = plan.mode === 'append' ? original : ''
|
|
909
893
|
if (plan.mode === 'append') {
|
|
910
894
|
let add = plan.newText
|
|
911
895
|
if (output !== '' && !output.endsWith('\n')) {
|
|
912
896
|
add = '\n' + add
|
|
913
|
-
warnings.push('[warn]
|
|
897
|
+
warnings.push('[warn] the target did not end with a newline; one was added before appending')
|
|
914
898
|
}
|
|
915
|
-
if (add !== '' && !add.endsWith('\n')) warnings.push('[warn]
|
|
899
|
+
if (add !== '' && !add.endsWith('\n')) warnings.push('[warn] the appended content does not end with a newline, so the file will not either')
|
|
916
900
|
output += add
|
|
917
901
|
} else {
|
|
918
902
|
let add = plan.newText
|
|
919
903
|
if (original !== '' && add !== '' && !add.endsWith('\n')) add += '\n'
|
|
920
904
|
output += add + original
|
|
921
905
|
}
|
|
922
|
-
applied.push([plan.mode, 1, 1, plan.mode,
|
|
906
|
+
applied.push([plan.mode, 1, 1, plan.mode, ''])
|
|
923
907
|
} else {
|
|
924
908
|
const plans = []
|
|
925
909
|
if (plan.mode === 'replace') {
|
|
926
|
-
let target = plan.oldText
|
|
927
910
|
let match
|
|
911
|
+
let anchorSpan
|
|
912
|
+
// 行首下标表:多处命中要按行块换算成字符区间,单处命中要换算成行号。
|
|
913
|
+
const charStarts = lineStarts(original)
|
|
928
914
|
if (plan.anchor) {
|
|
929
|
-
let anchorSpan
|
|
930
915
|
try {
|
|
931
|
-
|
|
916
|
+
// `count` 必须传进来:`grep` 命中多处时报错原文就写着"用 count 声明命中数"。
|
|
917
|
+
anchorSpan = resolveAnchor(plan.anchor.value, original, 0, plan.count)
|
|
932
918
|
} catch (error) {
|
|
933
|
-
|
|
934
|
-
anchorSpan = null
|
|
919
|
+
return fail(plan.filePath, error instanceof UsageError ? error.message : String(error))
|
|
935
920
|
}
|
|
936
|
-
|
|
937
|
-
|
|
938
|
-
const
|
|
939
|
-
match =
|
|
940
|
-
|
|
921
|
+
// 声明的多处命中逐处替换:`count` 已在上面校验过命中数,每一次命中都有自己的行块,拿它替换
|
|
922
|
+
// 即可(把多个非相邻行块拼成一段文本去匹配只会得到"old 不存在")。
|
|
923
|
+
const blocks = anchorSpan.blocks ?? []
|
|
924
|
+
match = blocks.length > 1
|
|
925
|
+
? {
|
|
926
|
+
ok: true,
|
|
927
|
+
spans: blocks.map((block) => [
|
|
928
|
+
charStarts[block.start - 1],
|
|
929
|
+
block.end < charStarts.length ? charStarts[block.end] : original.length,
|
|
930
|
+
]),
|
|
931
|
+
mode: `exact:count(${plan.count})`,
|
|
932
|
+
hits: [],
|
|
933
|
+
note: '',
|
|
934
|
+
candidates: [],
|
|
935
|
+
}
|
|
936
|
+
: matchLiteral(original, anchorSpan.raw, {})
|
|
937
|
+
if (match.ok) match.mode = match.mode.startsWith('exact:count(') ? match.mode : `anchor:${match.mode}`
|
|
941
938
|
} else {
|
|
942
|
-
match = matchLiteral(original,
|
|
943
|
-
nth: plan.nth ?? 0,
|
|
944
|
-
strict: plan.strict === true,
|
|
945
|
-
expect: plan.count,
|
|
946
|
-
})
|
|
939
|
+
match = matchLiteral(original, plan.oldText ?? '', { expect: plan.count })
|
|
947
940
|
}
|
|
948
941
|
if (!match.ok) {
|
|
949
942
|
return fail(plan.filePath, `${label}:${match.note}\n${hintText(match)}`)
|
|
950
943
|
}
|
|
951
|
-
|
|
952
|
-
|
|
953
|
-
|
|
944
|
+
// 锚点路径的 mode 带 `anchor:` 前缀;去掉它之后,"是不是精确命中"才可判。
|
|
945
|
+
const hitMode = String(match.mode).replace(/^anchor:/, '')
|
|
946
|
+
// 不拿 `match.hits.length` 比 `plan.count`:`count` 的校验各有其主——`grep` / `lines` 由
|
|
947
|
+
// `resolveAnchor` 在命中处校验,`old_text` 由 `matchLiteral` 的 `expect` 校验。锚点路径的
|
|
948
|
+
// hits 是"锚点文本在全文出现几次",逐处替换时天然为 1,拿它比 `count` 只会把正当的多处替换
|
|
949
|
+
// 误判成失败。
|
|
954
950
|
for (const [start, end] of match.spans) {
|
|
955
|
-
const startLine = offsetToLine(
|
|
956
|
-
const endLine = end > start ? offsetToLine(
|
|
957
|
-
plans.push({ start, end, mode:
|
|
951
|
+
const startLine = offsetToLine(charStarts, start)
|
|
952
|
+
const endLine = end > start ? offsetToLine(charStarts, end - 1) : startLine
|
|
953
|
+
plans.push({ start, end, mode: hitMode, startLine, endLine })
|
|
958
954
|
}
|
|
959
|
-
|
|
955
|
+
// 只通报宽松命中:`exact` 与 `exact:count(N)` 带的是"命中 N 处,全部替换"这类说明,不是警告。
|
|
956
|
+
if (hitMode !== 'exact' && !hitMode.startsWith('exact:count(')) {
|
|
960
957
|
warnings.push(`[warn] ${match.note}`)
|
|
961
958
|
}
|
|
959
|
+
// 锚点带着行尾换行符、替换文本却没有:被换掉的那一行会和下一行并成一行。只对 `old_text` 说
|
|
960
|
+
// 这一句——`lines`/`grep` 是调用方点名要的整行行块,README 已把"`new_text` 也要以换行结尾"
|
|
961
|
+
// 写成前置约定,每次调用都重复提示只是噪声;而抄来的 `old_text` 带着换行、`new_text` 忘了带
|
|
962
|
+
// 换行时,`+1/-2` 说明不了"少了一行"。
|
|
963
|
+
if (plan.anchor === null && match.spans.length === 1 && match.spans[0][1] > match.spans[0][0]
|
|
964
|
+
&& original.startsWith('\n', match.spans[0][1] - 1)
|
|
965
|
+
&& plan.newText !== '' && !plan.newText.endsWith('\n')) {
|
|
966
|
+
warnings.push('[warn] the anchor ends with a newline but the replacement does not — the replaced line has been joined with the next one;'
|
|
967
|
+
+ ' to keep the line boundary, end new_text with a newline too')
|
|
968
|
+
}
|
|
962
969
|
} else {
|
|
963
970
|
let anchorSpan
|
|
964
971
|
try {
|
|
965
|
-
|
|
972
|
+
// `count` 同样要传:声明了命中行数就与 `replace` 档一样校验,不能悄悄忽略。
|
|
973
|
+
anchorSpan = resolveAnchor(plan.anchor.value, original, 0, plan.count)
|
|
966
974
|
} catch (error) {
|
|
967
975
|
return fail(plan.filePath, error instanceof UsageError ? error.message : String(error))
|
|
968
976
|
}
|
|
@@ -984,92 +992,50 @@ export async function applyPlan(plan, context) {
|
|
|
984
992
|
const previous = sorted[i - 1]
|
|
985
993
|
const current = sorted[i]
|
|
986
994
|
if (current.start < previous.end) {
|
|
987
|
-
return fail(plan.filePath,
|
|
995
|
+
return fail(plan.filePath, `two edit ranges overlap (lines ${previous.startLine} and ${current.startLine}) — split this into two calls`)
|
|
988
996
|
}
|
|
989
997
|
}
|
|
990
998
|
|
|
991
|
-
|
|
992
|
-
|
|
993
|
-
|
|
994
|
-
|
|
999
|
+
// 按 `start` 升序把原文的"间隙"与替换文本拼起来。逐段推进游标,不做任何坐标换算——先前的
|
|
1000
|
+
// "从后往前替换 + 位移修正"等价于这里的一次遍历,却要在每处替换后重算下标,容易写错。
|
|
1001
|
+
const ordered = [...plans].sort((a, b) => a.start - b.start)
|
|
1002
|
+
const parts = []
|
|
1003
|
+
let cursor = 0
|
|
1004
|
+
for (const item of ordered) {
|
|
1005
|
+
parts.push(original.slice(cursor, item.start), plan.newText)
|
|
1006
|
+
cursor = item.end
|
|
1007
|
+
applied.push(['replace', item.startLine, item.endLine, item.mode, ''])
|
|
995
1008
|
}
|
|
996
|
-
|
|
1009
|
+
parts.push(original.slice(cursor))
|
|
1010
|
+
output = parts.join('')
|
|
997
1011
|
}
|
|
998
1012
|
}
|
|
999
1013
|
|
|
1000
1014
|
if (output === original && exists) {
|
|
1001
|
-
return fail(plan.filePath, '
|
|
1002
|
-
}
|
|
1003
|
-
if (!exists && output === '') {
|
|
1004
|
-
return fail(plan.filePath, '新建内容为空 —— 没有产生任何变化')
|
|
1015
|
+
return fail(plan.filePath, 'no change (old and new are identical, or the content already matches)')
|
|
1005
1016
|
}
|
|
1006
1017
|
|
|
1007
|
-
const diff = unifiedDiff(label, original, output, diffContext)
|
|
1008
1018
|
const stat = diffStat(original, output)
|
|
1009
1019
|
const created = !exists
|
|
1010
|
-
const head = plan.dryRun
|
|
1011
|
-
? `=== ${label}${created ? '(新建)' : ''} | DRY RUN(未落盘)===`
|
|
1012
|
-
: `=== ${label}${created ? '(新建)' : ''} | 已写入 ===`
|
|
1013
1020
|
const kinds = [...new Set(applied.map(([k, s, e]) => (k === 'replace' ? `replace@${s}${e !== s ? `-${e}` : ''}` : k)))]
|
|
1014
|
-
.join('
|
|
1015
|
-
|
|
1016
|
-
// 模型可见部分:警告 + 一行统计(+A/-B)+ 经上限裁剪的正文。
|
|
1017
|
-
// 正文使用 0 上下文行:调用方刚完成这次编辑,周边内容或已读取过,或可自行读取。
|
|
1018
|
-
// 完整 diff(含上下文行)只出现在 stdout 与 UI 卡片中,不进入这条路径。
|
|
1019
|
-
const brief = [...warnings, `${kinds} +${stat.added}/-${stat.removed}`].join('\n')
|
|
1020
|
-
const visibleDiff = diffBudget(withoutDiffHeaders(unifiedDiff(label, original, output, 0)), diffMode, maxDiffLines)
|
|
1021
|
+
.join(', ')
|
|
1021
1022
|
|
|
1022
|
-
if (
|
|
1023
|
-
|
|
1024
|
-
|
|
1025
|
-
|
|
1026
|
-
|
|
1027
|
-
dryRun: true,
|
|
1028
|
-
stdout: [...warnings, head, diff.trimEnd(), `DRY RUN ${label}:${kinds}(+${stat.added}/-${stat.removed})`].filter((s) => s !== '').join('\n') + '\n',
|
|
1029
|
-
stderr: '',
|
|
1030
|
-
brief,
|
|
1031
|
-
diff: visibleDiff,
|
|
1023
|
+
if (created) {
|
|
1024
|
+
try {
|
|
1025
|
+
ensureParentDir(absPath, root)
|
|
1026
|
+
} catch (error) {
|
|
1027
|
+
return fail(plan.filePath, ioFailure('create the directory', label, error))
|
|
1032
1028
|
}
|
|
1033
1029
|
}
|
|
1034
1030
|
|
|
1035
|
-
let backupName = null
|
|
1036
|
-
if (exists && backup) {
|
|
1037
|
-
backupName = makeBackup(artifactsDir, absPath, bytesBefore)
|
|
1038
|
-
}
|
|
1039
1031
|
try {
|
|
1040
1032
|
writeFileAtomic(absPath, encodeText(output, info))
|
|
1041
1033
|
} catch (error) {
|
|
1042
|
-
return fail(plan.filePath,
|
|
1043
|
-
}
|
|
1044
|
-
|
|
1045
|
-
if (log) {
|
|
1046
|
-
const first = applied[0] ?? ['?', 0, 0, '', '']
|
|
1047
|
-
appendLedger(artifactsDir, {
|
|
1048
|
-
tool,
|
|
1049
|
-
file: label,
|
|
1050
|
-
abspath: absPath,
|
|
1051
|
-
action: created ? 'create' : 'write',
|
|
1052
|
-
kinds: applied.map(([k]) => k),
|
|
1053
|
-
line_start: first[1],
|
|
1054
|
-
line_end: first[2],
|
|
1055
|
-
added: stat.added,
|
|
1056
|
-
removed: stat.removed,
|
|
1057
|
-
bom: info.bom,
|
|
1058
|
-
eol: info.eol === '\r\n' ? 'CRLF' : 'LF',
|
|
1059
|
-
backup: backupName,
|
|
1060
|
-
summary: `${kinds}${plan.note ? `;${plan.note}` : ''}`,
|
|
1061
|
-
})
|
|
1034
|
+
return fail(plan.filePath, ioFailure('write the target', label, error))
|
|
1062
1035
|
}
|
|
1063
1036
|
|
|
1064
|
-
|
|
1065
|
-
|
|
1066
|
-
|
|
1067
|
-
wrote: true,
|
|
1068
|
-
dryRun: false,
|
|
1069
|
-
stdout: [...warnings, head, diff.trimEnd(), `OK ${label}:${kinds}(+${stat.added}/-${stat.removed})${backupName ? `;备份 ${backupName}` : ''}`].filter((s) => s !== '').join('\n') + '\n',
|
|
1070
|
-
stderr: '',
|
|
1071
|
-
brief,
|
|
1072
|
-
diff: visibleDiff,
|
|
1073
|
-
}
|
|
1037
|
+
// 模型可见文本:警告行 + 一行统计(如 `replace@17 +1/-1`)。不回显改动内容(理由见文件头)。
|
|
1038
|
+
const brief = [...warnings, `${kinds} +${stat.added}/-${stat.removed}`].join('\n')
|
|
1039
|
+
return { path: plan.filePath, ok: true, brief, stderr: '' }
|
|
1074
1040
|
})
|
|
1075
1041
|
}
|