jev-agent-tools 0.1.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +66 -0
- package/LICENSE +21 -0
- package/README.md +119 -0
- package/docs/design.md +187 -0
- package/docs/tools/jev_ask.md +92 -0
- package/docs/tools/jev_ask_files.md +80 -0
- package/docs/tools/jev_check_diff.md +77 -0
- package/docs/tools/jev_find_files.md +59 -0
- package/docs/tools/jev_locate_in_file.md +51 -0
- package/docs/tools/jev_select_tests.md +59 -0
- package/package.json +85 -0
- package/rules/jev-ask.md +4 -0
- package/src/adapters/analysis-context.ts +100 -0
- package/src/adapters/ask-files.ts +236 -0
- package/src/adapters/ask-proof.ts +202 -0
- package/src/adapters/ask-syntax.ts +463 -0
- package/src/adapters/command.ts +240 -0
- package/src/adapters/docs.ts +222 -0
- package/src/adapters/files.ts +411 -0
- package/src/adapters/find.ts +151 -0
- package/src/adapters/git-base.ts +32 -0
- package/src/adapters/git-inventory.ts +94 -0
- package/src/adapters/git.ts +525 -0
- package/src/adapters/locate-file.ts +197 -0
- package/src/adapters/output-lines.ts +50 -0
- package/src/adapters/risk-callers.ts +525 -0
- package/src/adapters/runner-version.ts +102 -0
- package/src/adapters/syntax.ts +229 -0
- package/src/adapters/test-inventory.ts +168 -0
- package/src/adapters/usage.ts +21 -0
- package/src/adapters/utf8.ts +57 -0
- package/src/constants.ts +109 -0
- package/src/core/ask-closure.ts +419 -0
- package/src/core/ask-proof.ts +32 -0
- package/src/core/ask-references.ts +249 -0
- package/src/core/asks.ts +616 -0
- package/src/core/batches.ts +83 -0
- package/src/core/command-output.ts +249 -0
- package/src/core/diff.ts +226 -0
- package/src/core/docs.ts +399 -0
- package/src/core/find.ts +157 -0
- package/src/core/git.ts +5 -0
- package/src/core/import-boundaries.ts +102 -0
- package/src/core/imports.ts +691 -0
- package/src/core/integrity.ts +64 -0
- package/src/core/lexical.ts +130 -0
- package/src/core/locate.ts +213 -0
- package/src/core/output.ts +264 -0
- package/src/core/pointer.ts +51 -0
- package/src/core/risk-callers.ts +1270 -0
- package/src/core/runner-version.ts +66 -0
- package/src/core/sections.ts +269 -0
- package/src/core/state.ts +53 -0
- package/src/core/syntax.ts +8 -0
- package/src/core/test-commands.ts +430 -0
- package/src/core/test-coverage.ts +103 -0
- package/src/core/test-discovery.ts +1695 -0
- package/src/core/test-evidence.ts +649 -0
- package/src/core/test-state.ts +99 -0
- package/src/core/truncate.ts +14 -0
- package/src/core/units.ts +531 -0
- package/src/describe.ts +26 -0
- package/src/guide.ts +42 -0
- package/src/host.ts +22 -0
- package/src/index.ts +40 -0
- package/src/jev/client.ts +505 -0
- package/src/jev/pool.ts +60 -0
- package/src/jev/types.ts +60 -0
- package/src/presets/docs.ts +85 -0
- package/src/presets/risk.ts +263 -0
- package/src/presets/spec.ts +111 -0
- package/src/presets/witnesses.ts +313 -0
- package/src/render.ts +45 -0
- package/src/result.ts +4 -0
- package/src/run-end.ts +157 -0
- package/src/runtime.ts +16 -0
- package/src/session.ts +106 -0
- package/src/texts/ask-files.ts +2 -0
- package/src/texts/ask.ts +4 -0
- package/src/texts/check-diff.ts +24 -0
- package/src/texts/configuration.ts +2 -0
- package/src/texts/find.ts +14 -0
- package/src/texts/guide.ts +16 -0
- package/src/texts/locate.ts +10 -0
- package/src/texts/run-end.ts +13 -0
- package/src/texts/select-tests.ts +3 -0
- package/src/tools/ask-files.ts +263 -0
- package/src/tools/ask-schema.ts +116 -0
- package/src/tools/ask.ts +925 -0
- package/src/tools/check-diff.ts +510 -0
- package/src/tools/docs-check.ts +399 -0
- package/src/tools/find.ts +529 -0
- package/src/tools/locate.ts +369 -0
- package/src/tools/select-tests.ts +746 -0
- package/src/tools/spec-check.ts +210 -0
|
@@ -0,0 +1,525 @@
|
|
|
1
|
+
import { type FileHandle, lstat, readlink } from "node:fs/promises";
|
|
2
|
+
import { resolve } from "node:path";
|
|
3
|
+
import {
|
|
4
|
+
GIT_BLOB_BATCH_SIZE,
|
|
5
|
+
STATE_MAX_CHARS,
|
|
6
|
+
TIMEOUT_MS,
|
|
7
|
+
} from "../constants.ts";
|
|
8
|
+
import { type DiffFile, parseDiff } from "../core/diff.ts";
|
|
9
|
+
import type { GitExec } from "../core/git.ts";
|
|
10
|
+
import {
|
|
11
|
+
buildUnits,
|
|
12
|
+
type EvidenceUnit,
|
|
13
|
+
type SourceFile,
|
|
14
|
+
type SyntaxParser,
|
|
15
|
+
type UnitLimit,
|
|
16
|
+
} from "../core/units.ts";
|
|
17
|
+
import type { Result } from "../result.ts";
|
|
18
|
+
import { admitGitPath, admitRepoMetadata, openRepoFile } from "./files.ts";
|
|
19
|
+
import { shareGitInventory } from "./git-inventory.ts";
|
|
20
|
+
import { loadSyntaxParser } from "./syntax.ts";
|
|
21
|
+
import { decodeUtf8, verifyGitUtf8 } from "./utf8.ts";
|
|
22
|
+
|
|
23
|
+
export type CollectionResult<T> =
|
|
24
|
+
| Result<T>
|
|
25
|
+
| { ok: false; kind: "inconsistent_diff"; error: string };
|
|
26
|
+
export interface DiffOptions {
|
|
27
|
+
cwd: string;
|
|
28
|
+
base: string;
|
|
29
|
+
paths?: string[];
|
|
30
|
+
signal?: AbortSignal;
|
|
31
|
+
}
|
|
32
|
+
const config = [
|
|
33
|
+
"-c",
|
|
34
|
+
"color.ui=false",
|
|
35
|
+
"-c",
|
|
36
|
+
"core.quotePath=true",
|
|
37
|
+
"-c",
|
|
38
|
+
"diff.suppressBlankEmpty=false",
|
|
39
|
+
"-c",
|
|
40
|
+
"diff.relative=false",
|
|
41
|
+
"-c",
|
|
42
|
+
"diff.noprefix=false",
|
|
43
|
+
"-c",
|
|
44
|
+
"diff.mnemonicPrefix=false",
|
|
45
|
+
"-c",
|
|
46
|
+
"diff.algorithm=myers",
|
|
47
|
+
"-c",
|
|
48
|
+
"diff.indentHeuristic=false",
|
|
49
|
+
];
|
|
50
|
+
async function git(
|
|
51
|
+
exec: GitExec,
|
|
52
|
+
options: DiffOptions,
|
|
53
|
+
args: string[],
|
|
54
|
+
): Promise<Result<{ text: string }>> {
|
|
55
|
+
try {
|
|
56
|
+
options.signal?.throwIfAborted();
|
|
57
|
+
const result = await exec("git", [...config, ...args], {
|
|
58
|
+
cwd: options.cwd,
|
|
59
|
+
timeout: TIMEOUT_MS,
|
|
60
|
+
signal: options.signal,
|
|
61
|
+
});
|
|
62
|
+
if (result.killed)
|
|
63
|
+
return {
|
|
64
|
+
ok: false,
|
|
65
|
+
error: "Git interrupted (cancelled or timed out).",
|
|
66
|
+
};
|
|
67
|
+
if (result.code !== 0)
|
|
68
|
+
return { ok: false, error: `Git failed: ${result.stderr}` };
|
|
69
|
+
return { ok: true, text: result.stdout };
|
|
70
|
+
} catch (error) {
|
|
71
|
+
return { ok: false, error: `Unable to execute git: ${String(error)}` };
|
|
72
|
+
}
|
|
73
|
+
}
|
|
74
|
+
type ReadSource = {
|
|
75
|
+
text: string;
|
|
76
|
+
binary: boolean;
|
|
77
|
+
limitation?: "too_large" | "unreadable" | "not UTF-8 text";
|
|
78
|
+
};
|
|
79
|
+
async function readWorktree(
|
|
80
|
+
options: DiffOptions,
|
|
81
|
+
exec: GitExec,
|
|
82
|
+
path: string,
|
|
83
|
+
inventory: ReadonlySet<string>,
|
|
84
|
+
): Promise<ReadSource> {
|
|
85
|
+
let handle: FileHandle | undefined;
|
|
86
|
+
try {
|
|
87
|
+
options.signal?.throwIfAborted();
|
|
88
|
+
const location = await admitRepoMetadata(
|
|
89
|
+
options.cwd,
|
|
90
|
+
path,
|
|
91
|
+
exec,
|
|
92
|
+
options.signal,
|
|
93
|
+
inventory,
|
|
94
|
+
);
|
|
95
|
+
if (!location.ok)
|
|
96
|
+
return { text: "", binary: false, limitation: "unreadable" };
|
|
97
|
+
const stat = await lstat(location.abs);
|
|
98
|
+
if (stat.isSymbolicLink())
|
|
99
|
+
return { text: await readlink(location.abs), binary: false };
|
|
100
|
+
if (stat.size > STATE_MAX_CHARS)
|
|
101
|
+
return { text: "", binary: false, limitation: "too_large" };
|
|
102
|
+
const opened = await openRepoFile(options.cwd, path, {
|
|
103
|
+
exec,
|
|
104
|
+
signal: options.signal,
|
|
105
|
+
inventory,
|
|
106
|
+
});
|
|
107
|
+
if (!opened.ok)
|
|
108
|
+
return { text: "", binary: false, limitation: "unreadable" };
|
|
109
|
+
handle = opened.handle;
|
|
110
|
+
const buffer = Buffer.alloc(STATE_MAX_CHARS + 1);
|
|
111
|
+
let offset = 0;
|
|
112
|
+
while (offset < buffer.length) {
|
|
113
|
+
const { bytesRead } = await handle.read(
|
|
114
|
+
buffer,
|
|
115
|
+
offset,
|
|
116
|
+
buffer.length - offset,
|
|
117
|
+
null,
|
|
118
|
+
);
|
|
119
|
+
if (!bytesRead) break;
|
|
120
|
+
offset += bytesRead;
|
|
121
|
+
}
|
|
122
|
+
if (offset > STATE_MAX_CHARS)
|
|
123
|
+
return { text: "", binary: false, limitation: "too_large" };
|
|
124
|
+
const bytes = buffer.subarray(0, offset);
|
|
125
|
+
const decoded = decodeUtf8(bytes);
|
|
126
|
+
return decoded.ok
|
|
127
|
+
? { text: decoded.text, binary: bytes.includes(0) }
|
|
128
|
+
: { text: "", binary: false, limitation: "not UTF-8 text" };
|
|
129
|
+
} catch {
|
|
130
|
+
return { text: "", binary: false, limitation: "unreadable" };
|
|
131
|
+
} finally {
|
|
132
|
+
await handle?.close();
|
|
133
|
+
}
|
|
134
|
+
}
|
|
135
|
+
async function readGitBlobs(
|
|
136
|
+
exec: GitExec,
|
|
137
|
+
options: DiffOptions,
|
|
138
|
+
objects: readonly { id: string; size: number }[],
|
|
139
|
+
): Promise<
|
|
140
|
+
Result<{ blobs: ReadonlyMap<string, string>; invalid: ReadonlySet<string> }>
|
|
141
|
+
> {
|
|
142
|
+
const blobs = new Map<string, string>();
|
|
143
|
+
const invalid = new Set<string>();
|
|
144
|
+
for (let offset = 0; offset < objects.length; offset += GIT_BLOB_BATCH_SIZE) {
|
|
145
|
+
const batch = objects.slice(offset, offset + GIT_BLOB_BATCH_SIZE);
|
|
146
|
+
const result = await git(exec, options, [
|
|
147
|
+
"show",
|
|
148
|
+
"--no-ext-diff",
|
|
149
|
+
"--no-textconv",
|
|
150
|
+
"--end-of-options",
|
|
151
|
+
...batch.map((object) => object.id),
|
|
152
|
+
]);
|
|
153
|
+
if (!result.ok) return result;
|
|
154
|
+
const bytes = Buffer.from(result.text, "utf8");
|
|
155
|
+
if (bytes.length !== batch.reduce((sum, object) => sum + object.size, 0)) {
|
|
156
|
+
// Host decoding can replace invalid UTF-8: never split subsequent blobs at shifted offsets.
|
|
157
|
+
for (const object of batch) {
|
|
158
|
+
const single = await git(exec, options, [
|
|
159
|
+
"show",
|
|
160
|
+
"--no-ext-diff",
|
|
161
|
+
"--no-textconv",
|
|
162
|
+
"--end-of-options",
|
|
163
|
+
object.id,
|
|
164
|
+
]);
|
|
165
|
+
if (!single.ok) return single;
|
|
166
|
+
const decoded = verifyGitUtf8(single.text, object.id, object.size);
|
|
167
|
+
if (decoded.ok) blobs.set(object.id, decoded.text);
|
|
168
|
+
else invalid.add(object.id);
|
|
169
|
+
}
|
|
170
|
+
continue;
|
|
171
|
+
}
|
|
172
|
+
let position = 0;
|
|
173
|
+
for (const object of batch) {
|
|
174
|
+
const decoded = verifyGitUtf8(
|
|
175
|
+
bytes.subarray(position, position + object.size).toString("utf8"),
|
|
176
|
+
object.id,
|
|
177
|
+
object.size,
|
|
178
|
+
);
|
|
179
|
+
if (decoded.ok) blobs.set(object.id, decoded.text);
|
|
180
|
+
else invalid.add(object.id);
|
|
181
|
+
position += object.size;
|
|
182
|
+
}
|
|
183
|
+
}
|
|
184
|
+
return { ok: true, blobs, invalid };
|
|
185
|
+
}
|
|
186
|
+
export async function collectDiff(
|
|
187
|
+
exec: GitExec,
|
|
188
|
+
options: DiffOptions,
|
|
189
|
+
): Promise<CollectionResult<{ files: SourceFile[] }>> {
|
|
190
|
+
exec = shareGitInventory(exec);
|
|
191
|
+
const root = await git(exec, options, ["rev-parse", "--show-toplevel"]);
|
|
192
|
+
if (!root.ok) return root;
|
|
193
|
+
const cwd = root.text.replace(/\n$/, "");
|
|
194
|
+
const scoped = { ...options, cwd };
|
|
195
|
+
const location = await git(exec, options, ["rev-parse", "--show-prefix"]);
|
|
196
|
+
if (!location.ok) return location;
|
|
197
|
+
const repositoryPrefix = location.text.replace(/\n$/, "");
|
|
198
|
+
const paths = [
|
|
199
|
+
"--",
|
|
200
|
+
...(options.paths ?? []).map(
|
|
201
|
+
(path) => `:(top,literal)${repositoryPrefix}${path}`,
|
|
202
|
+
),
|
|
203
|
+
];
|
|
204
|
+
const rawResult = await git(exec, scoped, [
|
|
205
|
+
"diff",
|
|
206
|
+
"--raw",
|
|
207
|
+
"-z",
|
|
208
|
+
"--no-ext-diff",
|
|
209
|
+
"--no-textconv",
|
|
210
|
+
"--ignore-submodules=none",
|
|
211
|
+
"--end-of-options",
|
|
212
|
+
options.base,
|
|
213
|
+
...paths,
|
|
214
|
+
]);
|
|
215
|
+
if (!rawResult.ok) return rawResult;
|
|
216
|
+
const untrackedResult = await git(exec, scoped, [
|
|
217
|
+
"ls-files",
|
|
218
|
+
"--full-name",
|
|
219
|
+
"--others",
|
|
220
|
+
"--exclude-standard",
|
|
221
|
+
"-z",
|
|
222
|
+
...paths,
|
|
223
|
+
]);
|
|
224
|
+
if (!untrackedResult.ok) return untrackedResult;
|
|
225
|
+
let candidates: DiffFile[];
|
|
226
|
+
try {
|
|
227
|
+
candidates = parseDiff({
|
|
228
|
+
raw: rawResult.text,
|
|
229
|
+
numstat: "",
|
|
230
|
+
patch: "",
|
|
231
|
+
untracked: untrackedResult.text.split("\0").filter(Boolean),
|
|
232
|
+
});
|
|
233
|
+
} catch {
|
|
234
|
+
return {
|
|
235
|
+
ok: false,
|
|
236
|
+
error: "Inconsistent Git snapshots; collect again.",
|
|
237
|
+
kind: "inconsistent_diff",
|
|
238
|
+
};
|
|
239
|
+
}
|
|
240
|
+
const unsupported: string[] = [];
|
|
241
|
+
await Promise.all(
|
|
242
|
+
candidates.map(async (file) => {
|
|
243
|
+
try {
|
|
244
|
+
const stat = await lstat(resolve(cwd, file.path));
|
|
245
|
+
if (
|
|
246
|
+
!stat.isFile() &&
|
|
247
|
+
!stat.isSymbolicLink() &&
|
|
248
|
+
file.oldMode !== "160000" &&
|
|
249
|
+
file.newMode !== "160000"
|
|
250
|
+
)
|
|
251
|
+
unsupported.push(file.path);
|
|
252
|
+
} catch {
|
|
253
|
+
/* Missing tracked paths remain deletions. */
|
|
254
|
+
}
|
|
255
|
+
}),
|
|
256
|
+
);
|
|
257
|
+
for (const path of unsupported) paths.push(`:(top,exclude,literal)${path}`);
|
|
258
|
+
const prefix = [
|
|
259
|
+
"diff",
|
|
260
|
+
"--no-color",
|
|
261
|
+
"--no-ext-diff",
|
|
262
|
+
"--no-textconv",
|
|
263
|
+
"--find-renames",
|
|
264
|
+
"--src-prefix=a/",
|
|
265
|
+
"--dst-prefix=b/",
|
|
266
|
+
"--no-relative",
|
|
267
|
+
"--line-prefix=",
|
|
268
|
+
"--submodule=short",
|
|
269
|
+
"--ignore-submodules=none",
|
|
270
|
+
];
|
|
271
|
+
const results = await Promise.all([
|
|
272
|
+
git(exec, scoped, [
|
|
273
|
+
...prefix,
|
|
274
|
+
"--raw",
|
|
275
|
+
"-z",
|
|
276
|
+
"--end-of-options",
|
|
277
|
+
options.base,
|
|
278
|
+
...paths,
|
|
279
|
+
]),
|
|
280
|
+
git(exec, scoped, [
|
|
281
|
+
...prefix,
|
|
282
|
+
"--numstat",
|
|
283
|
+
"-z",
|
|
284
|
+
"--end-of-options",
|
|
285
|
+
options.base,
|
|
286
|
+
...paths,
|
|
287
|
+
]),
|
|
288
|
+
git(exec, scoped, [
|
|
289
|
+
...prefix,
|
|
290
|
+
"--patch",
|
|
291
|
+
"--unified=8",
|
|
292
|
+
"--end-of-options",
|
|
293
|
+
options.base,
|
|
294
|
+
...paths,
|
|
295
|
+
]),
|
|
296
|
+
git(exec, scoped, [
|
|
297
|
+
"ls-files",
|
|
298
|
+
"--full-name",
|
|
299
|
+
"--others",
|
|
300
|
+
"--exclude-standard",
|
|
301
|
+
"-z",
|
|
302
|
+
...paths,
|
|
303
|
+
]),
|
|
304
|
+
git(exec, scoped, [
|
|
305
|
+
"ls-files",
|
|
306
|
+
"--full-name",
|
|
307
|
+
"--cached",
|
|
308
|
+
"--others",
|
|
309
|
+
"--exclude-standard",
|
|
310
|
+
"-z",
|
|
311
|
+
]),
|
|
312
|
+
git(exec, scoped, [
|
|
313
|
+
"ls-tree",
|
|
314
|
+
"-r",
|
|
315
|
+
"--long",
|
|
316
|
+
"-z",
|
|
317
|
+
"--full-tree",
|
|
318
|
+
"--end-of-options",
|
|
319
|
+
options.base,
|
|
320
|
+
]),
|
|
321
|
+
]);
|
|
322
|
+
for (const result of results) if (!result.ok) return result;
|
|
323
|
+
const [raw, numstat, patch, untracked, current, base] = results;
|
|
324
|
+
if (
|
|
325
|
+
!raw.ok ||
|
|
326
|
+
!numstat.ok ||
|
|
327
|
+
!patch.ok ||
|
|
328
|
+
!untracked.ok ||
|
|
329
|
+
!current.ok ||
|
|
330
|
+
!base.ok
|
|
331
|
+
)
|
|
332
|
+
return { ok: false, error: "Incomplete Git collection." };
|
|
333
|
+
const currentInventory = new Set(current.text.split("\0").filter(Boolean));
|
|
334
|
+
const baseInventory = new Map<string, string>();
|
|
335
|
+
for (const entry of base.text.split("\0")) {
|
|
336
|
+
const separator = entry.indexOf("\t");
|
|
337
|
+
if (separator !== -1) baseInventory.set(entry.slice(separator + 1), entry);
|
|
338
|
+
}
|
|
339
|
+
let files: DiffFile[];
|
|
340
|
+
try {
|
|
341
|
+
files = parseDiff({
|
|
342
|
+
raw: raw.text,
|
|
343
|
+
numstat: numstat.text,
|
|
344
|
+
patch: patch.text,
|
|
345
|
+
verifyPaths: true,
|
|
346
|
+
untracked: untracked.text.split("\0").filter(Boolean),
|
|
347
|
+
});
|
|
348
|
+
} catch {
|
|
349
|
+
return {
|
|
350
|
+
ok: false,
|
|
351
|
+
error: "Inconsistent Git snapshots; collect again.",
|
|
352
|
+
kind: "inconsistent_diff",
|
|
353
|
+
};
|
|
354
|
+
}
|
|
355
|
+
files.push(...candidates.filter((file) => unsupported.includes(file.path)));
|
|
356
|
+
const objects = new Map<string, { id: string; size: number }>();
|
|
357
|
+
for (const file of files) {
|
|
358
|
+
if (
|
|
359
|
+
/^[A?]/.test(file.status) ||
|
|
360
|
+
file.binary ||
|
|
361
|
+
unsupported.includes(file.path)
|
|
362
|
+
)
|
|
363
|
+
continue;
|
|
364
|
+
const admission = await admitGitPath(
|
|
365
|
+
cwd,
|
|
366
|
+
file.oldPath,
|
|
367
|
+
options.base,
|
|
368
|
+
exec,
|
|
369
|
+
options.signal,
|
|
370
|
+
baseInventory,
|
|
371
|
+
);
|
|
372
|
+
if (!admission.ok) continue;
|
|
373
|
+
const entry = baseInventory.get(file.oldPath);
|
|
374
|
+
const match =
|
|
375
|
+
entry &&
|
|
376
|
+
/^(?:100644|100755|120000) blob ([0-9a-f]+)\s+(\d+)\t/.exec(entry);
|
|
377
|
+
if (match)
|
|
378
|
+
objects.set(file.oldPath, { id: match[1] ?? "", size: Number(match[2]) });
|
|
379
|
+
}
|
|
380
|
+
const uniqueObjects = new Map(
|
|
381
|
+
[...objects.values()]
|
|
382
|
+
.filter((object) => object.size <= STATE_MAX_CHARS)
|
|
383
|
+
.map((object) => [object.id, object]),
|
|
384
|
+
);
|
|
385
|
+
const loaded = await readGitBlobs(exec, scoped, [...uniqueObjects.values()]);
|
|
386
|
+
if (!loaded.ok) return loaded;
|
|
387
|
+
const collectFile = async (
|
|
388
|
+
file: DiffFile,
|
|
389
|
+
): Promise<CollectionResult<{ file: SourceFile }>> => {
|
|
390
|
+
const historical = !/^[A?]/.test(file.status);
|
|
391
|
+
const admission = historical
|
|
392
|
+
? await admitGitPath(
|
|
393
|
+
cwd,
|
|
394
|
+
file.oldPath,
|
|
395
|
+
options.base,
|
|
396
|
+
exec,
|
|
397
|
+
options.signal,
|
|
398
|
+
baseInventory,
|
|
399
|
+
)
|
|
400
|
+
: await admitRepoMetadata(
|
|
401
|
+
cwd,
|
|
402
|
+
file.path,
|
|
403
|
+
exec,
|
|
404
|
+
options.signal,
|
|
405
|
+
currentInventory,
|
|
406
|
+
);
|
|
407
|
+
if (!admission.ok)
|
|
408
|
+
return {
|
|
409
|
+
ok: true,
|
|
410
|
+
file: {
|
|
411
|
+
...file,
|
|
412
|
+
hunks: [],
|
|
413
|
+
before: null,
|
|
414
|
+
after: null,
|
|
415
|
+
limitation: "unreadable",
|
|
416
|
+
},
|
|
417
|
+
};
|
|
418
|
+
if (unsupported.includes(file.path)) {
|
|
419
|
+
return {
|
|
420
|
+
ok: true,
|
|
421
|
+
file: { ...file, before: null, after: null, limitation: "unreadable" },
|
|
422
|
+
};
|
|
423
|
+
}
|
|
424
|
+
if (file.binary) {
|
|
425
|
+
return { ok: true, file: { ...file, before: null, after: null } };
|
|
426
|
+
}
|
|
427
|
+
if (file.oldMode === "160000" || file.newMode === "160000") {
|
|
428
|
+
return {
|
|
429
|
+
ok: true,
|
|
430
|
+
file: { ...file, before: null, after: null, limitation: "unreadable" },
|
|
431
|
+
};
|
|
432
|
+
}
|
|
433
|
+
if (file.status === "D") {
|
|
434
|
+
try {
|
|
435
|
+
const replacement = await lstat(resolve(cwd, file.path));
|
|
436
|
+
if (!replacement.isFile() && !replacement.isSymbolicLink()) {
|
|
437
|
+
return {
|
|
438
|
+
ok: true,
|
|
439
|
+
file: {
|
|
440
|
+
...file,
|
|
441
|
+
before: null,
|
|
442
|
+
after: null,
|
|
443
|
+
limitation: "unreadable",
|
|
444
|
+
},
|
|
445
|
+
};
|
|
446
|
+
}
|
|
447
|
+
} catch {
|
|
448
|
+
/* A genuinely deleted path has no worktree version. */
|
|
449
|
+
}
|
|
450
|
+
}
|
|
451
|
+
let before = "";
|
|
452
|
+
let limitation: ReadSource["limitation"];
|
|
453
|
+
if (!/^[A?]/.test(file.status)) {
|
|
454
|
+
const object = objects.get(file.oldPath);
|
|
455
|
+
if (!object)
|
|
456
|
+
return { ok: false, error: "Incomplete Git base tree; collect again." };
|
|
457
|
+
if (object.size > STATE_MAX_CHARS) limitation = "too_large";
|
|
458
|
+
else if (loaded.invalid.has(object.id)) limitation = "not UTF-8 text";
|
|
459
|
+
else before = loaded.blobs.get(object.id) ?? "";
|
|
460
|
+
}
|
|
461
|
+
const after =
|
|
462
|
+
file.status === "D"
|
|
463
|
+
? { text: "", binary: false }
|
|
464
|
+
: await readWorktree(scoped, exec, file.path, currentInventory);
|
|
465
|
+
limitation = after.limitation ?? limitation;
|
|
466
|
+
const unverifiableHunk =
|
|
467
|
+
limitation === "too_large" &&
|
|
468
|
+
file.hunks.some(
|
|
469
|
+
(hunk) =>
|
|
470
|
+
(hunk.beforeText ?? "").includes("\uFFFD") ||
|
|
471
|
+
(hunk.afterText ?? "").includes("\uFFFD"),
|
|
472
|
+
);
|
|
473
|
+
if (unverifiableHunk) limitation = "not UTF-8 text";
|
|
474
|
+
return {
|
|
475
|
+
ok: true,
|
|
476
|
+
file: {
|
|
477
|
+
...file,
|
|
478
|
+
hunks:
|
|
479
|
+
after.limitation === "unreadable" || limitation === "not UTF-8 text"
|
|
480
|
+
? file.hunks.map(
|
|
481
|
+
({ beforeText: _before, afterText: _after, ...hunk }) => hunk,
|
|
482
|
+
)
|
|
483
|
+
: file.hunks,
|
|
484
|
+
binary: after.binary,
|
|
485
|
+
limitation,
|
|
486
|
+
before:
|
|
487
|
+
/^[A?]/.test(file.status) ||
|
|
488
|
+
limitation === "too_large" ||
|
|
489
|
+
limitation === "not UTF-8 text"
|
|
490
|
+
? null
|
|
491
|
+
: before,
|
|
492
|
+
after: file.status === "D" || limitation ? null : after.text,
|
|
493
|
+
},
|
|
494
|
+
};
|
|
495
|
+
};
|
|
496
|
+
const sources: SourceFile[] = [];
|
|
497
|
+
for (let offset = 0; offset < files.length; offset += 4) {
|
|
498
|
+
const batch = await Promise.all(
|
|
499
|
+
files.slice(offset, offset + 4).map(collectFile),
|
|
500
|
+
);
|
|
501
|
+
for (const result of batch) {
|
|
502
|
+
if (!result.ok) return result;
|
|
503
|
+
sources.push(result.file);
|
|
504
|
+
}
|
|
505
|
+
}
|
|
506
|
+
return { ok: true, files: sources };
|
|
507
|
+
}
|
|
508
|
+
export async function collectUnits(
|
|
509
|
+
exec: GitExec,
|
|
510
|
+
options: DiffOptions,
|
|
511
|
+
sourceParser?: SyntaxParser,
|
|
512
|
+
): Promise<
|
|
513
|
+
CollectionResult<{
|
|
514
|
+
files: SourceFile[];
|
|
515
|
+
units: EvidenceUnit[];
|
|
516
|
+
limits: UnitLimit[];
|
|
517
|
+
}>
|
|
518
|
+
> {
|
|
519
|
+
const [diff, parser] = await Promise.all([
|
|
520
|
+
collectDiff(exec, options),
|
|
521
|
+
sourceParser ?? loadSyntaxParser(),
|
|
522
|
+
]);
|
|
523
|
+
if (!diff.ok) return diff;
|
|
524
|
+
return { ok: true, files: diff.files, ...buildUnits(diff.files, parser) };
|
|
525
|
+
}
|
|
@@ -0,0 +1,197 @@
|
|
|
1
|
+
import {
|
|
2
|
+
LOCATE_LABEL_MAX_CHARS,
|
|
3
|
+
LOCATE_READ_BUFFER_BYTES,
|
|
4
|
+
LOCATE_WHOLE_MAX_BYTES,
|
|
5
|
+
LOCATE_WHOLE_MAX_CHARS,
|
|
6
|
+
LOCATE_WINDOW_LINES,
|
|
7
|
+
LOCATE_WINDOW_MAX_SERIALIZED_CHARS,
|
|
8
|
+
STATE_MAX_CHARS,
|
|
9
|
+
} from "../constants.ts";
|
|
10
|
+
import type { GitExec } from "../core/git.ts";
|
|
11
|
+
import { truncate } from "../core/truncate.ts";
|
|
12
|
+
import type { Result } from "../result.ts";
|
|
13
|
+
import { openRepoFile } from "./files.ts";
|
|
14
|
+
import { createUtf8Decoder, decodeUtf8 } from "./utf8.ts";
|
|
15
|
+
|
|
16
|
+
type Window = {
|
|
17
|
+
start: number;
|
|
18
|
+
end: number;
|
|
19
|
+
label: string;
|
|
20
|
+
text: string;
|
|
21
|
+
serializedTextChars: number;
|
|
22
|
+
};
|
|
23
|
+
type FileOutline =
|
|
24
|
+
| { kind: "whole"; bytes: number; lines: number; text: string }
|
|
25
|
+
| { kind: "outline"; bytes: number; lines: number; windows: Window[] };
|
|
26
|
+
type Scan = { bytes: number; lines: number; windows: Window[]; text: string };
|
|
27
|
+
/** The initial read is bounded; large-file outlines retain labels rather than source. */
|
|
28
|
+
export async function readLocateFile(
|
|
29
|
+
cwd: string,
|
|
30
|
+
path: string,
|
|
31
|
+
signal?: AbortSignal,
|
|
32
|
+
exec?: GitExec,
|
|
33
|
+
): Promise<Result<FileOutline>> {
|
|
34
|
+
if (/^[a-z][a-z0-9+.-]*:\/\//i.test(path))
|
|
35
|
+
return {
|
|
36
|
+
ok: false,
|
|
37
|
+
error: `Internal URLs are not files; use read for ${path}.`,
|
|
38
|
+
};
|
|
39
|
+
const opened = await openRepoFile(cwd, path, { exec, signal });
|
|
40
|
+
if (!opened.ok) return opened;
|
|
41
|
+
const handle = opened.handle;
|
|
42
|
+
try {
|
|
43
|
+
const stat = await handle.stat();
|
|
44
|
+
if (!stat.isFile()) return { ok: false, error: `Not a file: ${path}.` };
|
|
45
|
+
const buffer = Buffer.alloc(
|
|
46
|
+
Math.min(stat.size + 1, LOCATE_WHOLE_MAX_BYTES + 1),
|
|
47
|
+
);
|
|
48
|
+
let count = 0;
|
|
49
|
+
while (count < buffer.length) {
|
|
50
|
+
signal?.throwIfAborted();
|
|
51
|
+
const read = await handle.read(
|
|
52
|
+
buffer,
|
|
53
|
+
count,
|
|
54
|
+
buffer.length - count,
|
|
55
|
+
null,
|
|
56
|
+
);
|
|
57
|
+
if (!read.bytesRead) break;
|
|
58
|
+
count += read.bytesRead;
|
|
59
|
+
}
|
|
60
|
+
if (buffer.subarray(0, count).includes(0))
|
|
61
|
+
return { ok: false, error: `Not a text file: ${path}.` };
|
|
62
|
+
if (count < buffer.length) {
|
|
63
|
+
const decoded = decodeUtf8(buffer.subarray(0, count));
|
|
64
|
+
if (!decoded.ok) return { ok: false, error: `${path}: ${decoded.error}` };
|
|
65
|
+
const text = decoded.text;
|
|
66
|
+
if (text.length <= LOCATE_WHOLE_MAX_CHARS)
|
|
67
|
+
return {
|
|
68
|
+
ok: true,
|
|
69
|
+
kind: "whole",
|
|
70
|
+
bytes: count,
|
|
71
|
+
lines: text
|
|
72
|
+
? text.split("\n").length - Number(text.endsWith("\n"))
|
|
73
|
+
: 0,
|
|
74
|
+
text,
|
|
75
|
+
};
|
|
76
|
+
}
|
|
77
|
+
const scan = await scanRange(cwd, path, undefined, signal, exec);
|
|
78
|
+
return scan.ok
|
|
79
|
+
? {
|
|
80
|
+
ok: true,
|
|
81
|
+
kind: "outline",
|
|
82
|
+
bytes: stat.size,
|
|
83
|
+
lines: scan.lines,
|
|
84
|
+
windows: scan.windows,
|
|
85
|
+
}
|
|
86
|
+
: scan;
|
|
87
|
+
} catch (error) {
|
|
88
|
+
return { ok: false, error: `Cannot read ${path}: ${String(error)}` };
|
|
89
|
+
} finally {
|
|
90
|
+
await handle.close();
|
|
91
|
+
}
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
export async function scanRange(
|
|
95
|
+
cwd: string,
|
|
96
|
+
path: string,
|
|
97
|
+
range?: { start: number; end: number },
|
|
98
|
+
signal?: AbortSignal,
|
|
99
|
+
exec?: GitExec,
|
|
100
|
+
): Promise<Result<Scan>> {
|
|
101
|
+
const opened = await openRepoFile(cwd, path, { exec, signal });
|
|
102
|
+
if (!opened.ok) return opened;
|
|
103
|
+
const handle = opened.handle;
|
|
104
|
+
const decode = createUtf8Decoder();
|
|
105
|
+
const buffer = Buffer.alloc(LOCATE_READ_BUFFER_BYTES);
|
|
106
|
+
let reachedEof = false;
|
|
107
|
+
let line = 1,
|
|
108
|
+
label = "",
|
|
109
|
+
pending = false,
|
|
110
|
+
text = "",
|
|
111
|
+
bytes = 0;
|
|
112
|
+
const windows: Window[] = [];
|
|
113
|
+
let windowStart = 1,
|
|
114
|
+
head = "",
|
|
115
|
+
tail = "",
|
|
116
|
+
serializedTextChars = 2;
|
|
117
|
+
const finishWindow = (end: number) => {
|
|
118
|
+
windows.push({
|
|
119
|
+
start: windowStart,
|
|
120
|
+
end,
|
|
121
|
+
label: label || `lines ${windowStart}-${end}`,
|
|
122
|
+
text: `${head}\n…\n${tail}`,
|
|
123
|
+
serializedTextChars,
|
|
124
|
+
});
|
|
125
|
+
windowStart = end + 1;
|
|
126
|
+
label = "";
|
|
127
|
+
head = "";
|
|
128
|
+
tail = "";
|
|
129
|
+
serializedTextChars = 2;
|
|
130
|
+
};
|
|
131
|
+
const consume = (chunk: string) => {
|
|
132
|
+
for (const piece of chunk.split(/(?<=\n)/)) {
|
|
133
|
+
if (!piece) continue;
|
|
134
|
+
pending = true;
|
|
135
|
+
const excerptLimit = LOCATE_WINDOW_MAX_SERIALIZED_CHARS;
|
|
136
|
+
serializedTextChars += JSON.stringify(piece).length - 2;
|
|
137
|
+
head = truncate(head + piece, excerptLimit);
|
|
138
|
+
const combined = tail + piece;
|
|
139
|
+
const tailStart = Math.max(0, combined.length - excerptLimit);
|
|
140
|
+
tail = combined.slice(
|
|
141
|
+
tailStart +
|
|
142
|
+
(tailStart > 0 && /[\uDC00-\uDFFF]/.test(combined[tailStart] ?? "")
|
|
143
|
+
? 1
|
|
144
|
+
: 0),
|
|
145
|
+
);
|
|
146
|
+
if (
|
|
147
|
+
(line - 1) % LOCATE_WINDOW_LINES === 0 &&
|
|
148
|
+
label.length < LOCATE_LABEL_MAX_CHARS
|
|
149
|
+
)
|
|
150
|
+
label = truncate(label + piece.trim(), LOCATE_LABEL_MAX_CHARS);
|
|
151
|
+
if (range && line >= range.start && line <= range.end) text += piece;
|
|
152
|
+
if (text.length > STATE_MAX_CHARS) return false;
|
|
153
|
+
if (piece.endsWith("\n")) {
|
|
154
|
+
if (
|
|
155
|
+
line - windowStart + 1 >= LOCATE_WINDOW_LINES ||
|
|
156
|
+
serializedTextChars >= LOCATE_WINDOW_MAX_SERIALIZED_CHARS
|
|
157
|
+
)
|
|
158
|
+
finishWindow(line);
|
|
159
|
+
line++;
|
|
160
|
+
pending = false;
|
|
161
|
+
}
|
|
162
|
+
}
|
|
163
|
+
return true;
|
|
164
|
+
};
|
|
165
|
+
try {
|
|
166
|
+
while (true) {
|
|
167
|
+
signal?.throwIfAborted();
|
|
168
|
+
const read = await handle.read(buffer, 0, buffer.length, null);
|
|
169
|
+
if (!read.bytesRead) {
|
|
170
|
+
reachedEof = true;
|
|
171
|
+
break;
|
|
172
|
+
}
|
|
173
|
+
if (buffer.subarray(0, read.bytesRead).includes(0))
|
|
174
|
+
return { ok: false, error: `Not a text file: ${path}.` };
|
|
175
|
+
bytes += read.bytesRead;
|
|
176
|
+
const decoded = decode(buffer.subarray(0, read.bytesRead));
|
|
177
|
+
if (!decoded.ok) return { ok: false, error: `${path}: ${decoded.error}` };
|
|
178
|
+
if (!consume(decoded.text))
|
|
179
|
+
return {
|
|
180
|
+
ok: false,
|
|
181
|
+
error: `Selected range exceeds ${STATE_MAX_CHARS} chars; read a narrower range directly.`,
|
|
182
|
+
};
|
|
183
|
+
if (range && line > range.end) break;
|
|
184
|
+
}
|
|
185
|
+
const end = reachedEof ? decode() : { ok: true as const, text: "" };
|
|
186
|
+
if (!end.ok) return { ok: false, error: `${path}: ${end.error}` };
|
|
187
|
+
if (!consume(end.text))
|
|
188
|
+
return { ok: false, error: "Selected range too large." };
|
|
189
|
+
const lines = line - Number(!pending);
|
|
190
|
+
if (lines > 0 && (windows.at(-1)?.end ?? 0) < lines) finishWindow(lines);
|
|
191
|
+
return { ok: true, bytes, lines, windows, text };
|
|
192
|
+
} catch (error) {
|
|
193
|
+
return { ok: false, error: `Cannot read ${path}: ${String(error)}` };
|
|
194
|
+
} finally {
|
|
195
|
+
await handle.close();
|
|
196
|
+
}
|
|
197
|
+
}
|