@nexrall/code-core 1.4.63 → 1.4.65
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/agent/agentRegistry.d.ts +14 -3
- package/dist/agent/agentRegistry.d.ts.map +1 -1
- package/dist/agent/agentRegistry.js +136 -6
- package/dist/agent/agentTypes.d.ts +51 -1
- package/dist/agent/agentTypes.d.ts.map +1 -1
- package/dist/agent/agentTypes.js +194 -11
- package/dist/agent/backgroundAgents.d.ts +66 -0
- package/dist/agent/backgroundAgents.d.ts.map +1 -0
- package/dist/agent/backgroundAgents.js +145 -0
- package/dist/agent/loop.d.ts +43 -6
- package/dist/agent/loop.d.ts.map +1 -1
- package/dist/agent/loop.js +942 -259
- package/dist/agent/modelCatalogue.d.ts +15 -0
- package/dist/agent/modelCatalogue.d.ts.map +1 -1
- package/dist/agent/modelCatalogue.js +46 -0
- package/dist/agent/readDedupe.d.ts +15 -0
- package/dist/agent/readDedupe.d.ts.map +1 -0
- package/dist/agent/readDedupe.js +146 -0
- package/dist/agent/toolPrefetch.d.ts +44 -0
- package/dist/agent/toolPrefetch.d.ts.map +1 -0
- package/dist/agent/toolPrefetch.js +101 -0
- package/dist/agent/trust.d.ts +0 -5
- package/dist/agent/trust.d.ts.map +1 -1
- package/dist/agent/trust.js +41 -0
- package/dist/api/client.d.ts +14 -0
- package/dist/api/client.d.ts.map +1 -1
- package/dist/api/client.js +88 -8
- package/dist/index.d.ts +1 -0
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +1 -0
- package/dist/permissions/modePolicy.d.ts +10 -0
- package/dist/permissions/modePolicy.d.ts.map +1 -1
- package/dist/permissions/modePolicy.js +11 -0
- package/dist/tools/executor.d.ts +36 -0
- package/dist/tools/executor.d.ts.map +1 -1
- package/dist/tools/executor.js +408 -85
- package/dist/tools/tsLangService.d.ts.map +1 -1
- package/dist/tools/tsLangService.js +107 -21
- package/dist/types.d.ts +99 -5
- package/dist/types.d.ts.map +1 -1
- package/dist/types.js +12 -1
- package/dist/util/miniYaml.d.ts +10 -0
- package/dist/util/miniYaml.d.ts.map +1 -0
- package/dist/util/miniYaml.js +149 -0
- package/package.json +8 -17
package/dist/tools/executor.js
CHANGED
|
@@ -33,6 +33,9 @@ var __importStar = (this && this.__importStar) || (function () {
|
|
|
33
33
|
};
|
|
34
34
|
})();
|
|
35
35
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
36
|
+
exports.capExternalOutput = capExternalOutput;
|
|
37
|
+
exports.runCapture = runCapture;
|
|
38
|
+
exports.globToRegex = globToRegex;
|
|
36
39
|
exports.executeTool = executeTool;
|
|
37
40
|
const fs = __importStar(require("fs"));
|
|
38
41
|
const path = __importStar(require("path"));
|
|
@@ -64,7 +67,15 @@ const peerTransport_1 = require("../agent/peerTransport");
|
|
|
64
67
|
const DEFAULT_TIMEOUT_MS = 60000;
|
|
65
68
|
const MAX_FETCH_BYTES = 200 * 1024; // 200 KB
|
|
66
69
|
const MAX_READ_BYTES = 500 * 1024; // 500 KB — still used by notebook_read (JSON.parse needs it whole)
|
|
67
|
-
|
|
70
|
+
// bash / grep / list output cap. 30K chars matches Claude Code's Bash default; it was 100K,
|
|
71
|
+
// i.e. ~28K tokens re-sent on every later request for one noisy command. Full bash output
|
|
72
|
+
// still spills to a temp file (see GAP B below), so nothing is lost — only inlined less.
|
|
73
|
+
const MAX_OUTPUT_CHARS = 30000;
|
|
74
|
+
// fetch_url: text handed to the model AFTER html stripping. The download cap above stays
|
|
75
|
+
// larger because raw HTML is mostly markup that stripHtml discards.
|
|
76
|
+
const MAX_FETCH_OUTPUT_CHARS = 40000;
|
|
77
|
+
// glob / search_files(files) list cap, most-recent first — same as Claude Code's Glob.
|
|
78
|
+
const MAX_GLOB_RESULTS = 100;
|
|
68
79
|
// read_file has NO byte-size gate (unlike the old 500KB cutoff) — like Claude
|
|
69
80
|
// Code, a file of any size can be read; it's always streamed line-by-line so
|
|
70
81
|
// memory is bounded regardless of file size. Instead there are two independent
|
|
@@ -81,6 +92,10 @@ const MAX_READ_LINES = 2000;
|
|
|
81
92
|
const MAX_READ_TOKENS = 25000;
|
|
82
93
|
const READ_CHARS_PER_TOKEN = 4; // rough, conservative — consistent with loop.ts's resume estimator
|
|
83
94
|
const MAX_READ_CHARS = MAX_READ_TOKENS * READ_CHARS_PER_TOKEN; // 100,000 chars
|
|
95
|
+
// Per-line cap (Claude Code: 2000). Without it a line longer than the whole token budget
|
|
96
|
+
// (minified bundle, one-line JSON) could never be kept: the read returned "lines 1-0 (more
|
|
97
|
+
// lines follow — pass offset:0 to continue)" forever and the model fell back to `cat`.
|
|
98
|
+
const MAX_READ_LINE_CHARS = 2000;
|
|
84
99
|
// GAP B — output spillover. The inline bash result keeps only HEAD+TAIL (~100 KB),
|
|
85
100
|
// which loses the MIDDLE of a large log — often exactly where a stack trace's root
|
|
86
101
|
// cause or a failing assertion lives. To make the full log recoverable WITHOUT
|
|
@@ -270,12 +285,16 @@ async function readFileWindowed(resolved, offset, requestedLimit) {
|
|
|
270
285
|
let keptChars = 0;
|
|
271
286
|
let lineNo = 0; // 0-based index of the NEXT line to be completed
|
|
272
287
|
let carry = ''; // partial line spanning chunk boundaries
|
|
288
|
+
let carryOverflow = false; // carry already holds MAX_READ_LINE_CHARS+ of a longer line
|
|
273
289
|
let stopped = false;
|
|
274
290
|
let hitTokenCap = false;
|
|
275
291
|
let sawEof = false;
|
|
276
292
|
const stream = fs.createReadStream(resolved, { encoding: 'utf-8', highWaterMark: 256 * 1024 });
|
|
277
293
|
// Returns false if adding this line would exceed the token budget (line NOT kept).
|
|
278
|
-
const tryPushLine = (
|
|
294
|
+
const tryPushLine = (raw) => {
|
|
295
|
+
const line = raw.length > MAX_READ_LINE_CHARS
|
|
296
|
+
? `${raw.slice(0, MAX_READ_LINE_CHARS)}… [line truncated at ${MAX_READ_LINE_CHARS} chars]`
|
|
297
|
+
: raw;
|
|
279
298
|
const formatted = formatNumberedLine(offset, kept.length, line);
|
|
280
299
|
const added = formatted.length + 1; // +1 for the join newline
|
|
281
300
|
if (keptChars + added > MAX_READ_CHARS) {
|
|
@@ -308,9 +327,23 @@ async function readFileWindowed(resolved, offset, requestedLimit) {
|
|
|
308
327
|
resolve({ output: `[File: ${resolved} — lines ${first}-${last}${totalNote}${note}]\n${numbered}` });
|
|
309
328
|
};
|
|
310
329
|
stream.on('data', (chunk) => {
|
|
311
|
-
|
|
330
|
+
let str = typeof chunk === 'string' ? chunk : chunk.toString('utf-8');
|
|
331
|
+
if (carryOverflow) {
|
|
332
|
+
// Still inside an over-long line whose kept prefix is already in `carry`:
|
|
333
|
+
// discard until its newline so memory stays bounded for one-line files.
|
|
334
|
+
const nl = str.indexOf('\n');
|
|
335
|
+
if (nl < 0)
|
|
336
|
+
return;
|
|
337
|
+
str = str.slice(nl);
|
|
338
|
+
carryOverflow = false;
|
|
339
|
+
}
|
|
340
|
+
const text = carry + str;
|
|
312
341
|
const lines = text.split('\n');
|
|
313
342
|
carry = lines.pop() ?? ''; // last element is an incomplete line (or '')
|
|
343
|
+
if (carry.length > MAX_READ_LINE_CHARS) {
|
|
344
|
+
carry = carry.slice(0, MAX_READ_LINE_CHARS + 1); // +1 so tryPushLine still marks it truncated
|
|
345
|
+
carryOverflow = true;
|
|
346
|
+
}
|
|
314
347
|
for (const line of lines) {
|
|
315
348
|
if (lineNo >= offset) {
|
|
316
349
|
if (lineNo >= hardEndLine) {
|
|
@@ -434,7 +467,7 @@ async function writeFile(input, workDir) {
|
|
|
434
467
|
// A brand-new file has no "removed" side — every line is an addition.
|
|
435
468
|
return { output: `Created ${resolved} (${lines} lines, ${bytes} bytes)${sec}`, linesAdded: lines, linesRemoved: 0 };
|
|
436
469
|
}
|
|
437
|
-
const xfile = crossFileBreakageWarning(resolved, normalizeLF(priorContent), normalizeLF(content), workDir ?? process.cwd());
|
|
470
|
+
const xfile = await crossFileBreakageWarning(resolved, normalizeLF(priorContent), normalizeLF(content), workDir ?? process.cwd());
|
|
438
471
|
// Reward-hacking guard: a write_file that OVERWRITES an existing test file can
|
|
439
472
|
// silently drop assertions / test cases. The loop layer can't see the prior
|
|
440
473
|
// content — but we can (we just read it). Run the full old→new analysis and
|
|
@@ -1111,22 +1144,116 @@ async function bash(input, abortSignal, sandbox, workDir, onStream) {
|
|
|
1111
1144
|
child.on('close', () => clearInterval(pollAbort));
|
|
1112
1145
|
});
|
|
1113
1146
|
}
|
|
1147
|
+
/**
|
|
1148
|
+
* Cap output from a tool we do not control (MCP servers). Built-in tools all cap their
|
|
1149
|
+
* own output, but an MCP result went into the conversation verbatim: one call to a
|
|
1150
|
+
* DB/browser/log server could inline megabytes, which is then re-sent (and re-billed as
|
|
1151
|
+
* cache writes) on every later round until compaction. Keep HEAD+TAIL inline and spill
|
|
1152
|
+
* the full text to a temp file the model can read_file with offset/limit — the same
|
|
1153
|
+
* contract as bash's GAP B spill.
|
|
1154
|
+
*/
|
|
1155
|
+
function capExternalOutput(output, label, maxChars = MAX_OUTPUT_CHARS) {
|
|
1156
|
+
if (output.length <= maxChars)
|
|
1157
|
+
return output;
|
|
1158
|
+
let spillNote = '';
|
|
1159
|
+
try {
|
|
1160
|
+
fs.mkdirSync(SPILL_DIR, { recursive: true });
|
|
1161
|
+
const safe = label.replace(/[^a-zA-Z0-9_-]/g, '_').slice(0, 60);
|
|
1162
|
+
const spillPath = path.join(SPILL_DIR, `${safe}-${Date.now()}-${Math.random().toString(36).slice(2, 8)}.log`);
|
|
1163
|
+
fs.writeFileSync(spillPath, (0, safeSlice_1.sliceSafeEnd)(output, MAX_SPILL_BYTES));
|
|
1164
|
+
spillNote = ` Full output (${output.length} chars) saved to ${spillPath} — use read_file with offset/limit to inspect any section.`;
|
|
1165
|
+
}
|
|
1166
|
+
catch { /* spill is best-effort */ }
|
|
1167
|
+
const headLen = Math.floor(maxChars * 0.7);
|
|
1168
|
+
const tailLen = maxChars - headLen;
|
|
1169
|
+
const omitted = output.length - headLen - tailLen;
|
|
1170
|
+
return `${(0, safeSlice_1.sliceSafeEnd)(output, headLen)}\n\n[… ${omitted} chars omitted.${spillNote}]\n\n${(0, safeSlice_1.sliceSafeStart)(output, output.length - tailLen)}`;
|
|
1171
|
+
}
|
|
1172
|
+
function runCapture(cmd, args, opts) {
|
|
1173
|
+
return new Promise((resolve) => {
|
|
1174
|
+
let stdout = '';
|
|
1175
|
+
let stderr = '';
|
|
1176
|
+
let bytes = 0;
|
|
1177
|
+
let error;
|
|
1178
|
+
let settled = false;
|
|
1179
|
+
let child;
|
|
1180
|
+
try {
|
|
1181
|
+
child = (0, child_process_1.spawn)(cmd, args, { cwd: opts.cwd, stdio: ['ignore', 'pipe', 'pipe'] });
|
|
1182
|
+
}
|
|
1183
|
+
catch (e) {
|
|
1184
|
+
resolve({ stdout: '', stderr: '', status: null, signal: null, error: e });
|
|
1185
|
+
return;
|
|
1186
|
+
}
|
|
1187
|
+
const finish = (status, signal) => {
|
|
1188
|
+
if (settled)
|
|
1189
|
+
return;
|
|
1190
|
+
settled = true;
|
|
1191
|
+
clearTimeout(timer);
|
|
1192
|
+
resolve({ stdout, stderr, status, signal, ...(error ? { error } : {}) });
|
|
1193
|
+
};
|
|
1194
|
+
const kill = (code) => {
|
|
1195
|
+
if (!error)
|
|
1196
|
+
error = Object.assign(new Error(`${cmd} ${code}`), { code });
|
|
1197
|
+
try {
|
|
1198
|
+
child.kill('SIGTERM');
|
|
1199
|
+
}
|
|
1200
|
+
catch { /* already gone */ }
|
|
1201
|
+
};
|
|
1202
|
+
const timer = setTimeout(() => kill('ETIMEDOUT'), opts.timeout);
|
|
1203
|
+
child.stdout.setEncoding('utf-8');
|
|
1204
|
+
child.stderr.setEncoding('utf-8');
|
|
1205
|
+
child.stdout.on('data', (chunk) => {
|
|
1206
|
+
if (error)
|
|
1207
|
+
return;
|
|
1208
|
+
bytes += Buffer.byteLength(chunk);
|
|
1209
|
+
if (bytes > opts.maxBuffer) {
|
|
1210
|
+
// Keep what fits, like spawnSync's truncated stdout on ENOBUFS.
|
|
1211
|
+
stdout += chunk.slice(0, Math.max(0, chunk.length - (bytes - opts.maxBuffer)));
|
|
1212
|
+
kill('ENOBUFS');
|
|
1213
|
+
return;
|
|
1214
|
+
}
|
|
1215
|
+
stdout += chunk;
|
|
1216
|
+
});
|
|
1217
|
+
child.stderr.on('data', (chunk) => { if (stderr.length < 64 * 1024)
|
|
1218
|
+
stderr += chunk; });
|
|
1219
|
+
child.on('error', (e) => { if (!error)
|
|
1220
|
+
error = e; finish(null, null); });
|
|
1221
|
+
child.on('close', (code, signal) => finish(code, signal));
|
|
1222
|
+
});
|
|
1223
|
+
}
|
|
1114
1224
|
// Detect ripgrep once per process — preferred over grep (faster, respects .gitignore).
|
|
1225
|
+
//
|
|
1226
|
+
// Binary resolution: NEXRALL_RG_PATH first (the VS Code extension points it at the rg
|
|
1227
|
+
// that ships inside every VS Code / Cursor install, so users without a system rg still
|
|
1228
|
+
// get the fast, .gitignore-aware path), then `rg` on PATH. Before this, a machine with
|
|
1229
|
+
// no rg installed silently fell back to grep and a full, ignore-unaware directory walk.
|
|
1115
1230
|
let _rgChecked = false;
|
|
1116
1231
|
let _rgAvailable = false;
|
|
1232
|
+
let _rgBin = 'rg';
|
|
1117
1233
|
function ripgrepAvailable() {
|
|
1118
1234
|
if (!_rgChecked) {
|
|
1119
|
-
const probe = (0, child_process_1.spawnSync)('rg', ['--version'], { encoding: 'utf-8', timeout: 3000 });
|
|
1120
|
-
_rgAvailable = probe.status === 0;
|
|
1121
1235
|
_rgChecked = true;
|
|
1236
|
+
const candidates = [process.env.NEXRALL_RG_PATH, 'rg'].filter((c) => !!c);
|
|
1237
|
+
for (const bin of candidates) {
|
|
1238
|
+
const probe = (0, child_process_1.spawnSync)(bin, ['--version'], { encoding: 'utf-8', timeout: 3000 });
|
|
1239
|
+
if (probe.status === 0) {
|
|
1240
|
+
_rgBin = bin;
|
|
1241
|
+
_rgAvailable = true;
|
|
1242
|
+
break;
|
|
1243
|
+
}
|
|
1244
|
+
}
|
|
1122
1245
|
}
|
|
1123
1246
|
return _rgAvailable;
|
|
1124
1247
|
}
|
|
1248
|
+
/** The ripgrep binary to spawn. Only meaningful after ripgrepAvailable() returned true. */
|
|
1249
|
+
function rgBin() {
|
|
1250
|
+
return _rgBin;
|
|
1251
|
+
}
|
|
1125
1252
|
// ── Cross-file breakage warning ──────────────────────────────────────────────
|
|
1126
1253
|
// After an edit removes/renames an exported symbol, scan the rest of the repo for
|
|
1127
1254
|
// surviving references. Returns a short warning string (or '' when clean). Best-
|
|
1128
1255
|
// effort, time-boxed, and never throws — a scan failure must not fail the edit.
|
|
1129
|
-
function crossFileBreakageWarning(editedAbsPath, oldContent, newContent, workDir) {
|
|
1256
|
+
async function crossFileBreakageWarning(editedAbsPath, oldContent, newContent, workDir) {
|
|
1130
1257
|
if (process.env.NEXRALL_CROSSFILE_CHECK === '0')
|
|
1131
1258
|
return '';
|
|
1132
1259
|
let removed;
|
|
@@ -1147,8 +1274,9 @@ function crossFileBreakageWarning(editedAbsPath, oldContent, newContent, workDir
|
|
|
1147
1274
|
} })();
|
|
1148
1275
|
const hits = [];
|
|
1149
1276
|
const MAX_SYMBOLS = 8;
|
|
1150
|
-
|
|
1151
|
-
|
|
1277
|
+
// Scans run concurrently (they used to be up to 8 sequential blocking rg calls).
|
|
1278
|
+
const scanned = await Promise.all(removed.slice(0, MAX_SYMBOLS).map(async (sym) => ({ sym, refs: await scanReferences(sym.name, workDir, editedReal) })));
|
|
1279
|
+
for (const { sym, refs } of scanned) {
|
|
1152
1280
|
if (refs.length) {
|
|
1153
1281
|
const shown = refs.slice(0, 3).map((r) => ` ${r}`).join('\n');
|
|
1154
1282
|
const more = refs.length > 3 ? `\n … and ${refs.length - 3} more` : '';
|
|
@@ -1163,31 +1291,20 @@ function crossFileBreakageWarning(editedAbsPath, oldContent, newContent, workDir
|
|
|
1163
1291
|
}
|
|
1164
1292
|
// Find files (other than the edited one) that reference `name` as a whole word.
|
|
1165
1293
|
// Uses ripgrep when available (fast, .gitignore-aware), else a bounded grep -r.
|
|
1166
|
-
function scanReferences(name, workDir, excludeRealPath) {
|
|
1294
|
+
async function scanReferences(name, workDir, excludeRealPath) {
|
|
1167
1295
|
const pattern = `\\b${name.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')}\\b`;
|
|
1168
1296
|
const files = new Set();
|
|
1169
1297
|
try {
|
|
1170
|
-
|
|
1171
|
-
|
|
1172
|
-
|
|
1173
|
-
|
|
1174
|
-
|
|
1175
|
-
|
|
1176
|
-
|
|
1177
|
-
|
|
1178
|
-
|
|
1179
|
-
|
|
1180
|
-
else {
|
|
1181
|
-
const r = (0, child_process_1.spawnSync)('grep', ['-rlI', '--exclude-dir=node_modules', '--exclude-dir=.git',
|
|
1182
|
-
'--exclude-dir=dist', '--exclude-dir=.next', '-E', pattern, '.'], {
|
|
1183
|
-
cwd: workDir, encoding: 'utf-8', timeout: 8000, maxBuffer: 8 * 1024 * 1024,
|
|
1184
|
-
});
|
|
1185
|
-
if (r.stdout)
|
|
1186
|
-
for (const line of r.stdout.split('\n')) {
|
|
1187
|
-
if (line.trim())
|
|
1188
|
-
files.add(line.trim());
|
|
1189
|
-
}
|
|
1190
|
-
}
|
|
1298
|
+
const opts = { cwd: workDir, timeout: 8000, maxBuffer: 8 * 1024 * 1024 };
|
|
1299
|
+
const r = ripgrepAvailable()
|
|
1300
|
+
? await runCapture(rgBin(), ['-l', '--no-messages', '-e', pattern, '.'], opts)
|
|
1301
|
+
: await runCapture('grep', ['-rlI', '--exclude-dir=node_modules', '--exclude-dir=.git',
|
|
1302
|
+
'--exclude-dir=dist', '--exclude-dir=.next', '-E', pattern, '.'], opts);
|
|
1303
|
+
if (r.stdout)
|
|
1304
|
+
for (const line of r.stdout.split('\n')) {
|
|
1305
|
+
if (line.trim())
|
|
1306
|
+
files.add(line.trim());
|
|
1307
|
+
}
|
|
1191
1308
|
}
|
|
1192
1309
|
catch {
|
|
1193
1310
|
return [];
|
|
@@ -1209,9 +1326,13 @@ function scanReferences(name, workDir, excludeRealPath) {
|
|
|
1209
1326
|
async function searchFiles(input, workDir) {
|
|
1210
1327
|
const pattern = typeof input.pattern === 'string' ? input.pattern : '';
|
|
1211
1328
|
const searchPath = typeof input.path === 'string' ? input.path : '.';
|
|
1212
|
-
const searchType = typeof input.type === 'string' ? input.type : 'content';
|
|
1213
1329
|
const ignoreCase = input.case_insensitive === true;
|
|
1214
1330
|
const contextLines = typeof input.context_lines === 'number' ? Math.min(10, Math.max(0, input.context_lines)) : 0;
|
|
1331
|
+
// Default is "files" (paths of matching files only), like Claude Code's Grep
|
|
1332
|
+
// files_with_matches: a content dump of a common pattern was ~50× larger. Asking for
|
|
1333
|
+
// context_lines implies the caller wants content.
|
|
1334
|
+
const searchType = typeof input.type === 'string' ? input.type : (contextLines > 0 ? 'content' : 'files');
|
|
1335
|
+
const filesOnly = searchType === 'files' || searchType === 'files_with_matches';
|
|
1215
1336
|
const include = typeof input.include === 'string' ? input.include : ''; // e.g. "*.ts"
|
|
1216
1337
|
if (!pattern)
|
|
1217
1338
|
return { error: 'Missing required parameter: pattern' };
|
|
@@ -1220,7 +1341,8 @@ async function searchFiles(input, workDir) {
|
|
|
1220
1341
|
if (searchType === 'filename') {
|
|
1221
1342
|
const matches = [];
|
|
1222
1343
|
const patternLower = pattern.toLowerCase();
|
|
1223
|
-
|
|
1344
|
+
const listing = await listFiles(resolved);
|
|
1345
|
+
listing.files.forEach((filePath) => {
|
|
1224
1346
|
const name = path.basename(filePath);
|
|
1225
1347
|
const nameLower = name.toLowerCase();
|
|
1226
1348
|
// Also match against the path relative to the search root — a pattern
|
|
@@ -1238,9 +1360,15 @@ async function searchFiles(input, workDir) {
|
|
|
1238
1360
|
});
|
|
1239
1361
|
if (matches.length === 0)
|
|
1240
1362
|
return { output: 'No matching files found.' };
|
|
1241
|
-
|
|
1363
|
+
const relBase = workDir ? path.resolve(workDir) : process.cwd();
|
|
1364
|
+
let out = matches.map((f) => {
|
|
1365
|
+
const r = path.relative(relBase, f);
|
|
1366
|
+
return r.startsWith('..') || path.isAbsolute(r) ? f : r.replace(/\\/g, '/');
|
|
1367
|
+
}).join('\n');
|
|
1242
1368
|
if (out.length > MAX_OUTPUT_CHARS)
|
|
1243
1369
|
out = (0, safeSlice_1.sliceSafeEnd)(out, MAX_OUTPUT_CHARS) + '\n[Truncated]';
|
|
1370
|
+
if (listing.truncated)
|
|
1371
|
+
out += `\n[Scan stopped after ${WALK_MAX_FILES.toLocaleString('en-US')} files — narrow the path for complete results.]`;
|
|
1244
1372
|
return { output: out };
|
|
1245
1373
|
}
|
|
1246
1374
|
else {
|
|
@@ -1250,20 +1378,31 @@ async function searchFiles(input, workDir) {
|
|
|
1250
1378
|
// pattern can emit tens of MB and the default silently truncates + sets
|
|
1251
1379
|
// status:null (ENOBUFS), which used to surface as an empty "search failed".
|
|
1252
1380
|
const SEARCH_MAX_BUFFER = 64 * 1024 * 1024; // 64 MB
|
|
1253
|
-
|
|
1381
|
+
// Run from workDir with a RELATIVE target so every output line is workDir-relative
|
|
1382
|
+
// (what read_file resolves against). Absolute prefixes were ~25% of the tokens.
|
|
1383
|
+
// A target outside workDir keeps absolute paths.
|
|
1384
|
+
const base = workDir ? path.resolve(workDir) : process.cwd();
|
|
1385
|
+
const relTarget = path.relative(base, resolved);
|
|
1386
|
+
const inside = !relTarget.startsWith('..') && !path.isAbsolute(relTarget);
|
|
1387
|
+
const target = inside ? (relTarget || '.') : resolved;
|
|
1388
|
+
const spawnOpts = { timeout: DEFAULT_TIMEOUT_MS, maxBuffer: SEARCH_MAX_BUFFER, cwd: base };
|
|
1254
1389
|
let result;
|
|
1255
1390
|
if (ripgrepAvailable()) {
|
|
1256
|
-
const args =
|
|
1391
|
+
const args = filesOnly
|
|
1392
|
+
? ['--files-with-matches', '--color=never', '--hidden', '--glob', '!.git']
|
|
1393
|
+
: ['--line-number', '--no-heading', '--color=never', '--hidden', '--glob', '!.git',
|
|
1394
|
+
// One minified-bundle hit could otherwise fill the whole output cap.
|
|
1395
|
+
'--max-columns', '500', '--max-columns-preview'];
|
|
1257
1396
|
if (ignoreCase)
|
|
1258
1397
|
args.push('-i');
|
|
1259
|
-
if (contextLines > 0)
|
|
1398
|
+
if (contextLines > 0 && !filesOnly)
|
|
1260
1399
|
args.push(`-C${contextLines}`);
|
|
1261
1400
|
if (include)
|
|
1262
1401
|
args.push('--glob', include);
|
|
1263
1402
|
// -m is PER-FILE in rg/grep; the real global cap is applied on the
|
|
1264
1403
|
// output below. Keep a generous per-file cap so no single file floods.
|
|
1265
|
-
args.push('-m', '200', '--regexp', pattern,
|
|
1266
|
-
result = (
|
|
1404
|
+
args.push('-m', '200', '--regexp', pattern, target);
|
|
1405
|
+
result = await runCapture(rgBin(), args, spawnOpts);
|
|
1267
1406
|
}
|
|
1268
1407
|
else {
|
|
1269
1408
|
// -E (ERE) matters: without it, grep defaults to BRE, where a bare
|
|
@@ -1276,23 +1415,26 @@ async function searchFiles(input, workDir) {
|
|
|
1276
1415
|
// fallback path whenever ripgrep isn't installed on the host. ERE
|
|
1277
1416
|
// matches rg's semantics (bare parens group, \( \) literal) so the
|
|
1278
1417
|
// SAME pattern behaves identically whether or not rg is present.
|
|
1279
|
-
const args = ['-rnE', '--binary-files=without-match', '--color=never'];
|
|
1418
|
+
const args = [filesOnly ? '-rlE' : '-rnE', '--binary-files=without-match', '--color=never'];
|
|
1280
1419
|
if (ignoreCase)
|
|
1281
1420
|
args.push('-i');
|
|
1282
|
-
if (contextLines > 0)
|
|
1421
|
+
if (contextLines > 0 && !filesOnly)
|
|
1283
1422
|
args.push(`-C${contextLines}`);
|
|
1284
1423
|
if (include)
|
|
1285
1424
|
args.push(`--include=${include}`);
|
|
1286
|
-
args.push('--exclude-dir=.git', '--exclude-dir=node_modules', '--exclude-dir=dist', '--exclude-dir=.next', '--exclude-dir=__pycache__', '--exclude-dir=.turbo', '--exclude-dir=coverage', '--exclude-dir=.cache', '-m', '200', pattern,
|
|
1287
|
-
result =
|
|
1425
|
+
args.push('--exclude-dir=.git', '--exclude-dir=node_modules', '--exclude-dir=dist', '--exclude-dir=.next', '--exclude-dir=__pycache__', '--exclude-dir=.turbo', '--exclude-dir=coverage', '--exclude-dir=.cache', '--exclude-dir=.venv', '--exclude-dir=venv', '-m', '200', pattern, target);
|
|
1426
|
+
result = await runCapture('grep', args, spawnOpts);
|
|
1288
1427
|
}
|
|
1289
|
-
|
|
1428
|
+
// rg/grep print "./x" when the target is "." — drop the noise prefix.
|
|
1429
|
+
let output = (result.stdout ?? '').replace(/^\.\//gm, '');
|
|
1290
1430
|
const stderr = result.stderr ?? '';
|
|
1291
|
-
// Distinguish real failure modes. spawnSync sets `.error`
|
|
1292
|
-
// for timeout (ETIMEDOUT), buffer overflow (ENOBUFS) and spawn failures.
|
|
1431
|
+
// Distinguish real failure modes. runCapture (like spawnSync) sets `.error`
|
|
1432
|
+
// (not `.status`) for timeout (ETIMEDOUT), buffer overflow (ENOBUFS) and spawn failures.
|
|
1293
1433
|
const spawnErr = result.error;
|
|
1294
1434
|
if (spawnErr) {
|
|
1295
|
-
|
|
1435
|
+
// ENOBUFS first: the child is SIGTERM'd on overflow too, so testing the signal
|
|
1436
|
+
// first misreported a too-large result as a timeout.
|
|
1437
|
+
if (spawnErr.code === 'ETIMEDOUT' || (result.signal === 'SIGTERM' && spawnErr.code !== 'ENOBUFS')) {
|
|
1296
1438
|
const partial = output ? `\n\nPartial results before timeout:\n${globalCapMatches(output)}` : '';
|
|
1297
1439
|
return { error: `Search timed out after ${Math.round(DEFAULT_TIMEOUT_MS / 1000)}s — narrow the path or pattern (or add an "include" filter).${partial}` };
|
|
1298
1440
|
}
|
|
@@ -1312,11 +1454,23 @@ async function searchFiles(input, workDir) {
|
|
|
1312
1454
|
// otherwise surfaces as raw, cryptic engine stderr (e.g. "parentheses
|
|
1313
1455
|
// not balanced", "Unmatched ( or \("). Give an actionable hint instead
|
|
1314
1456
|
// of just relaying the tool's internal error message verbatim.
|
|
1315
|
-
const hint = /parenthes|bracket|brace|Unmatched|repetition-operator|invalid regex/i.test(stderr)
|
|
1457
|
+
const hint = /parenthes|bracket|brace|Unmatched|unclosed|repetition-operator|invalid regex/i.test(stderr)
|
|
1316
1458
|
? ' — the pattern has invalid/unbalanced regex syntax. If you meant to match literal parentheses/brackets, escape them (e.g. "\\(", "\\)"), or simplify the pattern.'
|
|
1317
1459
|
: '';
|
|
1318
1460
|
return { error: (stderr || 'search failed').trim() + hint };
|
|
1319
1461
|
}
|
|
1462
|
+
if (filesOnly) {
|
|
1463
|
+
const files = output.split('\n').filter(Boolean);
|
|
1464
|
+
if (!files.length)
|
|
1465
|
+
return { output: 'No matches found.' };
|
|
1466
|
+
const shown = files.slice(0, MAX_GLOB_RESULTS);
|
|
1467
|
+
const more = files.length > shown.length
|
|
1468
|
+
? `\n[${files.length - shown.length} more files not shown — narrow the pattern/path]` : '';
|
|
1469
|
+
return {
|
|
1470
|
+
output: `${files.length} file(s) contain matches (type="content" for the matching lines):\n` +
|
|
1471
|
+
shown.join('\n') + more,
|
|
1472
|
+
};
|
|
1473
|
+
}
|
|
1320
1474
|
output = globalCapMatches(output);
|
|
1321
1475
|
return { output: output || 'No matches found.' };
|
|
1322
1476
|
}
|
|
@@ -1344,7 +1498,7 @@ function globalCapMatches(output) {
|
|
|
1344
1498
|
}
|
|
1345
1499
|
if (capped.length > MAX_OUTPUT_CHARS) {
|
|
1346
1500
|
capped = (0, safeSlice_1.sliceSafeEnd)(capped, MAX_OUTPUT_CHARS);
|
|
1347
|
-
note = `\n[Output truncated at ${MAX_OUTPUT_CHARS
|
|
1501
|
+
note = `\n[Output truncated at ${MAX_OUTPUT_CHARS.toLocaleString('en-US')} chars.]`;
|
|
1348
1502
|
}
|
|
1349
1503
|
return capped + note;
|
|
1350
1504
|
}
|
|
@@ -1371,13 +1525,68 @@ function matchesPattern(name, pattern) {
|
|
|
1371
1525
|
return false;
|
|
1372
1526
|
}
|
|
1373
1527
|
}
|
|
1528
|
+
// Noise directories never worth walking. .venv/venv matter most: a single Python virtualenv
|
|
1529
|
+
// is hundreds of MB of third-party files that drowned glob / filename-search results.
|
|
1530
|
+
const WALK_SKIP_DIRS = new Set([
|
|
1531
|
+
'.git', 'node_modules', 'dist', '.next', '__pycache__', '.turbo', 'coverage',
|
|
1532
|
+
'.venv', 'venv', '.tox', '.mypy_cache', '.pytest_cache', '.ruff_cache', '.gradle',
|
|
1533
|
+
'.cache', '.parcel-cache', '.svelte-kit', '.nuxt', 'target', '.idea', '.DS_Store',
|
|
1534
|
+
]);
|
|
1535
|
+
/** Hard ceiling on files visited by the fallback directory walk (no ripgrep). */
|
|
1536
|
+
const WALK_MAX_FILES = 50000;
|
|
1537
|
+
/**
|
|
1538
|
+
* List candidate files under `root` for glob / filename search.
|
|
1539
|
+
*
|
|
1540
|
+
* With ripgrep: `rg --files`, which honours .gitignore / .ignore (and git's global
|
|
1541
|
+
* excludes) — so build output, vendored deps and generated files that the project
|
|
1542
|
+
* itself ignores no longer drown results. Measured on this repo: the old walk visited
|
|
1543
|
+
* 2,972 files vs 948 tracked. `--hidden` keeps dotfiles like .github/ (still minus
|
|
1544
|
+
* anything ignored), and the WALK_SKIP_DIRS noise list is applied on top so a repo
|
|
1545
|
+
* with no .gitignore still skips node_modules etc.
|
|
1546
|
+
*
|
|
1547
|
+
* Without ripgrep: the old walk, but capped at WALK_MAX_FILES so a 100k-file monorepo
|
|
1548
|
+
* cannot stall the process; `truncated` tells the caller to say so.
|
|
1549
|
+
*/
|
|
1550
|
+
async function listFiles(root) {
|
|
1551
|
+
if (ripgrepAvailable()) {
|
|
1552
|
+
// --follow: the walkDir this replaced followed symlinks (pnpm / linked monorepo
|
|
1553
|
+
// packages); without it rg silently drops every symlinked dir AND file. rg detects
|
|
1554
|
+
// symlink loops itself.
|
|
1555
|
+
const args = ['--files', '--hidden', '--follow', '--no-messages', '--color=never'];
|
|
1556
|
+
for (const d of WALK_SKIP_DIRS)
|
|
1557
|
+
args.push('--glob', `!**/${d}/**`);
|
|
1558
|
+
args.push('.');
|
|
1559
|
+
const r = await runCapture(rgBin(), args, { cwd: root, timeout: DEFAULT_TIMEOUT_MS, maxBuffer: 64 * 1024 * 1024 });
|
|
1560
|
+
// rg exits 1 when it finds no files at all — a real (empty) answer, not a failure.
|
|
1561
|
+
// Exit 2 = "some path errored" (a symlink loop, an unreadable dir) — the list of
|
|
1562
|
+
// everything else is still complete, so keep it rather than re-walking the tree.
|
|
1563
|
+
if (!r.error && (r.status === 0 || r.status === 1 || (r.status === 2 && r.stdout))) {
|
|
1564
|
+
const files = r.stdout.split('\n').filter(Boolean).map((f) => path.join(root, f.replace(/^\.\//, '')));
|
|
1565
|
+
return { files, truncated: false };
|
|
1566
|
+
}
|
|
1567
|
+
// Anything else (timeout, overflow, spawn failure): fall through to the walk.
|
|
1568
|
+
}
|
|
1569
|
+
const files = [];
|
|
1570
|
+
let truncated = false;
|
|
1571
|
+
walkDir(root, (f) => {
|
|
1572
|
+
if (files.length >= WALK_MAX_FILES) {
|
|
1573
|
+
truncated = true;
|
|
1574
|
+
return false;
|
|
1575
|
+
}
|
|
1576
|
+
files.push(f);
|
|
1577
|
+
return true;
|
|
1578
|
+
});
|
|
1579
|
+
return { files, truncated };
|
|
1580
|
+
}
|
|
1581
|
+
// callback returning `false` STOPS the walk (a cap that only stopped collecting would
|
|
1582
|
+
// still traverse the whole tree, so it would not bound the time at all).
|
|
1374
1583
|
function walkDir(dirPath, callback, _visited = new Set()) {
|
|
1375
1584
|
try {
|
|
1376
1585
|
// Resolve symlinks to detect cycles — a symlink pointing to a parent dir
|
|
1377
1586
|
// would cause infinite recursion without this guard.
|
|
1378
1587
|
const real = fs.realpathSync(dirPath);
|
|
1379
1588
|
if (_visited.has(real))
|
|
1380
|
-
return;
|
|
1589
|
+
return true;
|
|
1381
1590
|
_visited.add(real);
|
|
1382
1591
|
const entries = fs.readdirSync(dirPath, { withFileTypes: true });
|
|
1383
1592
|
for (const entry of entries) {
|
|
@@ -1394,18 +1603,20 @@ function walkDir(dirPath, callback, _visited = new Set()) {
|
|
|
1394
1603
|
})());
|
|
1395
1604
|
if (isDir) {
|
|
1396
1605
|
// Skip common noise directories
|
|
1397
|
-
if (
|
|
1606
|
+
if (WALK_SKIP_DIRS.has(entry.name))
|
|
1398
1607
|
continue;
|
|
1399
|
-
walkDir(fullPath, callback, _visited)
|
|
1608
|
+
if (walkDir(fullPath, callback, _visited) === false)
|
|
1609
|
+
return false;
|
|
1400
1610
|
}
|
|
1401
|
-
else {
|
|
1402
|
-
|
|
1611
|
+
else if (callback(fullPath) === false) {
|
|
1612
|
+
return false;
|
|
1403
1613
|
}
|
|
1404
1614
|
}
|
|
1405
1615
|
}
|
|
1406
1616
|
catch {
|
|
1407
1617
|
// Skip unreadable directories
|
|
1408
1618
|
}
|
|
1619
|
+
return true;
|
|
1409
1620
|
}
|
|
1410
1621
|
async function createDirectory(input, workDir) {
|
|
1411
1622
|
const dirPath = typeof input.path === 'string' ? input.path : '';
|
|
@@ -1637,7 +1848,7 @@ async function editFile(input, workDir) {
|
|
|
1637
1848
|
const diff = buildDiff(filePath, oldNorm, newNorm, origNorm);
|
|
1638
1849
|
const linesBefore = origNorm.split('\n').length;
|
|
1639
1850
|
const linesAfter = updated.split('\n').length;
|
|
1640
|
-
const xfile = crossFileBreakageWarning(resolved, origNorm, normalizeLF(updated), workDir ?? process.cwd());
|
|
1851
|
+
const xfile = await crossFileBreakageWarning(resolved, origNorm, normalizeLF(updated), workDir ?? process.cwd());
|
|
1641
1852
|
// Security-lint only the NEWLY INSERTED text, not the whole file. Scanning the
|
|
1642
1853
|
// full file would re-report pre-existing findings on every unrelated edit —
|
|
1643
1854
|
// noise that has nothing to do with the change being made, and the fastest way
|
|
@@ -1845,7 +2056,49 @@ function fetchBlocklistCheck(parsedUrl) {
|
|
|
1845
2056
|
}
|
|
1846
2057
|
return null;
|
|
1847
2058
|
}
|
|
1848
|
-
|
|
2059
|
+
/**
|
|
2060
|
+
* fetch_url as the model sees it. With `prompt`, the page (fetched HERE, so localhost
|
|
2061
|
+
* and intranet docs work and the server is never an open proxy) is answered by a cheap
|
|
2062
|
+
* model server-side (/api/code/tools/web_fetch_extract — DeepSeek Flash by default) and
|
|
2063
|
+
* only that answer is returned: Claude Code's WebFetch. Any failure of the extraction
|
|
2064
|
+
* step falls back to the ordinary capped page text, so the call never gets worse.
|
|
2065
|
+
*/
|
|
2066
|
+
async function fetchUrlTool(input, workDir, abortSignal) {
|
|
2067
|
+
const prompt = typeof input.prompt === 'string' ? input.prompt.trim() : '';
|
|
2068
|
+
if (!prompt)
|
|
2069
|
+
return fetchUrl(input, workDir, 0, abortSignal);
|
|
2070
|
+
const capture = {};
|
|
2071
|
+
const raw = await fetchUrl(input, workDir, 0, abortSignal, capture);
|
|
2072
|
+
if (raw.error !== undefined || !capture.text?.trim())
|
|
2073
|
+
return raw;
|
|
2074
|
+
const token = (0, auth_1.getToken)();
|
|
2075
|
+
if (!token)
|
|
2076
|
+
return raw;
|
|
2077
|
+
try {
|
|
2078
|
+
const r = await _httpsPost(`${client_2.API_BASE}/api/code/tools/web_fetch_extract`, JSON.stringify({ url: capture.url ?? input.url, content: capture.text, prompt, ...(0, client_1.billingOverrideFields)() }), { 'Content-Type': 'application/json', Authorization: `Bearer ${token}` }, 90000, abortSignal);
|
|
2079
|
+
let json = {};
|
|
2080
|
+
try {
|
|
2081
|
+
json = JSON.parse(r.body);
|
|
2082
|
+
}
|
|
2083
|
+
catch { /* fall through */ }
|
|
2084
|
+
if (r.status !== 200 || !json.text) {
|
|
2085
|
+
return { ...raw, output: `${raw.output ?? ''}\n\n[prompt extraction unavailable (${json.error ?? `HTTP ${r.status}`}) — raw page text shown instead]` };
|
|
2086
|
+
}
|
|
2087
|
+
return {
|
|
2088
|
+
output: `[Answer extracted from ${capture.url ?? input.url} by ${json.model ?? 'a low-cost model'} for: "${prompt.slice(0, 200)}"]\n\n` +
|
|
2089
|
+
`<untrusted_web_content url="${capture.url ?? input.url}">\n${json.text}\n</untrusted_web_content>`,
|
|
2090
|
+
};
|
|
2091
|
+
}
|
|
2092
|
+
catch (err) {
|
|
2093
|
+
if (err.name === 'AbortError')
|
|
2094
|
+
return { error: 'fetch_url stopped by user', interrupted: true };
|
|
2095
|
+
return { ...raw, output: `${raw.output ?? ''}\n\n[prompt extraction failed (${err.message}) — raw page text shown instead]` };
|
|
2096
|
+
}
|
|
2097
|
+
}
|
|
2098
|
+
async function fetchUrl(input, _workDir, _redirectCount = 0, abortSignal,
|
|
2099
|
+
// Receives the FULL stripped page text (before the 40K model-facing cap), for the
|
|
2100
|
+
// prompt-extraction path in fetchUrlTool. Threaded through redirects.
|
|
2101
|
+
capture) {
|
|
1849
2102
|
const url = typeof input.url === 'string' ? input.url : '';
|
|
1850
2103
|
if (!url)
|
|
1851
2104
|
return { error: 'Missing required parameter: url' };
|
|
@@ -1947,7 +2200,7 @@ async function fetchUrl(input, _workDir, _redirectCount = 0, abortSignal) {
|
|
|
1947
2200
|
finish({ error: `Invalid redirect location: ${res.headers.location}` });
|
|
1948
2201
|
return;
|
|
1949
2202
|
}
|
|
1950
|
-
fetchUrl({ url: nextUrl }, undefined, _redirectCount + 1, abortSignal).then(finish);
|
|
2203
|
+
fetchUrl({ url: nextUrl }, undefined, _redirectCount + 1, abortSignal, capture).then(finish);
|
|
1951
2204
|
return;
|
|
1952
2205
|
}
|
|
1953
2206
|
const contentType = res.headers['content-type'] ?? '';
|
|
@@ -1961,7 +2214,15 @@ async function fetchUrl(input, _workDir, _redirectCount = 0, abortSignal) {
|
|
|
1961
2214
|
let body = Buffer.concat(chunks).toString('utf-8');
|
|
1962
2215
|
if (isHtml)
|
|
1963
2216
|
body = stripHtml(body);
|
|
1964
|
-
|
|
2217
|
+
if (capture) {
|
|
2218
|
+
capture.text = body;
|
|
2219
|
+
capture.url = url;
|
|
2220
|
+
}
|
|
2221
|
+
let truncNote = truncated ? `\n\n[Truncated at ${MAX_FETCH_BYTES / 1024}KB]` : '';
|
|
2222
|
+
if (body.length > MAX_FETCH_OUTPUT_CHARS) {
|
|
2223
|
+
body = (0, safeSlice_1.sliceSafeEnd)(body, MAX_FETCH_OUTPUT_CHARS);
|
|
2224
|
+
truncNote = `\n\n[Truncated at ${MAX_FETCH_OUTPUT_CHARS.toLocaleString('en-US')} chars of page text]`;
|
|
2225
|
+
}
|
|
1965
2226
|
// Prompt-injection mitigation: fetched web content is fully attacker-controlled
|
|
1966
2227
|
// (anyone can put "ignore previous instructions..." on a page). Wrap it in an
|
|
1967
2228
|
// explicit untrusted-data delimiter so the model treats it as DATA to read, not
|
|
@@ -2083,7 +2344,7 @@ async function multiEdit(input, workDir) {
|
|
|
2083
2344
|
// preserving the original mode bits (+x on scripts etc.).
|
|
2084
2345
|
const finalContent = wasCRLF ? content.replace(/\n/g, '\r\n') : content;
|
|
2085
2346
|
atomicWritePreservingMode(resolved, finalContent, pre.mode);
|
|
2086
|
-
const xfile = crossFileBreakageWarning(resolved, originalNorm, content, workDir ?? process.cwd());
|
|
2347
|
+
const xfile = await crossFileBreakageWarning(resolved, originalNorm, content, workDir ?? process.cwd());
|
|
2087
2348
|
// Lint only the newly inserted text (see the same reasoning in editFile).
|
|
2088
2349
|
const sec = (0, securityLint_1.securityNoteText)((0, securityLint_1.checkSecurity)(insertedText.join('\n')));
|
|
2089
2350
|
return {
|
|
@@ -2106,38 +2367,97 @@ async function glob(input, workDir) {
|
|
|
2106
2367
|
try {
|
|
2107
2368
|
const resolved = resolvePath(searchDir, workDir);
|
|
2108
2369
|
const matches = [];
|
|
2109
|
-
|
|
2110
|
-
|
|
2111
|
-
|
|
2112
|
-
|
|
2113
|
-
|
|
2114
|
-
|
|
2115
|
-
|
|
2116
|
-
.replace(/\?/g, '[^/]'); // ? matches single non-slash char
|
|
2117
|
-
return new RegExp(`^${re}$`);
|
|
2118
|
-
};
|
|
2119
|
-
const regex = toRegex(pattern);
|
|
2120
|
-
walkDir(resolved, (filePath) => {
|
|
2370
|
+
const regex = globToRegex(pattern);
|
|
2371
|
+
// Only fall back to basename test for patterns without a path separator
|
|
2372
|
+
// (i.e. single-segment patterns like "*.ts"). Patterns with "/" (e.g.
|
|
2373
|
+
// "src/*.ts") must match the full relative path to avoid false positives.
|
|
2374
|
+
const useBasename = !pattern.includes('/') && !pattern.includes('**');
|
|
2375
|
+
const listing = await listFiles(resolved);
|
|
2376
|
+
for (const filePath of listing.files) {
|
|
2121
2377
|
const rel = path.relative(resolved, filePath).replace(/\\/g, '/');
|
|
2122
|
-
// Only fall back to basename test for patterns without a path separator
|
|
2123
|
-
// (i.e. single-segment patterns like "*.ts"). Patterns with "/" (e.g.
|
|
2124
|
-
// "src/*.ts") must match the full relative path to avoid false positives.
|
|
2125
|
-
const useBasename = !pattern.includes('/') && !pattern.includes('**');
|
|
2126
2378
|
if (regex.test(rel) || (useBasename && regex.test(path.basename(filePath)))) {
|
|
2127
2379
|
matches.push(filePath);
|
|
2128
2380
|
}
|
|
2129
|
-
}
|
|
2381
|
+
}
|
|
2382
|
+
const scanNote = listing.truncated
|
|
2383
|
+
? `\n[Scan stopped after ${WALK_MAX_FILES.toLocaleString('en-US')} files — narrow the path for complete results.]`
|
|
2384
|
+
: '';
|
|
2130
2385
|
if (!matches.length)
|
|
2131
|
-
return { output: `No files matched pattern: ${pattern}` };
|
|
2132
|
-
|
|
2133
|
-
|
|
2134
|
-
|
|
2135
|
-
|
|
2386
|
+
return { output: `No files matched pattern: ${pattern}${scanNote}` };
|
|
2387
|
+
// Most recently modified first, capped — same shape as Claude Code's Glob. Paths are
|
|
2388
|
+
// relative to the search dir: absolute prefixes were ~25% of the output's tokens.
|
|
2389
|
+
const mtime = (f) => { try {
|
|
2390
|
+
return fs.statSync(f).mtimeMs;
|
|
2391
|
+
}
|
|
2392
|
+
catch {
|
|
2393
|
+
return 0;
|
|
2394
|
+
} };
|
|
2395
|
+
const sorted = matches.map((f) => ({ f, t: mtime(f) })).sort((a, b) => b.t - a.t).map((x) => x.f);
|
|
2396
|
+
const shown = sorted.slice(0, MAX_GLOB_RESULTS).map((f) => path.relative(resolved, f).replace(/\\/g, '/'));
|
|
2397
|
+
const more = sorted.length > shown.length
|
|
2398
|
+
? `\n[${sorted.length - shown.length} more not shown — narrow the pattern or path]`
|
|
2399
|
+
: '';
|
|
2400
|
+
return { output: `${matches.length} file(s) matched "${pattern}" under ${resolved}:\n${shown.join('\n')}${more}${scanNote}` };
|
|
2136
2401
|
}
|
|
2137
2402
|
catch (err) {
|
|
2138
2403
|
return { error: err.message };
|
|
2139
2404
|
}
|
|
2140
2405
|
}
|
|
2406
|
+
/**
|
|
2407
|
+
* Glob → anchored RegExp. Scans the pattern once, so the `*` / `?` rules can never
|
|
2408
|
+
* rewrite the regex emitted for `**` (the old chained .replace() calls turned
|
|
2409
|
+
* `**\/` into `([^/]:[^/]+/)` — every `**` pattern silently matched nothing).
|
|
2410
|
+
* Supports `**`, `*`, `?`, `{a,b}` and `[...]` classes.
|
|
2411
|
+
*/
|
|
2412
|
+
function globToRegex(pattern) {
|
|
2413
|
+
let re = '';
|
|
2414
|
+
let inBrace = 0;
|
|
2415
|
+
for (let i = 0; i < pattern.length; i++) {
|
|
2416
|
+
const c = pattern[i];
|
|
2417
|
+
if (c === '*') {
|
|
2418
|
+
if (pattern[i + 1] === '*') {
|
|
2419
|
+
const slash = pattern[i + 2] === '/';
|
|
2420
|
+
re += slash ? '(?:[^/]*/)*' : '.*';
|
|
2421
|
+
i += slash ? 2 : 1;
|
|
2422
|
+
}
|
|
2423
|
+
else {
|
|
2424
|
+
re += '[^/]*';
|
|
2425
|
+
}
|
|
2426
|
+
}
|
|
2427
|
+
else if (c === '?') {
|
|
2428
|
+
re += '[^/]';
|
|
2429
|
+
}
|
|
2430
|
+
else if (c === '{') {
|
|
2431
|
+
inBrace++;
|
|
2432
|
+
re += '(?:';
|
|
2433
|
+
}
|
|
2434
|
+
else if (c === '}' && inBrace > 0) {
|
|
2435
|
+
inBrace--;
|
|
2436
|
+
re += ')';
|
|
2437
|
+
}
|
|
2438
|
+
else if (c === ',' && inBrace > 0) {
|
|
2439
|
+
re += '|';
|
|
2440
|
+
}
|
|
2441
|
+
else if (c === '[') {
|
|
2442
|
+
const end = pattern.indexOf(']', i + 1);
|
|
2443
|
+
if (end > i) {
|
|
2444
|
+
let cls = pattern.slice(i + 1, end).replace(/\\/g, '\\\\');
|
|
2445
|
+
if (cls.startsWith('!'))
|
|
2446
|
+
cls = '^' + cls.slice(1);
|
|
2447
|
+
re += `[${cls}]`;
|
|
2448
|
+
i = end;
|
|
2449
|
+
}
|
|
2450
|
+
else {
|
|
2451
|
+
re += '\\[';
|
|
2452
|
+
}
|
|
2453
|
+
}
|
|
2454
|
+
else {
|
|
2455
|
+
re += c.replace(/[.+^$()|[\]\\{}]/g, '\\$&');
|
|
2456
|
+
}
|
|
2457
|
+
}
|
|
2458
|
+
re += ')'.repeat(inBrace); // an unclosed `{` must not throw "Invalid regular expression"
|
|
2459
|
+
return new RegExp(`^${re}$`);
|
|
2460
|
+
}
|
|
2141
2461
|
const _todoScopes = new Map();
|
|
2142
2462
|
function todoScope(key) {
|
|
2143
2463
|
const k = key || 'default';
|
|
@@ -2396,7 +2716,7 @@ async function notebookRead(input, workDir) {
|
|
|
2396
2716
|
});
|
|
2397
2717
|
let output = lines.join('\n');
|
|
2398
2718
|
if (output.length > MAX_OUTPUT_CHARS) {
|
|
2399
|
-
output = (0, safeSlice_1.sliceSafeEnd)(output, MAX_OUTPUT_CHARS) +
|
|
2719
|
+
output = (0, safeSlice_1.sliceSafeEnd)(output, MAX_OUTPUT_CHARS) + `\n\n[Notebook output truncated at ${MAX_OUTPUT_CHARS.toLocaleString('en-US')} chars]`;
|
|
2400
2720
|
}
|
|
2401
2721
|
return { output };
|
|
2402
2722
|
}
|
|
@@ -2493,7 +2813,10 @@ function _httpsPost(url, body, headers, timeoutMs = 30000, abortSignal) {
|
|
|
2493
2813
|
const done = (fn) => { if (settled)
|
|
2494
2814
|
return; settled = true; if (abortPoll)
|
|
2495
2815
|
clearInterval(abortPoll); fn(); };
|
|
2496
|
-
|
|
2816
|
+
// http for an http:// API base (local dev / tests) — https.request would reject it —
|
|
2817
|
+
// and the explicit port, which was silently dropped (so :3000 went to :443).
|
|
2818
|
+
const mod = u.protocol === 'http:' ? http : https;
|
|
2819
|
+
const req = mod.request({ hostname: u.hostname, ...(u.port ? { port: Number(u.port) } : {}), path: u.pathname + u.search, method: 'POST', headers: { ...headers, 'Content-Length': Buffer.byteLength(body) }, timeout: timeoutMs }, (res) => {
|
|
2497
2820
|
res.on('data', (c) => chunks.push(c));
|
|
2498
2821
|
res.on('end', () => done(() => resolve({ status: res.statusCode ?? 0, body: Buffer.concat(chunks).toString('utf-8') })));
|
|
2499
2822
|
});
|
|
@@ -2839,7 +3162,7 @@ const TOOL_MAP = {
|
|
|
2839
3162
|
copy_file: copyFile,
|
|
2840
3163
|
move_file: moveFile,
|
|
2841
3164
|
delete_file: deleteFile,
|
|
2842
|
-
fetch_url:
|
|
3165
|
+
fetch_url: (input) => fetchUrlTool(input),
|
|
2843
3166
|
web_search: webSearch,
|
|
2844
3167
|
generate_image: generateImage,
|
|
2845
3168
|
stock_photo: stockPhoto,
|
|
@@ -2921,7 +3244,7 @@ selfPeer) {
|
|
|
2921
3244
|
// silently making the user wait out their own timeout (up to 170s for
|
|
2922
3245
|
// generate_image) instead of returning control immediately.
|
|
2923
3246
|
if (name === 'fetch_url')
|
|
2924
|
-
return await
|
|
3247
|
+
return await fetchUrlTool(input, workDir, abortSignal);
|
|
2925
3248
|
if (name === 'web_search')
|
|
2926
3249
|
return await webSearch(input, workDir, abortSignal);
|
|
2927
3250
|
if (name === 'generate_image')
|