@n0zer0d4y/vulcan-file-ops 1.2.13 → 1.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +99 -27
- package/README.md +153 -118
- package/dist/cli.js +16 -0
- package/dist/server/index.js +150 -39
- package/dist/tools/filesystem-tools.js +222 -39
- package/dist/tools/read-tools.js +89 -16
- package/dist/tools/shell-tool.js +48 -62
- package/dist/tools/write-tools.js +19 -17
- package/dist/types/index.js +6 -3
- package/dist/utils/command-path-extraction.js +169 -174
- package/dist/utils/command-validation.js +62 -10
- package/dist/utils/document-parser.js +43 -5
- package/dist/utils/html-image-sanitizer.js +478 -0
- package/dist/utils/html-to-document.js +75 -11
- package/dist/utils/lib.js +357 -80
- package/dist/utils/limits.js +47 -0
- package/dist/utils/regex-worker.js +262 -0
- package/dist/utils/shell-parser.js +225 -0
- package/dist/utils/zip-guard.js +266 -0
- package/package.json +11 -8
package/dist/utils/lib.js
CHANGED
|
@@ -5,6 +5,8 @@ import { createTwoFilesPatch } from "diff";
|
|
|
5
5
|
import { minimatch } from "minimatch";
|
|
6
6
|
import { normalizePath, expandHome } from "./path-utils.js";
|
|
7
7
|
import { isPathWithinAllowedDirectories } from "./path-validation.js";
|
|
8
|
+
import { RegexEvaluationSession, RegexEvaluationError, DEFAULT_GREP_REGEX_FILE_TIMEOUT_MS, DEFAULT_GREP_REGEX_TOTAL_TIMEOUT_MS, } from "./regex-worker.js";
|
|
9
|
+
import { MAX_TEXT_READ_BYTES } from "./limits.js";
|
|
8
10
|
// Global configuration - set by the main module
|
|
9
11
|
let allowedDirectories = [];
|
|
10
12
|
let ignoredFolders = [];
|
|
@@ -304,6 +306,12 @@ export async function validatePath(requestedPath, options) {
|
|
|
304
306
|
return absolute;
|
|
305
307
|
}
|
|
306
308
|
catch (parentError) {
|
|
309
|
+
// Only a missing parent is "does not exist"; an access-denied result
|
|
310
|
+
// (e.g. the parent resolves outside the allowed directories) must keep
|
|
311
|
+
// its own message instead of being reported as missing.
|
|
312
|
+
if (!isNotFoundError(parentError)) {
|
|
313
|
+
throw parentError;
|
|
314
|
+
}
|
|
307
315
|
if (!createParentIfMissing) {
|
|
308
316
|
throw new Error(`Parent directory does not exist: ${parentDir}`);
|
|
309
317
|
}
|
|
@@ -318,6 +326,161 @@ export async function validatePath(requestedPath, options) {
|
|
|
318
326
|
throw error;
|
|
319
327
|
}
|
|
320
328
|
}
|
|
329
|
+
// Canonical containment helpers
|
|
330
|
+
//
|
|
331
|
+
// isPathWithinAllowedDirectories() is a purely lexical check. Paths that are
|
|
332
|
+
// handed to something other than Node's fs (e.g. a shell command line) or that
|
|
333
|
+
// are created with mkdir must also be checked after symlink resolution.
|
|
334
|
+
function isNotFoundError(error) {
|
|
335
|
+
const code = error.code;
|
|
336
|
+
return code === "ENOENT" || code === "ENOTDIR";
|
|
337
|
+
}
|
|
338
|
+
/**
|
|
339
|
+
* Lexical normalization first, then realpath of the nearest existing ancestor
|
|
340
|
+
* with the non-existent remainder re-appended. Matches Win32 and Node fs
|
|
341
|
+
* semantics, where ".." is collapsed before symlinks are followed.
|
|
342
|
+
*/
|
|
343
|
+
export async function resolveLexicalCanonicalPath(absolutePath) {
|
|
344
|
+
const resolved = path.resolve(absolutePath);
|
|
345
|
+
const tail = [];
|
|
346
|
+
let current = resolved;
|
|
347
|
+
while (true) {
|
|
348
|
+
try {
|
|
349
|
+
const real = await fs.realpath(current);
|
|
350
|
+
return tail.length > 0 ? path.join(real, ...tail.reverse()) : real;
|
|
351
|
+
}
|
|
352
|
+
catch (error) {
|
|
353
|
+
if (!isNotFoundError(error)) {
|
|
354
|
+
throw error;
|
|
355
|
+
}
|
|
356
|
+
const parent = path.dirname(current);
|
|
357
|
+
if (parent === current) {
|
|
358
|
+
return resolved;
|
|
359
|
+
}
|
|
360
|
+
tail.push(path.basename(current));
|
|
361
|
+
current = parent;
|
|
362
|
+
}
|
|
363
|
+
}
|
|
364
|
+
}
|
|
365
|
+
/**
|
|
366
|
+
* Component-by-component resolution that follows symlinks before applying
|
|
367
|
+
* "..", matching how a POSIX kernel resolves the path a shell passes to it.
|
|
368
|
+
* On Windows this is stricter than the OS (which collapses ".." first), which
|
|
369
|
+
* only ever produces extra denials, never extra access.
|
|
370
|
+
*/
|
|
371
|
+
export async function resolvePhysicalCanonicalPath(inputPath, baseDir) {
|
|
372
|
+
const absolute = path.isAbsolute(inputPath)
|
|
373
|
+
? inputPath
|
|
374
|
+
: `${baseDir}${path.sep}${inputPath}`;
|
|
375
|
+
const rawRoot = path.parse(absolute).root;
|
|
376
|
+
const root = path.parse(path.resolve(absolute)).root;
|
|
377
|
+
const separator = path.sep === "\\" ? /[\\/]+/ : /\/+/;
|
|
378
|
+
const components = absolute.slice(rawRoot.length).split(separator);
|
|
379
|
+
let current = root;
|
|
380
|
+
for (const component of components) {
|
|
381
|
+
if (!component || component === ".") {
|
|
382
|
+
continue;
|
|
383
|
+
}
|
|
384
|
+
if (component === "..") {
|
|
385
|
+
current = path.dirname(current);
|
|
386
|
+
continue;
|
|
387
|
+
}
|
|
388
|
+
const next = path.join(current, component);
|
|
389
|
+
try {
|
|
390
|
+
current = await fs.realpath(next);
|
|
391
|
+
}
|
|
392
|
+
catch (error) {
|
|
393
|
+
if (!isNotFoundError(error)) {
|
|
394
|
+
throw error;
|
|
395
|
+
}
|
|
396
|
+
current = next;
|
|
397
|
+
}
|
|
398
|
+
}
|
|
399
|
+
return current;
|
|
400
|
+
}
|
|
401
|
+
/**
|
|
402
|
+
* True only if the path is inside the allowed directories lexically, after
|
|
403
|
+
* lexical-then-realpath resolution, and after physical resolution. Relative
|
|
404
|
+
* paths are resolved against baseDir. Any resolution error denies access.
|
|
405
|
+
*/
|
|
406
|
+
export async function isPathCanonicallyAllowed(inputPath, baseDir) {
|
|
407
|
+
const allowed = getAllowedDirectories();
|
|
408
|
+
if (allowed.length === 0 || !inputPath || inputPath.includes("\x00")) {
|
|
409
|
+
return false;
|
|
410
|
+
}
|
|
411
|
+
try {
|
|
412
|
+
const absolute = path.isAbsolute(inputPath)
|
|
413
|
+
? path.resolve(inputPath)
|
|
414
|
+
: path.resolve(baseDir, inputPath);
|
|
415
|
+
if (!isPathWithinAllowedDirectories(normalizePath(absolute), allowed)) {
|
|
416
|
+
return false;
|
|
417
|
+
}
|
|
418
|
+
const lexical = await resolveLexicalCanonicalPath(absolute);
|
|
419
|
+
if (!isPathWithinAllowedDirectories(normalizePath(lexical), allowed)) {
|
|
420
|
+
return false;
|
|
421
|
+
}
|
|
422
|
+
const physical = await resolvePhysicalCanonicalPath(inputPath, baseDir);
|
|
423
|
+
return isPathWithinAllowedDirectories(normalizePath(physical), allowed);
|
|
424
|
+
}
|
|
425
|
+
catch {
|
|
426
|
+
return false;
|
|
427
|
+
}
|
|
428
|
+
}
|
|
429
|
+
/**
|
|
430
|
+
* Create a directory (and any missing ancestors) only inside the allowed
|
|
431
|
+
* directories. Every directory created is re-checked with realpath so a
|
|
432
|
+
* symlink or junction swapped into the chain cannot redirect creation.
|
|
433
|
+
* Returns the real path of the directory.
|
|
434
|
+
*/
|
|
435
|
+
export async function ensureDirectoryWithinAllowed(dirPath) {
|
|
436
|
+
const absolute = path.resolve(expandHome(dirPath));
|
|
437
|
+
const allowed = getAllowedDirectories();
|
|
438
|
+
if (!(await isPathCanonicallyAllowed(absolute, process.cwd()))) {
|
|
439
|
+
throw new Error(`Access denied - path outside allowed directories: ${absolute} not in ${allowed.join(", ")}`);
|
|
440
|
+
}
|
|
441
|
+
const missing = [];
|
|
442
|
+
let current = absolute;
|
|
443
|
+
while (true) {
|
|
444
|
+
try {
|
|
445
|
+
await fs.lstat(current);
|
|
446
|
+
break;
|
|
447
|
+
}
|
|
448
|
+
catch (error) {
|
|
449
|
+
if (!isNotFoundError(error)) {
|
|
450
|
+
throw error;
|
|
451
|
+
}
|
|
452
|
+
missing.push(current);
|
|
453
|
+
const parent = path.dirname(current);
|
|
454
|
+
if (parent === current) {
|
|
455
|
+
break;
|
|
456
|
+
}
|
|
457
|
+
current = parent;
|
|
458
|
+
}
|
|
459
|
+
}
|
|
460
|
+
for (const dir of missing.reverse()) {
|
|
461
|
+
try {
|
|
462
|
+
await fs.mkdir(dir);
|
|
463
|
+
}
|
|
464
|
+
catch (error) {
|
|
465
|
+
if (error.code !== "EEXIST") {
|
|
466
|
+
throw error;
|
|
467
|
+
}
|
|
468
|
+
}
|
|
469
|
+
const realCreated = await fs.realpath(dir);
|
|
470
|
+
if (!isPathWithinAllowedDirectories(normalizePath(realCreated), getAllowedDirectories())) {
|
|
471
|
+
throw new Error(`Access denied - directory resolves outside allowed directories: ${realCreated}`);
|
|
472
|
+
}
|
|
473
|
+
}
|
|
474
|
+
const realFinal = await fs.realpath(absolute);
|
|
475
|
+
if (!isPathWithinAllowedDirectories(normalizePath(realFinal), getAllowedDirectories())) {
|
|
476
|
+
throw new Error(`Access denied - directory resolves outside allowed directories: ${realFinal}`);
|
|
477
|
+
}
|
|
478
|
+
const stats = await fs.stat(realFinal);
|
|
479
|
+
if (!stats.isDirectory()) {
|
|
480
|
+
throw new Error(`Path exists and is not a directory: ${absolute}`);
|
|
481
|
+
}
|
|
482
|
+
return realFinal;
|
|
483
|
+
}
|
|
321
484
|
// File Operations
|
|
322
485
|
export async function getFileStats(filePath) {
|
|
323
486
|
const stats = await fs.stat(filePath);
|
|
@@ -335,10 +498,21 @@ export async function readFileContent(filePath, encoding = "utf-8") {
|
|
|
335
498
|
return await fs.readFile(filePath, encoding);
|
|
336
499
|
}
|
|
337
500
|
export async function writeFileContent(filePath, content) {
|
|
501
|
+
await writeFileAtomic(filePath, content);
|
|
502
|
+
}
|
|
503
|
+
/**
|
|
504
|
+
* Binary counterpart of writeFileContent (PDF/DOCX output) with the same
|
|
505
|
+
* symlink-safe, atomic write semantics.
|
|
506
|
+
*/
|
|
507
|
+
export async function writeBinaryFileAtomic(filePath, data) {
|
|
508
|
+
await writeFileAtomic(filePath, data);
|
|
509
|
+
}
|
|
510
|
+
async function writeFileAtomic(filePath, data) {
|
|
511
|
+
const encoding = typeof data === "string" ? "utf-8" : undefined;
|
|
338
512
|
try {
|
|
339
513
|
// Security: 'wx' flag ensures exclusive creation - fails if file/symlink exists,
|
|
340
514
|
// preventing writes through pre-existing symlinks
|
|
341
|
-
await fs.writeFile(filePath,
|
|
515
|
+
await fs.writeFile(filePath, data, { encoding, flag: "wx" });
|
|
342
516
|
}
|
|
343
517
|
catch (error) {
|
|
344
518
|
if (error.code === "EEXIST") {
|
|
@@ -347,7 +521,7 @@ export async function writeFileContent(filePath, content) {
|
|
|
347
521
|
// replace the target file atomically and don't follow symlinks.
|
|
348
522
|
const tempPath = `${filePath}.${randomBytes(16).toString("hex")}.tmp`;
|
|
349
523
|
try {
|
|
350
|
-
await fs.writeFile(tempPath,
|
|
524
|
+
await fs.writeFile(tempPath, data, { encoding, flag: "wx" });
|
|
351
525
|
await fs.rename(tempPath, filePath);
|
|
352
526
|
}
|
|
353
527
|
catch (renameError) {
|
|
@@ -651,6 +825,17 @@ export async function applyFileEdits(filePath, edits, dryRun = false, matchingSt
|
|
|
651
825
|
}
|
|
652
826
|
return formattedDiff;
|
|
653
827
|
}
|
|
828
|
+
/**
|
|
829
|
+
* VFO-15: head/tail/range reads stream the file, but the text they hold
|
|
830
|
+
* (collected lines plus the current partial line) must stay bounded, e.g.
|
|
831
|
+
* for a file consisting of one huge line.
|
|
832
|
+
*/
|
|
833
|
+
function assertWithinTextReadLimit(heldChars) {
|
|
834
|
+
if (heldChars > MAX_TEXT_READ_BYTES) {
|
|
835
|
+
throw new Error(`Requested lines exceed the ${(MAX_TEXT_READ_BYTES / 1024 / 1024).toFixed(1)} MB ` +
|
|
836
|
+
`read limit (the file may contain very long lines). Request fewer lines.`);
|
|
837
|
+
}
|
|
838
|
+
}
|
|
654
839
|
// Memory-efficient implementation to get the last N lines of a file
|
|
655
840
|
export async function tailFile(filePath, numLines) {
|
|
656
841
|
const CHUNK_SIZE = 1024; // Read 1KB at a time
|
|
@@ -666,6 +851,10 @@ export async function tailFile(filePath, numLines) {
|
|
|
666
851
|
let chunk = Buffer.alloc(CHUNK_SIZE);
|
|
667
852
|
let linesFound = 0;
|
|
668
853
|
let remainingText = "";
|
|
854
|
+
let linesLength = 0;
|
|
855
|
+
// The newline that ends the last line does not start another (empty)
|
|
856
|
+
// line, so tail of "a\nb\n" with 1 line is "b", as with POSIX tail.
|
|
857
|
+
let atFileEnd = true;
|
|
669
858
|
// Read chunks from the end of the file until we have enough lines
|
|
670
859
|
while (position > 0 && linesFound < numLines) {
|
|
671
860
|
const size = Math.min(CHUNK_SIZE, position);
|
|
@@ -675,9 +864,22 @@ export async function tailFile(filePath, numLines) {
|
|
|
675
864
|
break;
|
|
676
865
|
// Get the chunk as a string and prepend any remaining text from previous iteration
|
|
677
866
|
const readData = chunk.slice(0, bytesRead).toString("utf-8");
|
|
867
|
+
// VFO-15: bound memory held for one (partial) line plus the output,
|
|
868
|
+
// and avoid re-splitting an ever-growing line on every chunk.
|
|
869
|
+
if (position > 0 && !/[\r\n]/.test(readData)) {
|
|
870
|
+
remainingText = readData + remainingText;
|
|
871
|
+
assertWithinTextReadLimit(remainingText.length + linesLength);
|
|
872
|
+
continue;
|
|
873
|
+
}
|
|
678
874
|
const chunkText = readData + remainingText;
|
|
679
875
|
// Split by newlines and count
|
|
680
876
|
const chunkLines = normalizeLineEndings(chunkText).split("\n");
|
|
877
|
+
if (atFileEnd) {
|
|
878
|
+
if (chunkLines.length > 1 && chunkLines[chunkLines.length - 1] === "") {
|
|
879
|
+
chunkLines.pop();
|
|
880
|
+
}
|
|
881
|
+
atFileEnd = false;
|
|
882
|
+
}
|
|
681
883
|
// If this isn't the end of the file, the first line is likely incomplete
|
|
682
884
|
// Save it to prepend to the next chunk
|
|
683
885
|
if (position > 0) {
|
|
@@ -688,7 +890,9 @@ export async function tailFile(filePath, numLines) {
|
|
|
688
890
|
for (let i = chunkLines.length - 1; i >= 0 && linesFound < numLines; i--) {
|
|
689
891
|
lines.unshift(chunkLines[i]);
|
|
690
892
|
linesFound++;
|
|
893
|
+
linesLength += chunkLines[i].length + 1;
|
|
691
894
|
}
|
|
895
|
+
assertWithinTextReadLimit(linesLength + remainingText.length);
|
|
692
896
|
}
|
|
693
897
|
return lines.join("\n");
|
|
694
898
|
}
|
|
@@ -703,6 +907,7 @@ export async function headFile(filePath, numLines) {
|
|
|
703
907
|
const lines = [];
|
|
704
908
|
let buffer = "";
|
|
705
909
|
let bytesRead = 0;
|
|
910
|
+
let linesLength = 0;
|
|
706
911
|
const chunk = Buffer.alloc(1024); // 1KB buffer
|
|
707
912
|
// Read chunks and count lines until we have enough or reach EOF
|
|
708
913
|
while (lines.length < numLines) {
|
|
@@ -710,17 +915,21 @@ export async function headFile(filePath, numLines) {
|
|
|
710
915
|
if (result.bytesRead === 0)
|
|
711
916
|
break; // End of file
|
|
712
917
|
bytesRead += result.bytesRead;
|
|
713
|
-
|
|
714
|
-
|
|
918
|
+
const text = chunk.slice(0, result.bytesRead).toString("utf-8");
|
|
919
|
+
buffer += text;
|
|
920
|
+
// Only rescan the buffer when the new chunk completes a line.
|
|
921
|
+
const newLineIndex = text.includes("\n") ? buffer.lastIndexOf("\n") : -1;
|
|
715
922
|
if (newLineIndex !== -1) {
|
|
716
923
|
const completeLines = buffer.slice(0, newLineIndex).split("\n");
|
|
717
924
|
buffer = buffer.slice(newLineIndex + 1);
|
|
718
925
|
for (const line of completeLines) {
|
|
719
926
|
lines.push(line);
|
|
927
|
+
linesLength += line.length + 1;
|
|
720
928
|
if (lines.length >= numLines)
|
|
721
929
|
break;
|
|
722
930
|
}
|
|
723
931
|
}
|
|
932
|
+
assertWithinTextReadLimit(linesLength + buffer.length);
|
|
724
933
|
}
|
|
725
934
|
// If there is leftover content and we still need lines, add it
|
|
726
935
|
if (buffer.length > 0 && lines.length < numLines) {
|
|
@@ -748,6 +957,7 @@ export async function rangeFile(filePath, startLine, endLine) {
|
|
|
748
957
|
let currentLineNumber = 0;
|
|
749
958
|
let buffer = "";
|
|
750
959
|
let bytesRead = 0;
|
|
960
|
+
let targetLength = 0;
|
|
751
961
|
const chunk = Buffer.alloc(CHUNK_SIZE);
|
|
752
962
|
// Read file sequentially until we reach the end line
|
|
753
963
|
while (currentLineNumber < endLine) {
|
|
@@ -764,7 +974,19 @@ export async function rangeFile(filePath, startLine, endLine) {
|
|
|
764
974
|
break;
|
|
765
975
|
}
|
|
766
976
|
bytesRead += result.bytesRead;
|
|
767
|
-
|
|
977
|
+
const text = chunk.slice(0, result.bytesRead).toString("utf-8");
|
|
978
|
+
if (!text.includes("\n")) {
|
|
979
|
+
if (currentLineNumber + 1 < startLine) {
|
|
980
|
+
// Middle of a line before the range: its content is never needed.
|
|
981
|
+
buffer = "";
|
|
982
|
+
}
|
|
983
|
+
else {
|
|
984
|
+
buffer += text;
|
|
985
|
+
assertWithinTextReadLimit(targetLength + buffer.length);
|
|
986
|
+
}
|
|
987
|
+
continue;
|
|
988
|
+
}
|
|
989
|
+
buffer += text;
|
|
768
990
|
// Process complete lines in buffer
|
|
769
991
|
let newLineIndex = buffer.indexOf("\n");
|
|
770
992
|
while (newLineIndex !== -1) {
|
|
@@ -774,6 +996,8 @@ export async function rangeFile(filePath, startLine, endLine) {
|
|
|
774
996
|
// Check if this line is within our target range
|
|
775
997
|
if (currentLineNumber >= startLine && currentLineNumber <= endLine) {
|
|
776
998
|
targetLines.push(line);
|
|
999
|
+
targetLength += line.length + 1;
|
|
1000
|
+
assertWithinTextReadLimit(targetLength);
|
|
777
1001
|
}
|
|
778
1002
|
// Early exit if we've collected all needed lines
|
|
779
1003
|
if (currentLineNumber >= endLine) {
|
|
@@ -792,9 +1016,30 @@ export async function rangeFile(filePath, startLine, endLine) {
|
|
|
792
1016
|
await fileHandle.close();
|
|
793
1017
|
}
|
|
794
1018
|
}
|
|
1019
|
+
/**
|
|
1020
|
+
* Maximum length of a grep_files regex pattern. Compiling is cheap, but long
|
|
1021
|
+
* model-supplied patterns only widen the backtracking attack surface (VFO-14).
|
|
1022
|
+
*/
|
|
1023
|
+
export const MAX_GREP_PATTERN_LENGTH = 1_000;
|
|
1024
|
+
/**
|
|
1025
|
+
* Maximum length of a glob pattern handed to minimatch (glob_files pattern,
|
|
1026
|
+
* excludePatterns, grep_files glob). minimatch has had ReDoS advisories; a
|
|
1027
|
+
* length cap is cheap defense in depth.
|
|
1028
|
+
*/
|
|
1029
|
+
export const MAX_GLOB_PATTERN_LENGTH = 1_000;
|
|
1030
|
+
export { DEFAULT_GREP_REGEX_FILE_TIMEOUT_MS, DEFAULT_GREP_REGEX_TOTAL_TIMEOUT_MS };
|
|
1031
|
+
function assertGlobPatternLength(value, what) {
|
|
1032
|
+
if (value.length > MAX_GLOB_PATTERN_LENGTH) {
|
|
1033
|
+
throw new Error(`${what} is too long (${value.length} characters); the maximum is ${MAX_GLOB_PATTERN_LENGTH}`);
|
|
1034
|
+
}
|
|
1035
|
+
}
|
|
795
1036
|
export async function searchFilesWithValidation(rootPath, pattern, allowedDirectories, options = {}) {
|
|
796
1037
|
const { excludePatterns = [] } = options;
|
|
797
1038
|
const results = [];
|
|
1039
|
+
assertGlobPatternLength(pattern, "Glob pattern");
|
|
1040
|
+
for (const excludePattern of excludePatterns) {
|
|
1041
|
+
assertGlobPatternLength(excludePattern, "Exclude pattern");
|
|
1042
|
+
}
|
|
798
1043
|
// Check if pattern requires recursive search (contains ** or has path separators)
|
|
799
1044
|
const needsRecursion = pattern.includes("**") || pattern.includes("/") || pattern.includes("\\");
|
|
800
1045
|
async function search(currentPath, currentDepth = 0) {
|
|
@@ -826,13 +1071,20 @@ export async function searchFilesWithValidation(rootPath, pattern, allowedDirect
|
|
|
826
1071
|
return results;
|
|
827
1072
|
}
|
|
828
1073
|
export async function grepFilesWithValidation(pattern, searchPath, allowedDirectories, options = {}) {
|
|
829
|
-
const { caseInsensitive = false, contextBefore = 0, contextAfter = 0, outputMode = "content", headLimit, multiline = false, fileType, globPattern, } = options;
|
|
830
|
-
|
|
1074
|
+
const { caseInsensitive = false, contextBefore = 0, contextAfter = 0, outputMode = "content", headLimit, multiline = false, fileType, globPattern, regexFileTimeoutMs, regexTotalTimeoutMs, } = options;
|
|
1075
|
+
if (pattern.length > MAX_GREP_PATTERN_LENGTH) {
|
|
1076
|
+
throw new Error(`Regex pattern is too long (${pattern.length} characters); the maximum is ${MAX_GREP_PATTERN_LENGTH}`);
|
|
1077
|
+
}
|
|
1078
|
+
if (globPattern) {
|
|
1079
|
+
assertGlobPatternLength(globPattern, "Glob pattern");
|
|
1080
|
+
}
|
|
1081
|
+
// Validate the pattern up front with the same flags the line matcher uses
|
|
1082
|
+
// (no 'g' flag to avoid stateful matching). Compiling is cheap; all matching
|
|
1083
|
+
// against file content happens in a time-bounded worker (VFO-14).
|
|
831
1084
|
const flags = caseInsensitive ? "i" : "";
|
|
832
1085
|
const dotAllFlag = multiline ? "s" : "";
|
|
833
|
-
let regex;
|
|
834
1086
|
try {
|
|
835
|
-
|
|
1087
|
+
new RegExp(pattern, flags + dotAllFlag);
|
|
836
1088
|
}
|
|
837
1089
|
catch (error) {
|
|
838
1090
|
throw new Error(`Invalid regex pattern: ${pattern} - ${error instanceof Error ? error.message : String(error)}`);
|
|
@@ -848,6 +1100,32 @@ export async function grepFilesWithValidation(pattern, searchPath, allowedDirect
|
|
|
848
1100
|
// Determine if we need to search recursively
|
|
849
1101
|
const stats = await fs.stat(searchPath);
|
|
850
1102
|
const isDirectory = stats.isDirectory();
|
|
1103
|
+
// One worker per call, reused across files and always terminated below.
|
|
1104
|
+
const regexSession = new RegexEvaluationSession({
|
|
1105
|
+
fileTimeoutMs: regexFileTimeoutMs,
|
|
1106
|
+
totalTimeoutMs: regexTotalTimeoutMs,
|
|
1107
|
+
});
|
|
1108
|
+
function buildMatch(filePath, lines, i) {
|
|
1109
|
+
const match = {
|
|
1110
|
+
file: filePath,
|
|
1111
|
+
line: i + 1,
|
|
1112
|
+
content: lines[i],
|
|
1113
|
+
};
|
|
1114
|
+
// Add context lines if requested
|
|
1115
|
+
if (contextBefore > 0) {
|
|
1116
|
+
match.contextBefore = [];
|
|
1117
|
+
for (let j = Math.max(0, i - contextBefore); j < i; j++) {
|
|
1118
|
+
match.contextBefore.push(lines[j]);
|
|
1119
|
+
}
|
|
1120
|
+
}
|
|
1121
|
+
if (contextAfter > 0) {
|
|
1122
|
+
match.contextAfter = [];
|
|
1123
|
+
for (let j = i + 1; j < Math.min(lines.length, i + 1 + contextAfter); j++) {
|
|
1124
|
+
match.contextAfter.push(lines[j]);
|
|
1125
|
+
}
|
|
1126
|
+
}
|
|
1127
|
+
return match;
|
|
1128
|
+
}
|
|
851
1129
|
async function searchFile(filePath) {
|
|
852
1130
|
// Validate path against allowed directories
|
|
853
1131
|
try {
|
|
@@ -866,10 +1144,13 @@ export async function grepFilesWithValidation(pattern, searchPath, allowedDirect
|
|
|
866
1144
|
}
|
|
867
1145
|
}
|
|
868
1146
|
}
|
|
869
|
-
// Apply glob filter
|
|
1147
|
+
// Apply glob filter. Like ripgrep's --glob, a pattern without a slash
|
|
1148
|
+
// (e.g. "*.md") matches the file name at any depth; a pattern with a
|
|
1149
|
+
// slash (e.g. "src/**/*.ts") is matched against the relative path.
|
|
870
1150
|
if (globPattern) {
|
|
871
1151
|
const relativePath = path.relative(searchPath, filePath);
|
|
872
|
-
|
|
1152
|
+
const matchBase = !/[\\/]/.test(globPattern);
|
|
1153
|
+
if (!minimatch(relativePath, globPattern, { dot: true, matchBase })) {
|
|
873
1154
|
return; // Skip files that don't match glob
|
|
874
1155
|
}
|
|
875
1156
|
}
|
|
@@ -890,81 +1171,64 @@ export async function grepFilesWithValidation(pattern, searchPath, allowedDirect
|
|
|
890
1171
|
const content = await fs.readFile(filePath, "utf-8");
|
|
891
1172
|
let fileMatchCount = 0;
|
|
892
1173
|
let fileHasMatch = false;
|
|
1174
|
+
// Matching lines still to be emitted before head_limit is reached in
|
|
1175
|
+
// content mode; lets the worker stop scanning early. Line output is
|
|
1176
|
+
// replayed below exactly as the former in-thread loop produced it.
|
|
1177
|
+
const remainingHead = outputMode === "content" && headLimit
|
|
1178
|
+
? Math.max(0, headLimit - result.matches.length)
|
|
1179
|
+
: undefined;
|
|
893
1180
|
if (multiline) {
|
|
894
|
-
// For multiline mode, search the entire content
|
|
895
|
-
|
|
896
|
-
|
|
1181
|
+
// For multiline mode, search the entire content (global regex without
|
|
1182
|
+
// dotAll), then find which lines match the dotAll regex.
|
|
1183
|
+
const evaluation = await regexSession.evaluate({
|
|
1184
|
+
content,
|
|
1185
|
+
pattern,
|
|
1186
|
+
flags,
|
|
1187
|
+
multiline: true,
|
|
1188
|
+
needLines: outputMode === "content",
|
|
1189
|
+
maxLineMatches: remainingHead,
|
|
1190
|
+
}, filePath);
|
|
1191
|
+
if (evaluation.hasGlobalMatch) {
|
|
897
1192
|
fileHasMatch = true;
|
|
898
|
-
fileMatchCount =
|
|
899
|
-
result.totalMatches +=
|
|
1193
|
+
fileMatchCount = evaluation.globalMatchCount;
|
|
1194
|
+
result.totalMatches += evaluation.globalMatchCount;
|
|
900
1195
|
if (outputMode === "content") {
|
|
901
|
-
// For multiline, we'll split by lines and find which lines have matches
|
|
902
1196
|
const lines = normalizeLineEndings(content).split("\n");
|
|
903
|
-
for (
|
|
904
|
-
|
|
905
|
-
|
|
906
|
-
if (headLimit && result.matches.length >= headLimit) {
|
|
907
|
-
break;
|
|
908
|
-
}
|
|
909
|
-
const match = {
|
|
910
|
-
file: filePath,
|
|
911
|
-
line: i + 1,
|
|
912
|
-
content: line,
|
|
913
|
-
};
|
|
914
|
-
// Add context lines if requested
|
|
915
|
-
if (contextBefore > 0) {
|
|
916
|
-
match.contextBefore = [];
|
|
917
|
-
for (let j = Math.max(0, i - contextBefore); j < i; j++) {
|
|
918
|
-
match.contextBefore.push(lines[j]);
|
|
919
|
-
}
|
|
920
|
-
}
|
|
921
|
-
if (contextAfter > 0) {
|
|
922
|
-
match.contextAfter = [];
|
|
923
|
-
for (let j = i + 1; j < Math.min(lines.length, i + 1 + contextAfter); j++) {
|
|
924
|
-
match.contextAfter.push(lines[j]);
|
|
925
|
-
}
|
|
926
|
-
}
|
|
927
|
-
result.matches.push(match);
|
|
1197
|
+
for (const i of evaluation.lineIndices) {
|
|
1198
|
+
if (headLimit && result.matches.length >= headLimit) {
|
|
1199
|
+
break;
|
|
928
1200
|
}
|
|
1201
|
+
result.matches.push(buildMatch(filePath, lines, i));
|
|
929
1202
|
}
|
|
930
1203
|
}
|
|
931
1204
|
}
|
|
932
1205
|
}
|
|
933
1206
|
else {
|
|
934
|
-
// For single-line mode, search line by line
|
|
935
|
-
|
|
936
|
-
|
|
937
|
-
|
|
938
|
-
|
|
939
|
-
|
|
940
|
-
|
|
941
|
-
|
|
942
|
-
|
|
943
|
-
|
|
944
|
-
|
|
945
|
-
|
|
946
|
-
|
|
947
|
-
|
|
948
|
-
|
|
949
|
-
|
|
950
|
-
|
|
951
|
-
|
|
952
|
-
|
|
953
|
-
|
|
954
|
-
|
|
955
|
-
|
|
956
|
-
|
|
957
|
-
match.contextBefore.push(lines[j]);
|
|
958
|
-
}
|
|
959
|
-
}
|
|
960
|
-
if (contextAfter > 0) {
|
|
961
|
-
match.contextAfter = [];
|
|
962
|
-
for (let j = i + 1; j < Math.min(lines.length, i + 1 + contextAfter); j++) {
|
|
963
|
-
match.contextAfter.push(lines[j]);
|
|
964
|
-
}
|
|
965
|
-
}
|
|
966
|
-
result.matches.push(match);
|
|
1207
|
+
// For single-line mode, search line by line. In content mode the
|
|
1208
|
+
// former loop counted the match that hit head_limit before breaking,
|
|
1209
|
+
// so one extra matching line is requested.
|
|
1210
|
+
const evaluation = await regexSession.evaluate({
|
|
1211
|
+
content,
|
|
1212
|
+
pattern,
|
|
1213
|
+
flags,
|
|
1214
|
+
multiline: false,
|
|
1215
|
+
needLines: true,
|
|
1216
|
+
maxLineMatches: remainingHead === undefined ? undefined : remainingHead + 1,
|
|
1217
|
+
}, filePath);
|
|
1218
|
+
const lines = outputMode === "content"
|
|
1219
|
+
? normalizeLineEndings(content).split("\n")
|
|
1220
|
+
: [];
|
|
1221
|
+
for (const i of evaluation.lineIndices) {
|
|
1222
|
+
fileHasMatch = true;
|
|
1223
|
+
fileMatchCount++;
|
|
1224
|
+
result.totalMatches++;
|
|
1225
|
+
// Handle different output modes
|
|
1226
|
+
if (outputMode === "content") {
|
|
1227
|
+
// Check head limit for content mode
|
|
1228
|
+
if (headLimit && result.matches.length >= headLimit) {
|
|
1229
|
+
break; // Stop processing this file
|
|
967
1230
|
}
|
|
1231
|
+
result.matches.push(buildMatch(filePath, lines, i));
|
|
968
1232
|
}
|
|
969
1233
|
}
|
|
970
1234
|
}
|
|
@@ -980,6 +1244,10 @@ export async function grepFilesWithValidation(pattern, searchPath, allowedDirect
|
|
|
980
1244
|
}
|
|
981
1245
|
}
|
|
982
1246
|
catch (error) {
|
|
1247
|
+
// Regex timeouts / worker failures abort the whole search.
|
|
1248
|
+
if (error instanceof RegexEvaluationError) {
|
|
1249
|
+
throw error;
|
|
1250
|
+
}
|
|
983
1251
|
// Skip binary files or files we can't read
|
|
984
1252
|
return;
|
|
985
1253
|
}
|
|
@@ -1015,17 +1283,26 @@ export async function grepFilesWithValidation(pattern, searchPath, allowedDirect
|
|
|
1015
1283
|
}
|
|
1016
1284
|
}
|
|
1017
1285
|
}
|
|
1018
|
-
catch {
|
|
1286
|
+
catch (error) {
|
|
1287
|
+
// Regex timeouts / worker failures abort the whole search.
|
|
1288
|
+
if (error instanceof RegexEvaluationError) {
|
|
1289
|
+
throw error;
|
|
1290
|
+
}
|
|
1019
1291
|
// Skip directories we can't read
|
|
1020
1292
|
return;
|
|
1021
1293
|
}
|
|
1022
1294
|
}
|
|
1023
1295
|
// Execute search
|
|
1024
|
-
|
|
1025
|
-
|
|
1296
|
+
try {
|
|
1297
|
+
if (isDirectory) {
|
|
1298
|
+
await searchDirectory(searchPath);
|
|
1299
|
+
}
|
|
1300
|
+
else {
|
|
1301
|
+
await searchFile(searchPath);
|
|
1302
|
+
}
|
|
1026
1303
|
}
|
|
1027
|
-
|
|
1028
|
-
await
|
|
1304
|
+
finally {
|
|
1305
|
+
await regexSession.dispose();
|
|
1029
1306
|
}
|
|
1030
1307
|
return result;
|
|
1031
1308
|
}
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Centralized resource limits.
|
|
3
|
+
*
|
|
4
|
+
* These caps protect the (single-process, stdio) MCP server from requests that
|
|
5
|
+
* would otherwise exhaust memory, hang, or crash it. Keep all tunables here so
|
|
6
|
+
* they are easy to find and adjust.
|
|
7
|
+
*/
|
|
8
|
+
const MB = 1024 * 1024;
|
|
9
|
+
// ---------------------------------------------------------------------------
|
|
10
|
+
// HTML -> PDF / DOCX generation (VFO-13)
|
|
11
|
+
// ---------------------------------------------------------------------------
|
|
12
|
+
/** Maximum decoded size of a single embedded `data:` image in generated PDF/DOCX. */
|
|
13
|
+
export const MAX_EMBEDDED_IMAGE_BYTES = 10 * MB;
|
|
14
|
+
/**
|
|
15
|
+
* Maximum pixel count of a PNG that the PDF renderer must fully decode
|
|
16
|
+
* (alpha channel, indexed transparency or interlacing). 16 MP = 4096x4096.
|
|
17
|
+
*/
|
|
18
|
+
export const MAX_DECODED_IMAGE_PIXELS = 16 * 1024 * 1024;
|
|
19
|
+
/** Hard upper bound for rendering one PDF from HTML before the write fails. */
|
|
20
|
+
export const PDF_GENERATION_TIMEOUT_MS = 30_000;
|
|
21
|
+
// ---------------------------------------------------------------------------
|
|
22
|
+
// Read paths (VFO-15)
|
|
23
|
+
// ---------------------------------------------------------------------------
|
|
24
|
+
/**
|
|
25
|
+
* Maximum size of a plain-text file returned by read_file/read_multiple_files
|
|
26
|
+
* in "full" mode, and maximum amount of text held in memory by head/tail/range
|
|
27
|
+
* reads (output plus the current partial line).
|
|
28
|
+
*/
|
|
29
|
+
export const MAX_TEXT_READ_BYTES = 10 * MB;
|
|
30
|
+
/** Maximum size of a single image returned by attach_image. */
|
|
31
|
+
export const MAX_IMAGE_ATTACH_BYTES = 10 * MB;
|
|
32
|
+
/** Maximum combined size of all images in one attach_image call. */
|
|
33
|
+
export const MAX_IMAGE_ATTACH_TOTAL_BYTES = 20 * MB;
|
|
34
|
+
/** Maximum on-disk size of a document (PDF/Office/ODF) passed to a parser. */
|
|
35
|
+
export const MAX_DOCUMENT_FILE_BYTES = 50 * MB;
|
|
36
|
+
// ---------------------------------------------------------------------------
|
|
37
|
+
// ZIP-based documents (.docx .xlsx .pptx .odt .ods .odp) - zip bomb guard
|
|
38
|
+
// ---------------------------------------------------------------------------
|
|
39
|
+
/** Maximum total uncompressed size of all entries. */
|
|
40
|
+
export const ZIP_MAX_TOTAL_UNCOMPRESSED_BYTES = 200 * MB;
|
|
41
|
+
/** Maximum number of entries in the central directory. */
|
|
42
|
+
export const ZIP_MAX_ENTRIES = 10_000;
|
|
43
|
+
/** Maximum compression ratio for an entry larger than ZIP_RATIO_CHECK_MIN_BYTES. */
|
|
44
|
+
export const ZIP_MAX_COMPRESSION_RATIO = 200;
|
|
45
|
+
/** Entries at or below this uncompressed size are exempt from the ratio check. */
|
|
46
|
+
export const ZIP_RATIO_CHECK_MIN_BYTES = 10 * MB;
|
|
47
|
+
//# sourceMappingURL=limits.js.map
|