@vincemakes/kiso-tools-node 0.36.0 → 0.38.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,96 @@
1
+ /**
2
+ * THE SEARCH CORPUS — one definition of "which files a search may see",
3
+ * shared by `search_text` and `list_dir`'s glob.
4
+ *
5
+ * They disagreed before this existed: `list_dir` listed every dot entry
6
+ * and `search_text` skipped all of them. Two walkers is how that happens,
7
+ * so there is one.
8
+ *
9
+ * The rule is the user's own declaration. A `.gitignore` FILE at the
10
+ * workspace root switches the corpus on — not the presence of `.git`, and
11
+ * the walk never goes up. A `.gitignore` is the declaration wherever the
12
+ * repository root happens to be, which is what lets a monorepo
13
+ * subdirectory carrying its own file be treated properly instead of being
14
+ * penalised for not being a repository root; and `.git` says nothing the
15
+ * file does not. With no declaration there is nothing to trust, so the
16
+ * old conservative rule stands: every dot entry skipped.
17
+ *
18
+ * Two things are never searched, declaration or not:
19
+ *
20
+ * - `.git`, which is machinery, not content.
21
+ * - the CREDENTIAL SET below — files whose conventional purpose is to
22
+ * hold credentials. A committed file is not a secret by the user's own
23
+ * declaration, so `.github/` and `.eslintrc` ARE searchable; but an
24
+ * incidental hit from a search for "KEY" is exposure without intent,
25
+ * and that is what this prevents. An explicit `read_file` of any of
26
+ * them is unchanged.
27
+ *
28
+ * `node_modules` is skipped unconditionally in both modes, as before. It
29
+ * is not a declaration question: DC-54 measured 296,924 files under a home
30
+ * directory, and a corpus that can reach a dependency tree is the freeze
31
+ * that finding was about.
32
+ */
33
+ import { type Ignore } from "ignore";
34
+ /** The walk stops here. `search_text` has always used 8; the glob inherits
35
+ * it so a pattern cannot reach deeper than a search can. */
36
+ export declare const CORPUS_MAX_DEPTH = 8;
37
+ export declare function isCredentialName(name: string): boolean;
38
+ export interface CorpusOptions {
39
+ workspaceRoot: string;
40
+ /** Default CORPUS_MAX_DEPTH. */
41
+ maxDepth?: number;
42
+ /** Stop after this many files. Default: no cap (search has its own
43
+ * budget; the glob passes 200). */
44
+ maxEntries?: number;
45
+ /** DC-49 exclude roots — the user's own knob, unchanged. */
46
+ isExcluded?: (fullPath: string) => boolean;
47
+ /** Keep only these. Applied BEFORE the cap, so `maxEntries` bounds the
48
+ * files returned and not the files walked past — capping the walk
49
+ * first and filtering after would silently return fewer matches than
50
+ * exist, which is the defect the cap note exists to prevent. */
51
+ accept?: (workspaceRelative: string) => boolean;
52
+ /** Start the walk here instead of at the root. The DECLARATION is still
53
+ * read at `workspaceRoot` and paths are still relative to it. */
54
+ walkFrom?: string;
55
+ }
56
+ export interface CorpusWalk {
57
+ /** Workspace-relative, POSIX-separated. */
58
+ files: string[];
59
+ /** The walk stopped descending somewhere. DIFFERENT from the cap: the
60
+ * remedy is "the file is deeper than a search reaches", not "narrow
61
+ * the directory". Reported separately, and never claimed when it did
62
+ * not happen — a walk that silently returns less is the same defect
63
+ * as a read that silently returns 200 lines. */
64
+ cutByDepth: boolean;
65
+ /** The entry cap was reached. */
66
+ cutByCap: boolean;
67
+ }
68
+ /** One directory's `.gitignore`, and the directory it is relative to. */
69
+ export interface Layer {
70
+ dir: string;
71
+ matcher: Ignore;
72
+ }
73
+ export declare function readLayer(dir: string): Layer | null;
74
+ /** Ignored by ANY ancestor's file, each tested against the path relative
75
+ * to the directory that declared it — which is what makes a nested
76
+ * `.gitignore` apply to its own subtree and not above it. */
77
+ export declare function ignoredBy(layers: readonly Layer[], full: string, isDir: boolean): boolean;
78
+ /** The SHARED predicate. `search_text` walks its own tree (its walk is
79
+ * interleaved with the budget and the matcher) and `walkCorpus` walks
80
+ * here, but both ask THIS — which is the whole point of there being one
81
+ * corpus rather than two walkers that agree by coincidence. */
82
+ export declare function corpusSkips(declared: boolean, layers: readonly Layer[], full: string, name: string, isDir: boolean): boolean;
83
+ /** The layers in force inside `dir`, given its parent's. */
84
+ export declare function layersEntering(dir: string, parent: readonly Layer[]): readonly Layer[];
85
+ export declare function walkCorpus(opts: CorpusOptions): CorpusWalk;
86
+ /** A minimal glob → RegExp, anchored, over workspace-relative POSIX paths.
87
+ *
88
+ * `*` any run except `/`, `?` one character except `/`, `**` any run
89
+ * including `/`. A leading `** /` (without the space) matches zero or
90
+ * more directories, so `**\/*.ts` finds a root-level `a.ts` — the
91
+ * behaviour people expect and the one a naive translation gets wrong by
92
+ * requiring at least one directory.
93
+ *
94
+ * Everything else is literal, including the regex metacharacters a path
95
+ * can legally contain. */
96
+ export declare function globToRegExp(pattern: string): RegExp;
package/dist/corpus.js ADDED
@@ -0,0 +1,218 @@
1
+ /**
2
+ * THE SEARCH CORPUS — one definition of "which files a search may see",
3
+ * shared by `search_text` and `list_dir`'s glob.
4
+ *
5
+ * They disagreed before this existed: `list_dir` listed every dot entry
6
+ * and `search_text` skipped all of them. Two walkers is how that happens,
7
+ * so there is one.
8
+ *
9
+ * The rule is the user's own declaration. A `.gitignore` FILE at the
10
+ * workspace root switches the corpus on — not the presence of `.git`, and
11
+ * the walk never goes up. A `.gitignore` is the declaration wherever the
12
+ * repository root happens to be, which is what lets a monorepo
13
+ * subdirectory carrying its own file be treated properly instead of being
14
+ * penalised for not being a repository root; and `.git` says nothing the
15
+ * file does not. With no declaration there is nothing to trust, so the
16
+ * old conservative rule stands: every dot entry skipped.
17
+ *
18
+ * Two things are never searched, declaration or not:
19
+ *
20
+ * - `.git`, which is machinery, not content.
21
+ * - the CREDENTIAL SET below — files whose conventional purpose is to
22
+ * hold credentials. A committed file is not a secret by the user's own
23
+ * declaration, so `.github/` and `.eslintrc` ARE searchable; but an
24
+ * incidental hit from a search for "KEY" is exposure without intent,
25
+ * and that is what this prevents. An explicit `read_file` of any of
26
+ * them is unchanged.
27
+ *
28
+ * `node_modules` is skipped unconditionally in both modes, as before. It
29
+ * is not a declaration question: DC-54 measured 296,924 files under a home
30
+ * directory, and a corpus that can reach a dependency tree is the freeze
31
+ * that finding was about.
32
+ */
33
+ import { readFileSync, readdirSync } from "node:fs";
34
+ import { join, relative, sep } from "node:path";
35
+ import ignore, {} from "ignore";
36
+ /** The walk stops here. `search_text` has always used 8; the glob inherits
37
+ * it so a pattern cannot reach deeper than a search can. */
38
+ export const CORPUS_MAX_DEPTH = 8;
39
+ /** THE CREDENTIAL SET — files whose CONVENTIONAL PURPOSE is to hold
40
+ * credentials. That class is the rule; the list is its application, and
41
+ * the next candidate is judged by the class rather than by resemblance.
42
+ * Changes to this list are RULINGS, not edits.
43
+ *
44
+ * Enumerated rather than matched by prefix: `.env*` as a prefix would
45
+ * also swallow `.environment` and `.envoy.yaml`, which are ordinary
46
+ * files that happen to start the same way.
47
+ *
48
+ * IN, and why:
49
+ * `.env`, `.env.*` the convention itself
50
+ * `.envrc` direnv — routinely `export AWS_SECRET_...`
51
+ * `.netrc` machine credentials, by definition
52
+ * id_rsa, id_dsa, private keys, by name. Their `.pub` counterparts
53
+ * id_ecdsa, are different names and stay searchable, which is
54
+ * id_ed25519 correct: a public key is public.
55
+ * *.pem the same, by extension
56
+ *
57
+ * OUT, deliberately, so the omissions are decisions and not oversights:
58
+ * `.npmrc` a config file by convention; tokens in it are normally
59
+ * `${VAR}` placeholders, so excluding it would cost more
60
+ * than it protects
61
+ * `*.key` too many non-secret uses to be a credential by name
62
+ *
63
+ * An explicit `read_file` of ANY of these is unchanged. Reading on
64
+ * purpose is the user's model doing what it was told; an incidental hit
65
+ * from a search for "KEY" is exposure without intent, and only the
66
+ * second is what this prevents. */
67
+ const ENV_TEMPLATES = new Set([".env.example", ".env.sample", ".env.template"]);
68
+ const CREDENTIAL_NAMES = new Set([".envrc", ".netrc", "id_rsa", "id_dsa", "id_ecdsa", "id_ed25519"]);
69
+ export function isCredentialName(name) {
70
+ if (ENV_TEMPLATES.has(name))
71
+ return false;
72
+ if (name === ".env" || name.startsWith(".env."))
73
+ return true;
74
+ if (CREDENTIAL_NAMES.has(name))
75
+ return true;
76
+ return name.endsWith(".pem");
77
+ }
78
+ export function readLayer(dir) {
79
+ let text;
80
+ try {
81
+ text = readFileSync(join(dir, ".gitignore"), "utf8");
82
+ }
83
+ catch {
84
+ return null;
85
+ }
86
+ return { dir, matcher: ignore().add(text) };
87
+ }
88
+ /** Ignored by ANY ancestor's file, each tested against the path relative
89
+ * to the directory that declared it — which is what makes a nested
90
+ * `.gitignore` apply to its own subtree and not above it. */
91
+ export function ignoredBy(layers, full, isDir) {
92
+ for (const layer of layers) {
93
+ const rel = relative(layer.dir, full).split(sep).join("/");
94
+ if (rel === "" || rel.startsWith("../"))
95
+ continue;
96
+ if (layer.matcher.ignores(isDir ? `${rel}/` : rel))
97
+ return true;
98
+ }
99
+ return false;
100
+ }
101
+ /** The SHARED predicate. `search_text` walks its own tree (its walk is
102
+ * interleaved with the budget and the matcher) and `walkCorpus` walks
103
+ * here, but both ask THIS — which is the whole point of there being one
104
+ * corpus rather than two walkers that agree by coincidence. */
105
+ export function corpusSkips(declared, layers, full, name, isDir) {
106
+ if (name === "node_modules" || name === ".git")
107
+ return true;
108
+ if (isCredentialName(name))
109
+ return true;
110
+ if (!declared)
111
+ return name.startsWith(".");
112
+ return ignoredBy(layers, full, isDir);
113
+ }
114
+ /** The layers in force inside `dir`, given its parent's. */
115
+ export function layersEntering(dir, parent) {
116
+ const own = readLayer(dir);
117
+ return own === null ? parent : [...parent, own];
118
+ }
119
+ export function walkCorpus(opts) {
120
+ const root = opts.workspaceRoot;
121
+ const maxDepth = opts.maxDepth ?? CORPUS_MAX_DEPTH;
122
+ const maxEntries = opts.maxEntries ?? Number.POSITIVE_INFINITY;
123
+ const rootLayer = readLayer(root);
124
+ /** No declaration, nothing to trust: the pre-corpus rule, unchanged. */
125
+ const declared = rootLayer !== null;
126
+ const files = [];
127
+ let cutByDepth = false;
128
+ let cutByCap = false;
129
+ const walk = (dir, depth, layers) => {
130
+ if (cutByCap)
131
+ return;
132
+ let entries;
133
+ try {
134
+ entries = readdirSync(dir, { withFileTypes: true });
135
+ }
136
+ catch {
137
+ return; // unreadable directory: skipped, as before
138
+ }
139
+ const here = depth === 0 ? layers : layersEntering(dir, layers);
140
+ for (const entry of entries) {
141
+ if (cutByCap)
142
+ return;
143
+ const name = entry.name;
144
+ const full = join(dir, name);
145
+ const isDir = entry.isDirectory();
146
+ if (corpusSkips(declared, here, full, name, isDir))
147
+ continue;
148
+ if (isDir) {
149
+ if (opts.isExcluded?.(full) === true)
150
+ continue;
151
+ if (depth + 1 > maxDepth) {
152
+ cutByDepth = true;
153
+ continue;
154
+ }
155
+ walk(full, depth + 1, here);
156
+ continue;
157
+ }
158
+ const rel = relative(root, full).split(sep).join("/");
159
+ if (opts.accept !== undefined && !opts.accept(rel))
160
+ continue;
161
+ if (files.length >= maxEntries) {
162
+ cutByCap = true;
163
+ return;
164
+ }
165
+ files.push(rel);
166
+ }
167
+ };
168
+ const start = opts.walkFrom ?? root;
169
+ // entering a subtree, the layers between root and it still apply
170
+ let startLayers = rootLayer === null ? [] : [rootLayer];
171
+ if (start !== root) {
172
+ let cur = root;
173
+ for (const part of relative(root, start).split(sep).filter(Boolean)) {
174
+ cur = join(cur, part);
175
+ startLayers = layersEntering(cur, startLayers);
176
+ }
177
+ }
178
+ walk(start, 0, startLayers);
179
+ return { files, cutByDepth, cutByCap };
180
+ }
181
+ /** A minimal glob → RegExp, anchored, over workspace-relative POSIX paths.
182
+ *
183
+ * `*` any run except `/`, `?` one character except `/`, `**` any run
184
+ * including `/`. A leading `** /` (without the space) matches zero or
185
+ * more directories, so `**\/*.ts` finds a root-level `a.ts` — the
186
+ * behaviour people expect and the one a naive translation gets wrong by
187
+ * requiring at least one directory.
188
+ *
189
+ * Everything else is literal, including the regex metacharacters a path
190
+ * can legally contain. */
191
+ export function globToRegExp(pattern) {
192
+ let out = "";
193
+ for (let i = 0; i < pattern.length; i += 1) {
194
+ const c = pattern[i];
195
+ if (c === "*") {
196
+ if (pattern[i + 1] === "*") {
197
+ // `**/` spans zero or more directories; a bare `**` spans anything
198
+ if (pattern[i + 2] === "/") {
199
+ out += "(?:.*/)?";
200
+ i += 2;
201
+ }
202
+ else {
203
+ out += ".*";
204
+ i += 1;
205
+ }
206
+ }
207
+ else
208
+ out += "[^/]*";
209
+ continue;
210
+ }
211
+ if (c === "?") {
212
+ out += "[^/]";
213
+ continue;
214
+ }
215
+ out += c.replace(/[.+^${}()|[\]\\]/g, "\\$&");
216
+ }
217
+ return new RegExp(`^${out}$`);
218
+ }
package/dist/index.d.ts CHANGED
@@ -145,6 +145,7 @@ export declare function readFileTool(opts: WorkspaceToolsOptions): Tool<{
145
145
  }>;
146
146
  export declare function listDirTool(opts: WorkspaceToolsOptions): Tool<{
147
147
  path?: string;
148
+ glob?: string;
148
149
  }>;
149
150
  /** The instrument behind gate (d): live workers and queue depth. */
150
151
  export declare function searchWorkerStats(): {
package/dist/index.js CHANGED
@@ -24,11 +24,13 @@ import { Worker } from "node:worker_threads";
24
24
  import { fileURLToPath } from "node:url";
25
25
  import { createHash } from "node:crypto";
26
26
  import { tmpdir } from "node:os";
27
- import { basename, dirname, isAbsolute, join, relative, resolve } from "node:path";
27
+ import { basename, dirname, isAbsolute, join, relative, resolve, sep } from "node:path";
28
28
  import { defineTool } from "@vincemakes/kiso-core";
29
29
  // WR-1/WR-1A — the revision-guard primitives (unit-tested in wr1a-coda):
30
30
  import { strippedShellEnv } from "./secret-env.js";
31
31
  import { contentRevision, normalizeRevision, postEffectEscape, precondition, publishNewFile, revalidateBeforeRename } from "./wr1.js";
32
+ import { CORPUS_MAX_DEPTH, globToRegExp, walkCorpus } from "./corpus.js";
33
+ import { describeSearchMiss } from "./search-miss.js";
32
34
  /**
33
35
  * TUI2-R1 (C) — THE SHELL PROGRESS SIDECAR.
34
36
  *
@@ -68,6 +70,11 @@ export function shellProgressPath(sessionId, command) {
68
70
  const key = createHash("sha256").update(`${sessionId ?? ""} ${command}`).digest("hex").slice(0, 16);
69
71
  return join(SHELL_PROGRESS_DIR, `${key}.log`);
70
72
  }
73
+ /** ACI-2: at most this many DISTINCT match lines are named in a refusal. */
74
+ const ACI2_LINES_SHOWN = 5;
75
+ /** Match offsets kept for that report. Past this the refusal says "more"
76
+ * without a number rather than one it cannot stand behind. */
77
+ const ACI2_OFFSETS_KEPT = 500;
71
78
  const OUTPUT_CAP = 100_000; // chars of output a tool result may carry
72
79
  const DEFAULT_SHELL_TIMEOUT_MS = 30_000;
73
80
  // the token round: the scoped-read defaults — read_file shows the head 200 lines
@@ -76,6 +83,12 @@ const DEFAULT_SHELL_TIMEOUT_MS = 30_000;
76
83
  // red line: every truncation names its continuation — the model always
77
84
  // has a path to the full content.
78
85
  const DEFAULT_READ_LINES = 200;
86
+ /** The default window's SECOND bound, and the one lines cannot express: 200
87
+ * lines of minified source is megabytes, 200 lines of prose is a few KB.
88
+ * Whichever binds first wins, and the cut is always at a line boundary. An
89
+ * explicit `limit` is the caller saying what they want and is not capped
90
+ * here — the 100k output cap still applies to it. */
91
+ const DEFAULT_READ_CHARS = 16_000;
79
92
  /** R3: how many files a search may read before it hands the event loop
80
93
  * back. Small enough that the 200ms motion cadence never misses a beat,
81
94
  * large enough that the yield costs nothing on a small tree. */
@@ -305,13 +318,17 @@ const INODE_SCAN_MS = 2_000;
305
318
  /** The "… N more lines" note — the actionable continuation: the exact
306
319
  * line the next read must start at, so the model can always reach the
307
320
  * full content in ranges (the red line). */
308
- function moreLinesNote(nextOffset, remaining) {
309
- return `\n… ${remaining} more ${remaining === 1 ? "line" : "lines"} (call again with offset=${nextOffset})`;
321
+ function moreLinesNote(nextOffset, remaining, limit) {
322
+ // BOTH parameters. Naming only `offset` was an instruction to read the
323
+ // rest of the file: with `limit` absent the read runs to EOF, so a model
324
+ // following its own continuation note defeated the window from the second
325
+ // read onward. Measured before the fix: 7.3% of real reads took that path.
326
+ return `\n… ${remaining} more ${remaining === 1 ? "line" : "lines"} (call again with offset=${nextOffset} limit=${limit})`;
310
327
  }
311
328
  export function readFileTool(opts) {
312
329
  return defineTool({
313
330
  name: "read_file",
314
- description: "Read a workspace file or a range (offset/limit; default: the first 200 lines, with a continuation note). The final [rev:X] line identifies the version read.",
331
+ description: "Read a workspace file or a range. Without `limit`: 200 lines or 16000 chars from `offset` (default 1), whichever binds, then a note naming the next offset and limit. The final [rev:X] line identifies the version read.",
315
332
  parameters: {
316
333
  type: "object",
317
334
  properties: {
@@ -398,21 +415,44 @@ export function readFileTool(opts) {
398
415
  errorKind: "invalid_input",
399
416
  };
400
417
  }
401
- const end = limit === undefined ? total : Math.min(start + limit - 1, total);
402
- // DEFAULT: the head 200 lines; a larger file ends with the
403
- // honest continuation note (small files ≤ 200 lines are
404
- // byte-identical to the pre-token-round behavior).
418
+ // THE DEFAULT WINDOW applies whenever `limit` is ABSENT, from
419
+ // `offset ?? 1`. It used to apply only when BOTH were absent,
420
+ // so an offset alone read to the end of the file — and since
421
+ // the note named only `offset`, a model following its own
422
+ // continuation note left the window behind after one read.
405
423
  let text;
406
424
  let note = "";
407
- if (offset === undefined && limit === undefined) {
408
- text = total <= DEFAULT_READ_LINES ? content : parts.slice(0, DEFAULT_READ_LINES).join("\n");
409
- if (total > DEFAULT_READ_LINES)
410
- note = moreLinesNote(DEFAULT_READ_LINES + 1, total - DEFAULT_READ_LINES);
425
+ if (limit === undefined) {
426
+ const lastLine = Math.min(start + DEFAULT_READ_LINES - 1, total);
427
+ const slice = parts.slice(start - 1, lastLine);
428
+ let body = slice.join("\n");
429
+ let shown = slice.length;
430
+ if (body.length > DEFAULT_READ_CHARS) {
431
+ const cut = body.lastIndexOf("\n", DEFAULT_READ_CHARS);
432
+ if (cut > 0) {
433
+ body = body.slice(0, cut);
434
+ shown = body.split("\n").length;
435
+ }
436
+ else {
437
+ // No newline inside the budget: the first line alone
438
+ // is over it. One WHOLE line is the smallest honest
439
+ // answer — a cut mid-line is a lie about the file.
440
+ body = slice[0] ?? "";
441
+ shown = 1;
442
+ }
443
+ }
444
+ // The whole file from line 1 is returned VERBATIM, trailing
445
+ // newline included: `parts.join` would drop it.
446
+ text = start === 1 && shown === total ? content : body;
447
+ const next = start + shown;
448
+ if (next <= total)
449
+ note = moreLinesNote(next, total - next + 1, DEFAULT_READ_LINES);
411
450
  }
412
451
  else {
452
+ const end = Math.min(start + limit - 1, total);
413
453
  text = parts.slice(start - 1, end).join("\n");
414
454
  if (end < total)
415
- note = moreLinesNote(end + 1, total - end);
455
+ note = moreLinesNote(end + 1, total - end, limit);
416
456
  }
417
457
  // The output cap's cut must STAY actionable: cut at a line
418
458
  // boundary and name the exact next offset (the generic cap()
@@ -449,10 +489,13 @@ export function readFileTool(opts) {
449
489
  export function listDirTool(opts) {
450
490
  return defineTool({
451
491
  name: "list_dir",
452
- description: "List the entries of a directory. Omit path to list the workspace root. Capped at 200 entries with an overflow note (narrow to a subdirectory for more).",
492
+ description: "List the entries of a directory, or with glob, search the tree recursively for workspace-relative paths matching it. Capped at 200 with an overflow note.",
453
493
  parameters: {
454
494
  type: "object",
455
- properties: { path: { type: "string", description: "Workspace-relative directory to list" } },
495
+ properties: {
496
+ path: { type: "string", description: "Workspace-relative directory to list" },
497
+ glob: { type: "string", description: "Recursive search: * and ? within a segment, ** across them; matched against workspace-relative paths" },
498
+ },
456
499
  additionalProperties: false,
457
500
  },
458
501
  idempotent: true,
@@ -460,9 +503,35 @@ export function listDirTool(opts) {
460
503
  effects: { precommitSafe: true, concurrency: "shared" },
461
504
  promptSnippet: "list_dir — directory entries (the workspace ls)",
462
505
  promptGuidelines: ["narrow to a subdirectory when the listing caps at 200 entries"],
463
- execute: async ({ path }) => {
506
+ execute: async ({ path, glob }) => {
464
507
  try {
465
508
  const dir = resolveWithinRoot(opts.workspaceRoot, path ?? ".");
509
+ // ACI-8: with a pattern this is a recursive search over the
510
+ // SAME corpus `search_text` uses — one definition, so the two
511
+ // cannot drift the way they had.
512
+ if (glob !== undefined) {
513
+ const re = globToRegExp(glob);
514
+ const walk = walkCorpus({
515
+ workspaceRoot: opts.workspaceRoot,
516
+ walkFrom: dir,
517
+ maxEntries: MAX_DIR_ENTRIES,
518
+ accept: (rel) => re.test(rel),
519
+ isExcluded: (full) => {
520
+ const r = relative(opts.workspaceRoot, full).split(sep).join("/");
521
+ return (opts.excludeRoots ?? []).some((ex) => r === ex || r.startsWith(`${ex}/`));
522
+ },
523
+ });
524
+ const body = walk.files.length ? cap(walk.files.join("\n")) : `(no match for ${glob})`;
525
+ // TWO truncations, different remedies, and neither claimed
526
+ // when it did not happen: a walk that silently returns less
527
+ // is the defect a note exists to prevent.
528
+ const notes = [];
529
+ if (walk.cutByCap)
530
+ notes.push(`${MAX_DIR_ENTRIES} shown (narrow the pattern for more)`);
531
+ if (walk.cutByDepth)
532
+ notes.push(`the walk stopped at depth ${CORPUS_MAX_DEPTH} — anything deeper is not listed`);
533
+ return { content: notes.length ? `${body}\n… ${notes.join("; ")}` : body, isError: false };
534
+ }
466
535
  // DC-54 — TRUNCATED BEFORE IT BUILDS. It was a `.map` over
467
536
  // EVERY entry with the 200-entry slice only after: a directory
468
537
  // of 200,000 entries built 200,000 strings to show 200 of
@@ -663,7 +732,12 @@ export function searchTextTool(opts) {
663
732
  // CX-1 F4 (audit F4): the walk-and-match runs on its OWN thread, which
664
733
  // the deadline and the abort both TERMINATE. A catastrophic regex used
665
734
  // to block this loop — no budget check, timer or abort could run.
666
- const outcome = await runSearchWorker({ token: 0, root: searchRootReal, single, pattern, flags, excluded, maxFileBytes, maxFiles, deadline, maxMatches: MAX_SEARCH_MATCHES, sniffBytes: BINARY_SNIFF_BYTES }, deadline, ctx.signal);
735
+ const outcome = await runSearchWorker(
736
+ // The workspace root is realpath'd with the SAME helper the search
737
+ // root uses: `full` is walked from a realpath'd root, and making
738
+ // a path relative between a resolved and an unresolved base
739
+ // yields `../..` the moment a symlink sits between them.
740
+ { token: 0, root: searchRootReal, workspaceRoot: realOrSelf(opts.workspaceRoot), single, pattern, flags, excluded, maxFileBytes, maxFiles, deadline, maxMatches: MAX_SEARCH_MATCHES, sniffBytes: BINARY_SNIFF_BYTES }, deadline, ctx.signal);
667
741
  if (outcome.kind === "aborted")
668
742
  return { content: "search_text aborted", isError: true, errorKind: "fatal" };
669
743
  if (outcome.kind === "error")
@@ -838,10 +912,62 @@ export function writeFileTool(opts) {
838
912
  },
839
913
  });
840
914
  }
915
+ /** ACI-2: every offset where `search` occurs, OVERLAPPING ones included —
916
+ * "aa" occurs twice in "aaa", and the two resolutions differ, so the call
917
+ * is ambiguous. A scan that steps past each match by its own length
918
+ * reports one and edits blind. This is the ONLY scan: `count === 0` is the
919
+ * missing pattern and `offsets[0]` is where a unique one resolved. */
920
+ function occurrencesOf(text, search) {
921
+ const offsets = [];
922
+ let count = 0;
923
+ for (let at = text.indexOf(search); at !== -1;) {
924
+ count += 1;
925
+ if (offsets.length < ACI2_OFFSETS_KEPT)
926
+ offsets.push(at);
927
+ // The cursor must ADVANCE. indexOf("", n) clamps n to the string
928
+ // length and then returns the same offset forever — a synchronous
929
+ // spin no test timeout can interrupt. The single indexOf this
930
+ // replaced could not hang; a loop has to earn that.
931
+ const next = text.indexOf(search, at + 1);
932
+ if (next <= at)
933
+ break;
934
+ at = next;
935
+ }
936
+ return { count, offsets };
937
+ }
938
+ /** The distinct 1-based lines the (ASCENDING) offsets fall on, in ONE pass —
939
+ * a line lookup per offset re-walks the file and a common pattern has
940
+ * hundreds of them. Ascending order is what lets adjacent-dedupe stand in
941
+ * for a set. */
942
+ function linesOfOffsets(text, offsets) {
943
+ const lines = [];
944
+ let line = 1;
945
+ let i = 0;
946
+ for (const at of offsets) {
947
+ for (; i < at; i += 1)
948
+ if (text.charCodeAt(i) === 10)
949
+ line += 1;
950
+ if (lines[lines.length - 1] !== line)
951
+ lines.push(line);
952
+ }
953
+ return lines;
954
+ }
955
+ /** "line 4" / "lines 1, 2" / "lines 1, 2, 3, 4, 5 and 35 more" — the tail
956
+ * counts LINES not yet named, never matches: a pattern hit ten times on
957
+ * one line reads "line 1", because "and 9 more" would send the caller
958
+ * looking for nine lines that are not there. `exhaustive` false means the
959
+ * offsets themselves were capped, so the tail carries no number at all. */
960
+ function describeMatchLines(text, offsets, exhaustive) {
961
+ const lines = linesOfOffsets(text, offsets);
962
+ const shown = lines.slice(0, ACI2_LINES_SHOWN);
963
+ const hidden = lines.length - shown.length;
964
+ const tail = !exhaustive ? " and more" : hidden > 0 ? ` and ${hidden} more` : "";
965
+ return `${shown.length === 1 && tail === "" ? "line" : "lines"} ${shown.join(", ")}${tail}`;
966
+ }
841
967
  export function editFileTool(opts) {
842
968
  return defineTool({
843
969
  name: "edit_file",
844
- description: "Edit a workspace file at its latest revision (expectedRevision). ONE of: search+replace (first exact occurrence), or edits (1-32 disjoint hunks resolved against the same snapshot, applied atomically).",
970
+ description: "Edit a workspace file at its latest revision (expectedRevision). ONE of: search+replace (must match exactly once), or edits (1-32 disjoint hunks resolved against the same snapshot, applied atomically).",
845
971
  parameters: {
846
972
  type: "object",
847
973
  properties: {
@@ -936,19 +1062,37 @@ export function editFileTool(opts) {
936
1062
  const text = bytes.toString("utf8");
937
1063
  // WR-1E2: EVERY hunk resolves against THIS snapshot — never the
938
1064
  // output of an earlier hunk. All spans are known before any
939
- // staging; overlaps refuse (duplicate searches both resolve
940
- // first-occurrence and therefore overlap — never retargeted).
1065
+ // staging; overlaps refuse. Since ACI-2 a non-unique search is
1066
+ // already refused above, so the overlap left to catch is two
1067
+ // hunks aimed at the same unique text — never retargeted.
941
1068
  const spans = [];
942
1069
  for (let i = 0; i < hunks.length; i += 1) {
943
1070
  const h = hunks[i];
944
- const at = text.indexOf(h.search);
945
- if (at === -1) {
1071
+ const { count, offsets } = occurrencesOf(text, h.search);
1072
+ if (count === 0) {
946
1073
  // WR-1A ④: the WORLD lacks the pattern (the input is
947
1074
  // fine) and nothing ran — precondition; the note never
948
1075
  // rides an edit that wrote nothing.
949
- return precondition(hunks.length === 1 && edits === undefined ? `edit_file: pattern not found in ${path}` : `edit_file: pattern not found in ${path} (hunk ${i + 1})`);
1076
+ // The headline says WHAT failed; the detail says WHERE.
1077
+ // A refusal that names the divergence costs one line
1078
+ // here and saves a whole file read at the caller.
1079
+ const headline = hunks.length === 1 && edits === undefined
1080
+ ? `edit_file: pattern not found in ${path}`
1081
+ : `edit_file: pattern not found in ${path} (hunk ${i + 1})`;
1082
+ const detail = describeSearchMiss(text, h.search);
1083
+ return precondition(detail ? `${headline}\n${detail}` : headline);
1084
+ }
1085
+ // ACI-2: more than one resolution is a QUESTION, not an edit.
1086
+ // Taking the first one wrote the wrong place and reported
1087
+ // success — the one failure shape a mutation tool must not
1088
+ // have. The refusal carries the count and the lines so the
1089
+ // call can be fixed without reading the file again.
1090
+ if (count > 1) {
1091
+ const lines = describeMatchLines(text, offsets, count <= ACI2_OFFSETS_KEPT);
1092
+ const where = hunks.length === 1 && edits === undefined ? lines : `hunk ${i + 1}, ${lines}`;
1093
+ return precondition(`edit_file: pattern matches ${count} places in ${path} (${where}) — include enough surrounding text to make it unique`);
950
1094
  }
951
- spans.push({ start: at, end: at + h.search.length, replace: h.replace });
1095
+ spans.push({ start: offsets[0], end: offsets[0] + h.search.length, replace: h.replace });
952
1096
  }
953
1097
  const bySpan = [...spans].sort((a, b) => a.start - b.start);
954
1098
  for (let i = 1; i < bySpan.length; i += 1) {
@@ -0,0 +1,24 @@
1
+ /**
2
+ * When a search does not match, say WHERE it stopped matching.
3
+ *
4
+ * `edit_file: pattern not found in src/report.js (hunk 2)` is true and
5
+ * useless. The tool has already established, by failing, that the text is
6
+ * not there — but it also knows, or can cheaply find out, how much of the
7
+ * search DID match and what the file has instead. Withholding that leaves
8
+ * one recourse: read the whole file again.
9
+ *
10
+ * This is not a guess about what callers need. Ten refused edits in one
11
+ * measured session were all `pattern not found`, none stale, none
12
+ * overlapping, and in every one of them a long prefix matched before the
13
+ * search ran into text the caller had not written yet — 93 of 223
14
+ * characters, 94 of 286, 306 of 913. Four more searched for an import
15
+ * line with the new symbol ALREADY IN IT. The failure has one shape: the
16
+ * search describes the file as it will be, not as it is. A message that
17
+ * names the divergence answers that in one line; the current one costs a
18
+ * whole file read to discover.
19
+ */
20
+ /**
21
+ * The detail lines for a failed search, or "" when there is nothing useful
22
+ * to say. The caller owns the headline; this is what follows it.
23
+ */
24
+ export declare function describeSearchMiss(text: string, search: string): string;
@@ -0,0 +1,94 @@
1
+ /**
2
+ * When a search does not match, say WHERE it stopped matching.
3
+ *
4
+ * `edit_file: pattern not found in src/report.js (hunk 2)` is true and
5
+ * useless. The tool has already established, by failing, that the text is
6
+ * not there — but it also knows, or can cheaply find out, how much of the
7
+ * search DID match and what the file has instead. Withholding that leaves
8
+ * one recourse: read the whole file again.
9
+ *
10
+ * This is not a guess about what callers need. Ten refused edits in one
11
+ * measured session were all `pattern not found`, none stale, none
12
+ * overlapping, and in every one of them a long prefix matched before the
13
+ * search ran into text the caller had not written yet — 93 of 223
14
+ * characters, 94 of 286, 306 of 913. Four more searched for an import
15
+ * line with the new symbol ALREADY IN IT. The failure has one shape: the
16
+ * search describes the file as it will be, not as it is. A message that
17
+ * names the divergence answers that in one line; the current one costs a
18
+ * whole file read to discover.
19
+ */
20
+ /** The longest prefix of `needle` that occurs in `hay`, by length.
21
+ *
22
+ * Monotone — if a prefix occurs then so does every shorter one — so this
23
+ * binary-searches instead of walking. A linear walk is O(m) substring
24
+ * searches, which on a large file and a long search is the kind of cost
25
+ * that turns a better error message into a worse tool.
26
+ */
27
+ function longestMatchingPrefix(hay, needle) {
28
+ let lo = 0;
29
+ let hi = needle.length;
30
+ while (lo < hi) {
31
+ const mid = (lo + hi + 1) >> 1;
32
+ if (hay.includes(needle.slice(0, mid)))
33
+ lo = mid;
34
+ else
35
+ hi = mid - 1;
36
+ }
37
+ return lo;
38
+ }
39
+ /** 1-based line number of an offset. */
40
+ function lineAt(text, offset) {
41
+ let n = 1;
42
+ for (let i = 0; i < offset && i < text.length; i += 1)
43
+ if (text.charCodeAt(i) === 10)
44
+ n += 1;
45
+ return n;
46
+ }
47
+ /** A fragment for a one-line message: escaped, and bounded. */
48
+ function fragment(s, max = 60) {
49
+ const cut = s.slice(0, max);
50
+ const shown = JSON.stringify(cut).slice(1, -1); // drop the quotes, keep \n and \t visible
51
+ return s.length > max ? `${shown}…` : shown;
52
+ }
53
+ /**
54
+ * The detail lines for a failed search, or "" when there is nothing useful
55
+ * to say. The caller owns the headline; this is what follows it.
56
+ */
57
+ export function describeSearchMiss(text, search) {
58
+ if (search.length === 0)
59
+ return "";
60
+ const matched = longestMatchingPrefix(text, search);
61
+ if (matched === 0) {
62
+ const firstLine = search.split("\n", 1)[0] ?? "";
63
+ return ` no part of it appears in the file — it begins "${fragment(firstLine)}"`;
64
+ }
65
+ // The FIRST place the prefix appears. Since ACI-2 the tool no longer
66
+ // has first-occurrence semantics to match, so the reason is now this
67
+ // function's own: a miss report needs one location and the earliest is
68
+ // the deterministic choice. A prefix occurring in several places is
69
+ // reported at the earliest of them, which can be further from where the
70
+ // caller was aiming than the report admits.
71
+ const at = text.indexOf(search.slice(0, matched));
72
+ const endOfMatch = at + matched;
73
+ const line = lineAt(text, at);
74
+ const endLine = lineAt(text, endOfMatch);
75
+ const head = ` ${matched} of ${search.length} characters matched, from line ${line} to line ${endLine}`;
76
+ const rest = search.slice(matched);
77
+ // RUNNING PAST THE END IS THE COMMON CASE AND ITS OWN SENTENCE. Four of
78
+ // the ten refusals in the measured session ended exactly here, and
79
+ // rendering that as `the file then has: ""` buries the one fact worth
80
+ // having: there is no more file. A caller that appended what it meant
81
+ // to ADD onto the end of what it meant to FIND reads its own mistake
82
+ // off this line.
83
+ if (endOfMatch >= text.length) {
84
+ return [
85
+ head,
86
+ ` the file ENDS there — your search continues for ${rest.length} more characters: "${fragment(rest)}"`,
87
+ ].join("\n");
88
+ }
89
+ return [
90
+ head,
91
+ ` the file then has: "${fragment(text.slice(endOfMatch))}"`,
92
+ ` your search wanted: "${fragment(rest)}"`,
93
+ ].join("\n");
94
+ }
@@ -18,6 +18,12 @@
18
18
  export interface SearchRequest {
19
19
  readonly token: number;
20
20
  readonly root: string;
21
+ /** The WORKSPACE root, which is not always the search root: a search under
22
+ * `packages/runtime` must still name `packages/runtime/src/run.ts` so the
23
+ * result can be handed to `read_file` unchanged. Realpath'd by the
24
+ * caller, because `full` is walked from a realpath'd root and a mixed
25
+ * pair produces `../..` the moment a symlink is involved. */
26
+ readonly workspaceRoot: string;
21
27
  /** a single file to scan instead of walking `root` */
22
28
  readonly single: string | null;
23
29
  readonly pattern: string;
@@ -16,8 +16,33 @@
16
16
  * message from a superseded worker is ignored.
17
17
  */
18
18
  import { open, readdir } from "node:fs/promises";
19
- import { join, relative } from "node:path";
19
+ import { basename, join, relative } from "node:path";
20
+ import { corpusSkips, layersEntering, readLayer } from "./corpus.js";
20
21
  import { isMainThread, parentPort } from "node:worker_threads";
22
+ /** ACI-5 — the excerpt WINDOWS THE MATCH instead of taking the line's head.
23
+ *
24
+ * `line.trim().slice(0, 160)` answers "what does this line start with",
25
+ * and the model asked "where is my pattern". Measured over 171 real
26
+ * search results and 1,931 excerpt lines: 11.7% hit the 160-char cut and
27
+ * 6.1% did not contain the pattern they matched — a hit the model cannot
28
+ * act on without spending a read to find out what it found.
29
+ *
30
+ * A short line is returned exactly as before, markers and all absent, so
31
+ * the common case is byte-identical. */
32
+ const EXCERPT_RADIUS = 80;
33
+ function excerptAround(line, regex) {
34
+ const trimmed = line.trim();
35
+ if (trimmed.length <= EXCERPT_RADIUS * 2)
36
+ return trimmed;
37
+ // A fresh non-global copy: `lastIndex` on a shared /g regex would make
38
+ // the excerpt depend on which line was scanned before it.
39
+ const found = new RegExp(regex.source, regex.flags.replace("g", "")).exec(trimmed);
40
+ const at = found ? found.index : 0;
41
+ const hit = found ? found[0].length : 0;
42
+ const start = Math.max(0, at - EXCERPT_RADIUS);
43
+ const end = Math.min(trimmed.length, at + hit + EXCERPT_RADIUS);
44
+ return `${start > 0 ? "…" : ""}${trimmed.slice(start, end)}${end < trimmed.length ? "…" : ""}`;
45
+ }
21
46
  export async function runSearch(req) {
22
47
  const regex = new RegExp(req.pattern, req.flags);
23
48
  const matches = [];
@@ -77,8 +102,11 @@ export async function runSearch(req) {
77
102
  for (const [i, line] of text.split("\n").entries()) {
78
103
  if (regex.test(line)) {
79
104
  totalMatches += 1;
105
+ // WORKSPACE-RELATIVE, not absolute: `read_file` refuses an
106
+ // absolute path, so an absolute hit here is a result the
107
+ // model cannot feed back without rewriting it by hand.
80
108
  if (matches.length < req.maxMatches)
81
- matches.push(`${full}:${i + 1}: ${line.trim().slice(0, 160)}`);
109
+ matches.push(`${relative(req.workspaceRoot, full) || basename(full)}:${i + 1}: ${excerptAround(line, regex)}`);
82
110
  }
83
111
  }
84
112
  }
@@ -86,7 +114,11 @@ export async function runSearch(req) {
86
114
  // unreadable file: skipped, like before
87
115
  }
88
116
  };
89
- const walk = async (dir, depth) => {
117
+ // ACI-4/ACI-8: the corpus is declared by a `.gitignore` FILE at the
118
+ // workspace root — not by `.git`, and the walk never goes up.
119
+ const rootLayer = readLayer(req.root);
120
+ const declared = rootLayer !== null;
121
+ const walk = async (dir, depth, layers) => {
90
122
  if (depth > 8 || outOfBudget())
91
123
  return;
92
124
  let entries;
@@ -101,18 +133,20 @@ export async function runSearch(req) {
101
133
  }
102
134
  throw err;
103
135
  }
136
+ const here = depth === 0 ? layers : layersEntering(dir, layers);
104
137
  for (const entry of entries) {
105
138
  if (outOfBudget())
106
139
  return;
107
- if (entry.name.startsWith(".") || entry.name === "node_modules")
108
- continue;
109
140
  const full = join(dir, entry.name);
110
- if (entry.isDirectory()) {
141
+ const isDir = entry.isDirectory();
142
+ if (corpusSkips(declared, here, full, entry.name, isDir))
143
+ continue;
144
+ if (isDir) {
111
145
  if (isExcluded(full)) {
112
146
  excludedDirs += 1;
113
147
  continue;
114
148
  }
115
- await walk(full, depth + 1);
149
+ await walk(full, depth + 1, here);
116
150
  }
117
151
  else if (entry.isFile())
118
152
  await scanFile(full);
@@ -122,7 +156,7 @@ export async function runSearch(req) {
122
156
  if (req.single !== null)
123
157
  await scanFile(req.single);
124
158
  else
125
- await walk(req.root, 0);
159
+ await walk(req.root, 0, rootLayer === null ? [] : [rootLayer]);
126
160
  }
127
161
  catch (err) {
128
162
  return { token: req.token, matches, totalMatches, filesSeen, skippedFiles, multiLink, unreadableDirs, excludedDirs, stopped, stoppedAt, error: err.message };
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@vincemakes/kiso-tools-node",
3
- "version": "0.36.0",
3
+ "version": "0.38.0",
4
4
  "description": "kiso coding tools for Node hosts \u2014 read file, list directory, search text, write/edit file, shell command.",
5
5
  "type": "module",
6
6
  "license": "MIT",
@@ -25,7 +25,8 @@
25
25
  "test": "vitest run"
26
26
  },
27
27
  "dependencies": {
28
- "@vincemakes/kiso-core": "0.36.0"
28
+ "@vincemakes/kiso-core": "0.38.0",
29
+ "ignore": "^7.0.9"
29
30
  },
30
31
  "devDependencies": {
31
32
  "@types/node": "^26.1.2",