@vincemakes/kiso-tools-node 0.37.0 → 0.39.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/corpus.d.ts +96 -0
- package/dist/corpus.js +218 -0
- package/dist/index.d.ts +1 -0
- package/dist/index.js +108 -10
- package/dist/search-miss.js +6 -1
- package/dist/search-worker.js +38 -7
- package/package.json +3 -2
package/dist/corpus.d.ts
ADDED
|
@@ -0,0 +1,96 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* THE SEARCH CORPUS — one definition of "which files a search may see",
|
|
3
|
+
* shared by `search_text` and `list_dir`'s glob.
|
|
4
|
+
*
|
|
5
|
+
* They disagreed before this existed: `list_dir` listed every dot entry
|
|
6
|
+
* and `search_text` skipped all of them. Two walkers is how that happens,
|
|
7
|
+
* so there is one.
|
|
8
|
+
*
|
|
9
|
+
* The rule is the user's own declaration. A `.gitignore` FILE at the
|
|
10
|
+
* workspace root switches the corpus on — not the presence of `.git`, and
|
|
11
|
+
* the walk never goes up. A `.gitignore` is the declaration wherever the
|
|
12
|
+
* repository root happens to be, which is what lets a monorepo
|
|
13
|
+
* subdirectory carrying its own file be treated properly instead of being
|
|
14
|
+
* penalised for not being a repository root; and `.git` says nothing the
|
|
15
|
+
* file does not. With no declaration there is nothing to trust, so the
|
|
16
|
+
* old conservative rule stands: every dot entry skipped.
|
|
17
|
+
*
|
|
18
|
+
* Two things are never searched, declaration or not:
|
|
19
|
+
*
|
|
20
|
+
* - `.git`, which is machinery, not content.
|
|
21
|
+
* - the CREDENTIAL SET below — files whose conventional purpose is to
|
|
22
|
+
* hold credentials. A committed file is not a secret by the user's own
|
|
23
|
+
* declaration, so `.github/` and `.eslintrc` ARE searchable; but an
|
|
24
|
+
* incidental hit from a search for "KEY" is exposure without intent,
|
|
25
|
+
* and that is what this prevents. An explicit `read_file` of any of
|
|
26
|
+
* them is unchanged.
|
|
27
|
+
*
|
|
28
|
+
* `node_modules` is skipped unconditionally in both modes, as before. It
|
|
29
|
+
* is not a declaration question: DC-54 measured 296,924 files under a home
|
|
30
|
+
* directory, and a corpus that can reach a dependency tree is the freeze
|
|
31
|
+
* that finding was about.
|
|
32
|
+
*/
|
|
33
|
+
import { type Ignore } from "ignore";
|
|
34
|
+
/** The walk stops here. `search_text` has always used 8; the glob inherits
|
|
35
|
+
* it so a pattern cannot reach deeper than a search can. */
|
|
36
|
+
export declare const CORPUS_MAX_DEPTH = 8;
|
|
37
|
+
export declare function isCredentialName(name: string): boolean;
|
|
38
|
+
export interface CorpusOptions {
|
|
39
|
+
workspaceRoot: string;
|
|
40
|
+
/** Default CORPUS_MAX_DEPTH. */
|
|
41
|
+
maxDepth?: number;
|
|
42
|
+
/** Stop after this many files. Default: no cap (search has its own
|
|
43
|
+
* budget; the glob passes 200). */
|
|
44
|
+
maxEntries?: number;
|
|
45
|
+
/** DC-49 exclude roots — the user's own knob, unchanged. */
|
|
46
|
+
isExcluded?: (fullPath: string) => boolean;
|
|
47
|
+
/** Keep only these. Applied BEFORE the cap, so `maxEntries` bounds the
|
|
48
|
+
* files returned and not the files walked past — capping the walk
|
|
49
|
+
* first and filtering after would silently return fewer matches than
|
|
50
|
+
* exist, which is the defect the cap note exists to prevent. */
|
|
51
|
+
accept?: (workspaceRelative: string) => boolean;
|
|
52
|
+
/** Start the walk here instead of at the root. The DECLARATION is still
|
|
53
|
+
* read at `workspaceRoot` and paths are still relative to it. */
|
|
54
|
+
walkFrom?: string;
|
|
55
|
+
}
|
|
56
|
+
export interface CorpusWalk {
|
|
57
|
+
/** Workspace-relative, POSIX-separated. */
|
|
58
|
+
files: string[];
|
|
59
|
+
/** The walk stopped descending somewhere. DIFFERENT from the cap: the
|
|
60
|
+
* remedy is "the file is deeper than a search reaches", not "narrow
|
|
61
|
+
* the directory". Reported separately, and never claimed when it did
|
|
62
|
+
* not happen — a walk that silently returns less is the same defect
|
|
63
|
+
* as a read that silently returns 200 lines. */
|
|
64
|
+
cutByDepth: boolean;
|
|
65
|
+
/** The entry cap was reached. */
|
|
66
|
+
cutByCap: boolean;
|
|
67
|
+
}
|
|
68
|
+
/** One directory's `.gitignore`, and the directory it is relative to. */
|
|
69
|
+
export interface Layer {
|
|
70
|
+
dir: string;
|
|
71
|
+
matcher: Ignore;
|
|
72
|
+
}
|
|
73
|
+
export declare function readLayer(dir: string): Layer | null;
|
|
74
|
+
/** Ignored by ANY ancestor's file, each tested against the path relative
|
|
75
|
+
* to the directory that declared it — which is what makes a nested
|
|
76
|
+
* `.gitignore` apply to its own subtree and not above it. */
|
|
77
|
+
export declare function ignoredBy(layers: readonly Layer[], full: string, isDir: boolean): boolean;
|
|
78
|
+
/** The SHARED predicate. `search_text` walks its own tree (its walk is
|
|
79
|
+
* interleaved with the budget and the matcher) and `walkCorpus` walks
|
|
80
|
+
* here, but both ask THIS — which is the whole point of there being one
|
|
81
|
+
* corpus rather than two walkers that agree by coincidence. */
|
|
82
|
+
export declare function corpusSkips(declared: boolean, layers: readonly Layer[], full: string, name: string, isDir: boolean): boolean;
|
|
83
|
+
/** The layers in force inside `dir`, given its parent's. */
|
|
84
|
+
export declare function layersEntering(dir: string, parent: readonly Layer[]): readonly Layer[];
|
|
85
|
+
export declare function walkCorpus(opts: CorpusOptions): CorpusWalk;
|
|
86
|
+
/** A minimal glob → RegExp, anchored, over workspace-relative POSIX paths.
|
|
87
|
+
*
|
|
88
|
+
* `*` any run except `/`, `?` one character except `/`, `**` any run
|
|
89
|
+
* including `/`. A leading `** /` (without the space) matches zero or
|
|
90
|
+
* more directories, so `**\/*.ts` finds a root-level `a.ts` — the
|
|
91
|
+
* behaviour people expect and the one a naive translation gets wrong by
|
|
92
|
+
* requiring at least one directory.
|
|
93
|
+
*
|
|
94
|
+
* Everything else is literal, including the regex metacharacters a path
|
|
95
|
+
* can legally contain. */
|
|
96
|
+
export declare function globToRegExp(pattern: string): RegExp;
|
package/dist/corpus.js
ADDED
|
@@ -0,0 +1,218 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* THE SEARCH CORPUS — one definition of "which files a search may see",
|
|
3
|
+
* shared by `search_text` and `list_dir`'s glob.
|
|
4
|
+
*
|
|
5
|
+
* They disagreed before this existed: `list_dir` listed every dot entry
|
|
6
|
+
* and `search_text` skipped all of them. Two walkers is how that happens,
|
|
7
|
+
* so there is one.
|
|
8
|
+
*
|
|
9
|
+
* The rule is the user's own declaration. A `.gitignore` FILE at the
|
|
10
|
+
* workspace root switches the corpus on — not the presence of `.git`, and
|
|
11
|
+
* the walk never goes up. A `.gitignore` is the declaration wherever the
|
|
12
|
+
* repository root happens to be, which is what lets a monorepo
|
|
13
|
+
* subdirectory carrying its own file be treated properly instead of being
|
|
14
|
+
* penalised for not being a repository root; and `.git` says nothing the
|
|
15
|
+
* file does not. With no declaration there is nothing to trust, so the
|
|
16
|
+
* old conservative rule stands: every dot entry skipped.
|
|
17
|
+
*
|
|
18
|
+
* Two things are never searched, declaration or not:
|
|
19
|
+
*
|
|
20
|
+
* - `.git`, which is machinery, not content.
|
|
21
|
+
* - the CREDENTIAL SET below — files whose conventional purpose is to
|
|
22
|
+
* hold credentials. A committed file is not a secret by the user's own
|
|
23
|
+
* declaration, so `.github/` and `.eslintrc` ARE searchable; but an
|
|
24
|
+
* incidental hit from a search for "KEY" is exposure without intent,
|
|
25
|
+
* and that is what this prevents. An explicit `read_file` of any of
|
|
26
|
+
* them is unchanged.
|
|
27
|
+
*
|
|
28
|
+
* `node_modules` is skipped unconditionally in both modes, as before. It
|
|
29
|
+
* is not a declaration question: DC-54 measured 296,924 files under a home
|
|
30
|
+
* directory, and a corpus that can reach a dependency tree is the freeze
|
|
31
|
+
* that finding was about.
|
|
32
|
+
*/
|
|
33
|
+
import { readFileSync, readdirSync } from "node:fs";
|
|
34
|
+
import { join, relative, sep } from "node:path";
|
|
35
|
+
import ignore, {} from "ignore";
|
|
36
|
+
/** The walk stops here. `search_text` has always used 8; the glob inherits
|
|
37
|
+
* it so a pattern cannot reach deeper than a search can. */
|
|
38
|
+
export const CORPUS_MAX_DEPTH = 8;
|
|
39
|
+
/** THE CREDENTIAL SET — files whose CONVENTIONAL PURPOSE is to hold
|
|
40
|
+
* credentials. That class is the rule; the list is its application, and
|
|
41
|
+
* the next candidate is judged by the class rather than by resemblance.
|
|
42
|
+
* Changes to this list are RULINGS, not edits.
|
|
43
|
+
*
|
|
44
|
+
* Enumerated rather than matched by prefix: `.env*` as a prefix would
|
|
45
|
+
* also swallow `.environment` and `.envoy.yaml`, which are ordinary
|
|
46
|
+
* files that happen to start the same way.
|
|
47
|
+
*
|
|
48
|
+
* IN, and why:
|
|
49
|
+
* `.env`, `.env.*` the convention itself
|
|
50
|
+
* `.envrc` direnv — routinely `export AWS_SECRET_...`
|
|
51
|
+
* `.netrc` machine credentials, by definition
|
|
52
|
+
* id_rsa, id_dsa, private keys, by name. Their `.pub` counterparts
|
|
53
|
+
* id_ecdsa, are different names and stay searchable, which is
|
|
54
|
+
* id_ed25519 correct: a public key is public.
|
|
55
|
+
* *.pem the same, by extension
|
|
56
|
+
*
|
|
57
|
+
* OUT, deliberately, so the omissions are decisions and not oversights:
|
|
58
|
+
* `.npmrc` a config file by convention; tokens in it are normally
|
|
59
|
+
* `${VAR}` placeholders, so excluding it would cost more
|
|
60
|
+
* than it protects
|
|
61
|
+
* `*.key` too many non-secret uses to be a credential by name
|
|
62
|
+
*
|
|
63
|
+
* An explicit `read_file` of ANY of these is unchanged. Reading on
|
|
64
|
+
* purpose is the user's model doing what it was told; an incidental hit
|
|
65
|
+
* from a search for "KEY" is exposure without intent, and only the
|
|
66
|
+
* second is what this prevents. */
|
|
67
|
+
const ENV_TEMPLATES = new Set([".env.example", ".env.sample", ".env.template"]);
|
|
68
|
+
const CREDENTIAL_NAMES = new Set([".envrc", ".netrc", "id_rsa", "id_dsa", "id_ecdsa", "id_ed25519"]);
|
|
69
|
+
export function isCredentialName(name) {
|
|
70
|
+
if (ENV_TEMPLATES.has(name))
|
|
71
|
+
return false;
|
|
72
|
+
if (name === ".env" || name.startsWith(".env."))
|
|
73
|
+
return true;
|
|
74
|
+
if (CREDENTIAL_NAMES.has(name))
|
|
75
|
+
return true;
|
|
76
|
+
return name.endsWith(".pem");
|
|
77
|
+
}
|
|
78
|
+
export function readLayer(dir) {
|
|
79
|
+
let text;
|
|
80
|
+
try {
|
|
81
|
+
text = readFileSync(join(dir, ".gitignore"), "utf8");
|
|
82
|
+
}
|
|
83
|
+
catch {
|
|
84
|
+
return null;
|
|
85
|
+
}
|
|
86
|
+
return { dir, matcher: ignore().add(text) };
|
|
87
|
+
}
|
|
88
|
+
/** Ignored by ANY ancestor's file, each tested against the path relative
|
|
89
|
+
* to the directory that declared it — which is what makes a nested
|
|
90
|
+
* `.gitignore` apply to its own subtree and not above it. */
|
|
91
|
+
export function ignoredBy(layers, full, isDir) {
|
|
92
|
+
for (const layer of layers) {
|
|
93
|
+
const rel = relative(layer.dir, full).split(sep).join("/");
|
|
94
|
+
if (rel === "" || rel.startsWith("../"))
|
|
95
|
+
continue;
|
|
96
|
+
if (layer.matcher.ignores(isDir ? `${rel}/` : rel))
|
|
97
|
+
return true;
|
|
98
|
+
}
|
|
99
|
+
return false;
|
|
100
|
+
}
|
|
101
|
+
/** The SHARED predicate. `search_text` walks its own tree (its walk is
|
|
102
|
+
* interleaved with the budget and the matcher) and `walkCorpus` walks
|
|
103
|
+
* here, but both ask THIS — which is the whole point of there being one
|
|
104
|
+
* corpus rather than two walkers that agree by coincidence. */
|
|
105
|
+
export function corpusSkips(declared, layers, full, name, isDir) {
|
|
106
|
+
if (name === "node_modules" || name === ".git")
|
|
107
|
+
return true;
|
|
108
|
+
if (isCredentialName(name))
|
|
109
|
+
return true;
|
|
110
|
+
if (!declared)
|
|
111
|
+
return name.startsWith(".");
|
|
112
|
+
return ignoredBy(layers, full, isDir);
|
|
113
|
+
}
|
|
114
|
+
/** The layers in force inside `dir`, given its parent's. */
|
|
115
|
+
export function layersEntering(dir, parent) {
|
|
116
|
+
const own = readLayer(dir);
|
|
117
|
+
return own === null ? parent : [...parent, own];
|
|
118
|
+
}
|
|
119
|
+
export function walkCorpus(opts) {
|
|
120
|
+
const root = opts.workspaceRoot;
|
|
121
|
+
const maxDepth = opts.maxDepth ?? CORPUS_MAX_DEPTH;
|
|
122
|
+
const maxEntries = opts.maxEntries ?? Number.POSITIVE_INFINITY;
|
|
123
|
+
const rootLayer = readLayer(root);
|
|
124
|
+
/** No declaration, nothing to trust: the pre-corpus rule, unchanged. */
|
|
125
|
+
const declared = rootLayer !== null;
|
|
126
|
+
const files = [];
|
|
127
|
+
let cutByDepth = false;
|
|
128
|
+
let cutByCap = false;
|
|
129
|
+
const walk = (dir, depth, layers) => {
|
|
130
|
+
if (cutByCap)
|
|
131
|
+
return;
|
|
132
|
+
let entries;
|
|
133
|
+
try {
|
|
134
|
+
entries = readdirSync(dir, { withFileTypes: true });
|
|
135
|
+
}
|
|
136
|
+
catch {
|
|
137
|
+
return; // unreadable directory: skipped, as before
|
|
138
|
+
}
|
|
139
|
+
const here = depth === 0 ? layers : layersEntering(dir, layers);
|
|
140
|
+
for (const entry of entries) {
|
|
141
|
+
if (cutByCap)
|
|
142
|
+
return;
|
|
143
|
+
const name = entry.name;
|
|
144
|
+
const full = join(dir, name);
|
|
145
|
+
const isDir = entry.isDirectory();
|
|
146
|
+
if (corpusSkips(declared, here, full, name, isDir))
|
|
147
|
+
continue;
|
|
148
|
+
if (isDir) {
|
|
149
|
+
if (opts.isExcluded?.(full) === true)
|
|
150
|
+
continue;
|
|
151
|
+
if (depth + 1 > maxDepth) {
|
|
152
|
+
cutByDepth = true;
|
|
153
|
+
continue;
|
|
154
|
+
}
|
|
155
|
+
walk(full, depth + 1, here);
|
|
156
|
+
continue;
|
|
157
|
+
}
|
|
158
|
+
const rel = relative(root, full).split(sep).join("/");
|
|
159
|
+
if (opts.accept !== undefined && !opts.accept(rel))
|
|
160
|
+
continue;
|
|
161
|
+
if (files.length >= maxEntries) {
|
|
162
|
+
cutByCap = true;
|
|
163
|
+
return;
|
|
164
|
+
}
|
|
165
|
+
files.push(rel);
|
|
166
|
+
}
|
|
167
|
+
};
|
|
168
|
+
const start = opts.walkFrom ?? root;
|
|
169
|
+
// entering a subtree, the layers between root and it still apply
|
|
170
|
+
let startLayers = rootLayer === null ? [] : [rootLayer];
|
|
171
|
+
if (start !== root) {
|
|
172
|
+
let cur = root;
|
|
173
|
+
for (const part of relative(root, start).split(sep).filter(Boolean)) {
|
|
174
|
+
cur = join(cur, part);
|
|
175
|
+
startLayers = layersEntering(cur, startLayers);
|
|
176
|
+
}
|
|
177
|
+
}
|
|
178
|
+
walk(start, 0, startLayers);
|
|
179
|
+
return { files, cutByDepth, cutByCap };
|
|
180
|
+
}
|
|
181
|
+
/** A minimal glob → RegExp, anchored, over workspace-relative POSIX paths.
|
|
182
|
+
*
|
|
183
|
+
* `*` any run except `/`, `?` one character except `/`, `**` any run
|
|
184
|
+
* including `/`. A leading `** /` (without the space) matches zero or
|
|
185
|
+
* more directories, so `**\/*.ts` finds a root-level `a.ts` — the
|
|
186
|
+
* behaviour people expect and the one a naive translation gets wrong by
|
|
187
|
+
* requiring at least one directory.
|
|
188
|
+
*
|
|
189
|
+
* Everything else is literal, including the regex metacharacters a path
|
|
190
|
+
* can legally contain. */
|
|
191
|
+
export function globToRegExp(pattern) {
|
|
192
|
+
let out = "";
|
|
193
|
+
for (let i = 0; i < pattern.length; i += 1) {
|
|
194
|
+
const c = pattern[i];
|
|
195
|
+
if (c === "*") {
|
|
196
|
+
if (pattern[i + 1] === "*") {
|
|
197
|
+
// `**/` spans zero or more directories; a bare `**` spans anything
|
|
198
|
+
if (pattern[i + 2] === "/") {
|
|
199
|
+
out += "(?:.*/)?";
|
|
200
|
+
i += 2;
|
|
201
|
+
}
|
|
202
|
+
else {
|
|
203
|
+
out += ".*";
|
|
204
|
+
i += 1;
|
|
205
|
+
}
|
|
206
|
+
}
|
|
207
|
+
else
|
|
208
|
+
out += "[^/]*";
|
|
209
|
+
continue;
|
|
210
|
+
}
|
|
211
|
+
if (c === "?") {
|
|
212
|
+
out += "[^/]";
|
|
213
|
+
continue;
|
|
214
|
+
}
|
|
215
|
+
out += c.replace(/[.+^${}()|[\]\\]/g, "\\$&");
|
|
216
|
+
}
|
|
217
|
+
return new RegExp(`^${out}$`);
|
|
218
|
+
}
|
package/dist/index.d.ts
CHANGED
|
@@ -145,6 +145,7 @@ export declare function readFileTool(opts: WorkspaceToolsOptions): Tool<{
|
|
|
145
145
|
}>;
|
|
146
146
|
export declare function listDirTool(opts: WorkspaceToolsOptions): Tool<{
|
|
147
147
|
path?: string;
|
|
148
|
+
glob?: string;
|
|
148
149
|
}>;
|
|
149
150
|
/** The instrument behind gate (d): live workers and queue depth. */
|
|
150
151
|
export declare function searchWorkerStats(): {
|
package/dist/index.js
CHANGED
|
@@ -24,11 +24,12 @@ import { Worker } from "node:worker_threads";
|
|
|
24
24
|
import { fileURLToPath } from "node:url";
|
|
25
25
|
import { createHash } from "node:crypto";
|
|
26
26
|
import { tmpdir } from "node:os";
|
|
27
|
-
import { basename, dirname, isAbsolute, join, relative, resolve } from "node:path";
|
|
27
|
+
import { basename, dirname, isAbsolute, join, relative, resolve, sep } from "node:path";
|
|
28
28
|
import { defineTool } from "@vincemakes/kiso-core";
|
|
29
29
|
// WR-1/WR-1A — the revision-guard primitives (unit-tested in wr1a-coda):
|
|
30
30
|
import { strippedShellEnv } from "./secret-env.js";
|
|
31
31
|
import { contentRevision, normalizeRevision, postEffectEscape, precondition, publishNewFile, revalidateBeforeRename } from "./wr1.js";
|
|
32
|
+
import { CORPUS_MAX_DEPTH, globToRegExp, walkCorpus } from "./corpus.js";
|
|
32
33
|
import { describeSearchMiss } from "./search-miss.js";
|
|
33
34
|
/**
|
|
34
35
|
* TUI2-R1 (C) — THE SHELL PROGRESS SIDECAR.
|
|
@@ -69,6 +70,11 @@ export function shellProgressPath(sessionId, command) {
|
|
|
69
70
|
const key = createHash("sha256").update(`${sessionId ?? ""} ${command}`).digest("hex").slice(0, 16);
|
|
70
71
|
return join(SHELL_PROGRESS_DIR, `${key}.log`);
|
|
71
72
|
}
|
|
73
|
+
/** ACI-2: at most this many DISTINCT match lines are named in a refusal. */
|
|
74
|
+
const ACI2_LINES_SHOWN = 5;
|
|
75
|
+
/** Match offsets kept for that report. Past this the refusal says "more"
|
|
76
|
+
* without a number rather than one it cannot stand behind. */
|
|
77
|
+
const ACI2_OFFSETS_KEPT = 500;
|
|
72
78
|
const OUTPUT_CAP = 100_000; // chars of output a tool result may carry
|
|
73
79
|
const DEFAULT_SHELL_TIMEOUT_MS = 30_000;
|
|
74
80
|
// the token round: the scoped-read defaults — read_file shows the head 200 lines
|
|
@@ -483,10 +489,13 @@ export function readFileTool(opts) {
|
|
|
483
489
|
export function listDirTool(opts) {
|
|
484
490
|
return defineTool({
|
|
485
491
|
name: "list_dir",
|
|
486
|
-
description: "List the entries of a directory
|
|
492
|
+
description: "List the entries of a directory, or with glob, search the tree recursively for workspace-relative paths matching it. Capped at 200 with an overflow note.",
|
|
487
493
|
parameters: {
|
|
488
494
|
type: "object",
|
|
489
|
-
properties: {
|
|
495
|
+
properties: {
|
|
496
|
+
path: { type: "string", description: "Workspace-relative directory to list" },
|
|
497
|
+
glob: { type: "string", description: "Recursive search: * and ? within a segment, ** across them; matched against workspace-relative paths" },
|
|
498
|
+
},
|
|
490
499
|
additionalProperties: false,
|
|
491
500
|
},
|
|
492
501
|
idempotent: true,
|
|
@@ -494,9 +503,35 @@ export function listDirTool(opts) {
|
|
|
494
503
|
effects: { precommitSafe: true, concurrency: "shared" },
|
|
495
504
|
promptSnippet: "list_dir — directory entries (the workspace ls)",
|
|
496
505
|
promptGuidelines: ["narrow to a subdirectory when the listing caps at 200 entries"],
|
|
497
|
-
execute: async ({ path }) => {
|
|
506
|
+
execute: async ({ path, glob }) => {
|
|
498
507
|
try {
|
|
499
508
|
const dir = resolveWithinRoot(opts.workspaceRoot, path ?? ".");
|
|
509
|
+
// ACI-8: with a pattern this is a recursive search over the
|
|
510
|
+
// SAME corpus `search_text` uses — one definition, so the two
|
|
511
|
+
// cannot drift the way they had.
|
|
512
|
+
if (glob !== undefined) {
|
|
513
|
+
const re = globToRegExp(glob);
|
|
514
|
+
const walk = walkCorpus({
|
|
515
|
+
workspaceRoot: opts.workspaceRoot,
|
|
516
|
+
walkFrom: dir,
|
|
517
|
+
maxEntries: MAX_DIR_ENTRIES,
|
|
518
|
+
accept: (rel) => re.test(rel),
|
|
519
|
+
isExcluded: (full) => {
|
|
520
|
+
const r = relative(opts.workspaceRoot, full).split(sep).join("/");
|
|
521
|
+
return (opts.excludeRoots ?? []).some((ex) => r === ex || r.startsWith(`${ex}/`));
|
|
522
|
+
},
|
|
523
|
+
});
|
|
524
|
+
const body = walk.files.length ? cap(walk.files.join("\n")) : `(no match for ${glob})`;
|
|
525
|
+
// TWO truncations, different remedies, and neither claimed
|
|
526
|
+
// when it did not happen: a walk that silently returns less
|
|
527
|
+
// is the defect a note exists to prevent.
|
|
528
|
+
const notes = [];
|
|
529
|
+
if (walk.cutByCap)
|
|
530
|
+
notes.push(`${MAX_DIR_ENTRIES} shown (narrow the pattern for more)`);
|
|
531
|
+
if (walk.cutByDepth)
|
|
532
|
+
notes.push(`the walk stopped at depth ${CORPUS_MAX_DEPTH} — anything deeper is not listed`);
|
|
533
|
+
return { content: notes.length ? `${body}\n… ${notes.join("; ")}` : body, isError: false };
|
|
534
|
+
}
|
|
500
535
|
// DC-54 — TRUNCATED BEFORE IT BUILDS. It was a `.map` over
|
|
501
536
|
// EVERY entry with the 200-entry slice only after: a directory
|
|
502
537
|
// of 200,000 entries built 200,000 strings to show 200 of
|
|
@@ -877,10 +912,62 @@ export function writeFileTool(opts) {
|
|
|
877
912
|
},
|
|
878
913
|
});
|
|
879
914
|
}
|
|
915
|
+
/** ACI-2: every offset where `search` occurs, OVERLAPPING ones included —
|
|
916
|
+
* "aa" occurs twice in "aaa", and the two resolutions differ, so the call
|
|
917
|
+
* is ambiguous. A scan that steps past each match by its own length
|
|
918
|
+
* reports one and edits blind. This is the ONLY scan: `count === 0` is the
|
|
919
|
+
* missing pattern and `offsets[0]` is where a unique one resolved. */
|
|
920
|
+
function occurrencesOf(text, search) {
|
|
921
|
+
const offsets = [];
|
|
922
|
+
let count = 0;
|
|
923
|
+
for (let at = text.indexOf(search); at !== -1;) {
|
|
924
|
+
count += 1;
|
|
925
|
+
if (offsets.length < ACI2_OFFSETS_KEPT)
|
|
926
|
+
offsets.push(at);
|
|
927
|
+
// The cursor must ADVANCE. indexOf("", n) clamps n to the string
|
|
928
|
+
// length and then returns the same offset forever — a synchronous
|
|
929
|
+
// spin no test timeout can interrupt. The single indexOf this
|
|
930
|
+
// replaced could not hang; a loop has to earn that.
|
|
931
|
+
const next = text.indexOf(search, at + 1);
|
|
932
|
+
if (next <= at)
|
|
933
|
+
break;
|
|
934
|
+
at = next;
|
|
935
|
+
}
|
|
936
|
+
return { count, offsets };
|
|
937
|
+
}
|
|
938
|
+
/** The distinct 1-based lines the (ASCENDING) offsets fall on, in ONE pass —
|
|
939
|
+
* a line lookup per offset re-walks the file and a common pattern has
|
|
940
|
+
* hundreds of them. Ascending order is what lets adjacent-dedupe stand in
|
|
941
|
+
* for a set. */
|
|
942
|
+
function linesOfOffsets(text, offsets) {
|
|
943
|
+
const lines = [];
|
|
944
|
+
let line = 1;
|
|
945
|
+
let i = 0;
|
|
946
|
+
for (const at of offsets) {
|
|
947
|
+
for (; i < at; i += 1)
|
|
948
|
+
if (text.charCodeAt(i) === 10)
|
|
949
|
+
line += 1;
|
|
950
|
+
if (lines[lines.length - 1] !== line)
|
|
951
|
+
lines.push(line);
|
|
952
|
+
}
|
|
953
|
+
return lines;
|
|
954
|
+
}
|
|
955
|
+
/** "line 4" / "lines 1, 2" / "lines 1, 2, 3, 4, 5 and 35 more" — the tail
|
|
956
|
+
* counts LINES not yet named, never matches: a pattern hit ten times on
|
|
957
|
+
* one line reads "line 1", because "and 9 more" would send the caller
|
|
958
|
+
* looking for nine lines that are not there. `exhaustive` false means the
|
|
959
|
+
* offsets themselves were capped, so the tail carries no number at all. */
|
|
960
|
+
function describeMatchLines(text, offsets, exhaustive) {
|
|
961
|
+
const lines = linesOfOffsets(text, offsets);
|
|
962
|
+
const shown = lines.slice(0, ACI2_LINES_SHOWN);
|
|
963
|
+
const hidden = lines.length - shown.length;
|
|
964
|
+
const tail = !exhaustive ? " and more" : hidden > 0 ? ` and ${hidden} more` : "";
|
|
965
|
+
return `${shown.length === 1 && tail === "" ? "line" : "lines"} ${shown.join(", ")}${tail}`;
|
|
966
|
+
}
|
|
880
967
|
export function editFileTool(opts) {
|
|
881
968
|
return defineTool({
|
|
882
969
|
name: "edit_file",
|
|
883
|
-
description: "Edit a workspace file at its latest revision (expectedRevision). ONE of: search+replace (
|
|
970
|
+
description: "Edit a workspace file at its latest revision (expectedRevision). ONE of: search+replace (must match exactly once), or edits (1-32 disjoint hunks resolved against the same snapshot, applied atomically).",
|
|
884
971
|
parameters: {
|
|
885
972
|
type: "object",
|
|
886
973
|
properties: {
|
|
@@ -975,13 +1062,14 @@ export function editFileTool(opts) {
|
|
|
975
1062
|
const text = bytes.toString("utf8");
|
|
976
1063
|
// WR-1E2: EVERY hunk resolves against THIS snapshot — never the
|
|
977
1064
|
// output of an earlier hunk. All spans are known before any
|
|
978
|
-
// staging; overlaps refuse
|
|
979
|
-
//
|
|
1065
|
+
// staging; overlaps refuse. Since ACI-2 a non-unique search is
|
|
1066
|
+
// already refused above, so the overlap left to catch is two
|
|
1067
|
+
// hunks aimed at the same unique text — never retargeted.
|
|
980
1068
|
const spans = [];
|
|
981
1069
|
for (let i = 0; i < hunks.length; i += 1) {
|
|
982
1070
|
const h = hunks[i];
|
|
983
|
-
const
|
|
984
|
-
if (
|
|
1071
|
+
const { count, offsets } = occurrencesOf(text, h.search);
|
|
1072
|
+
if (count === 0) {
|
|
985
1073
|
// WR-1A ④: the WORLD lacks the pattern (the input is
|
|
986
1074
|
// fine) and nothing ran — precondition; the note never
|
|
987
1075
|
// rides an edit that wrote nothing.
|
|
@@ -994,7 +1082,17 @@ export function editFileTool(opts) {
|
|
|
994
1082
|
const detail = describeSearchMiss(text, h.search);
|
|
995
1083
|
return precondition(detail ? `${headline}\n${detail}` : headline);
|
|
996
1084
|
}
|
|
997
|
-
|
|
1085
|
+
// ACI-2: more than one resolution is a QUESTION, not an edit.
|
|
1086
|
+
// Taking the first one wrote the wrong place and reported
|
|
1087
|
+
// success — the one failure shape a mutation tool must not
|
|
1088
|
+
// have. The refusal carries the count and the lines so the
|
|
1089
|
+
// call can be fixed without reading the file again.
|
|
1090
|
+
if (count > 1) {
|
|
1091
|
+
const lines = describeMatchLines(text, offsets, count <= ACI2_OFFSETS_KEPT);
|
|
1092
|
+
const where = hunks.length === 1 && edits === undefined ? lines : `hunk ${i + 1}, ${lines}`;
|
|
1093
|
+
return precondition(`edit_file: pattern matches ${count} places in ${path} (${where}) — include enough surrounding text to make it unique`);
|
|
1094
|
+
}
|
|
1095
|
+
spans.push({ start: offsets[0], end: offsets[0] + h.search.length, replace: h.replace });
|
|
998
1096
|
}
|
|
999
1097
|
const bySpan = [...spans].sort((a, b) => a.start - b.start);
|
|
1000
1098
|
for (let i = 1; i < bySpan.length; i += 1) {
|
package/dist/search-miss.js
CHANGED
|
@@ -62,7 +62,12 @@ export function describeSearchMiss(text, search) {
|
|
|
62
62
|
const firstLine = search.split("\n", 1)[0] ?? "";
|
|
63
63
|
return ` no part of it appears in the file — it begins "${fragment(firstLine)}"`;
|
|
64
64
|
}
|
|
65
|
-
//
|
|
65
|
+
// The FIRST place the prefix appears. Since ACI-2 the tool no longer
|
|
66
|
+
// has first-occurrence semantics to match, so the reason is now this
|
|
67
|
+
// function's own: a miss report needs one location and the earliest is
|
|
68
|
+
// the deterministic choice. A prefix occurring in several places is
|
|
69
|
+
// reported at the earliest of them, which can be further from where the
|
|
70
|
+
// caller was aiming than the report admits.
|
|
66
71
|
const at = text.indexOf(search.slice(0, matched));
|
|
67
72
|
const endOfMatch = at + matched;
|
|
68
73
|
const line = lineAt(text, at);
|
package/dist/search-worker.js
CHANGED
|
@@ -17,7 +17,32 @@
|
|
|
17
17
|
*/
|
|
18
18
|
import { open, readdir } from "node:fs/promises";
|
|
19
19
|
import { basename, join, relative } from "node:path";
|
|
20
|
+
import { corpusSkips, layersEntering, readLayer } from "./corpus.js";
|
|
20
21
|
import { isMainThread, parentPort } from "node:worker_threads";
|
|
22
|
+
/** ACI-5 — the excerpt WINDOWS THE MATCH instead of taking the line's head.
|
|
23
|
+
*
|
|
24
|
+
* `line.trim().slice(0, 160)` answers "what does this line start with",
|
|
25
|
+
* and the model asked "where is my pattern". Measured over 171 real
|
|
26
|
+
* search results and 1,931 excerpt lines: 11.7% hit the 160-char cut and
|
|
27
|
+
* 6.1% did not contain the pattern they matched — a hit the model cannot
|
|
28
|
+
* act on without spending a read to find out what it found.
|
|
29
|
+
*
|
|
30
|
+
* A short line is returned exactly as before, markers and all absent, so
|
|
31
|
+
* the common case is byte-identical. */
|
|
32
|
+
const EXCERPT_RADIUS = 80;
|
|
33
|
+
function excerptAround(line, regex) {
|
|
34
|
+
const trimmed = line.trim();
|
|
35
|
+
if (trimmed.length <= EXCERPT_RADIUS * 2)
|
|
36
|
+
return trimmed;
|
|
37
|
+
// A fresh non-global copy: `lastIndex` on a shared /g regex would make
|
|
38
|
+
// the excerpt depend on which line was scanned before it.
|
|
39
|
+
const found = new RegExp(regex.source, regex.flags.replace("g", "")).exec(trimmed);
|
|
40
|
+
const at = found ? found.index : 0;
|
|
41
|
+
const hit = found ? found[0].length : 0;
|
|
42
|
+
const start = Math.max(0, at - EXCERPT_RADIUS);
|
|
43
|
+
const end = Math.min(trimmed.length, at + hit + EXCERPT_RADIUS);
|
|
44
|
+
return `${start > 0 ? "…" : ""}${trimmed.slice(start, end)}${end < trimmed.length ? "…" : ""}`;
|
|
45
|
+
}
|
|
21
46
|
export async function runSearch(req) {
|
|
22
47
|
const regex = new RegExp(req.pattern, req.flags);
|
|
23
48
|
const matches = [];
|
|
@@ -81,7 +106,7 @@ export async function runSearch(req) {
|
|
|
81
106
|
// absolute path, so an absolute hit here is a result the
|
|
82
107
|
// model cannot feed back without rewriting it by hand.
|
|
83
108
|
if (matches.length < req.maxMatches)
|
|
84
|
-
matches.push(`${relative(req.workspaceRoot, full) || basename(full)}:${i + 1}: ${line
|
|
109
|
+
matches.push(`${relative(req.workspaceRoot, full) || basename(full)}:${i + 1}: ${excerptAround(line, regex)}`);
|
|
85
110
|
}
|
|
86
111
|
}
|
|
87
112
|
}
|
|
@@ -89,7 +114,11 @@ export async function runSearch(req) {
|
|
|
89
114
|
// unreadable file: skipped, like before
|
|
90
115
|
}
|
|
91
116
|
};
|
|
92
|
-
|
|
117
|
+
// ACI-4/ACI-8: the corpus is declared by a `.gitignore` FILE at the
|
|
118
|
+
// workspace root — not by `.git`, and the walk never goes up.
|
|
119
|
+
const rootLayer = readLayer(req.root);
|
|
120
|
+
const declared = rootLayer !== null;
|
|
121
|
+
const walk = async (dir, depth, layers) => {
|
|
93
122
|
if (depth > 8 || outOfBudget())
|
|
94
123
|
return;
|
|
95
124
|
let entries;
|
|
@@ -104,18 +133,20 @@ export async function runSearch(req) {
|
|
|
104
133
|
}
|
|
105
134
|
throw err;
|
|
106
135
|
}
|
|
136
|
+
const here = depth === 0 ? layers : layersEntering(dir, layers);
|
|
107
137
|
for (const entry of entries) {
|
|
108
138
|
if (outOfBudget())
|
|
109
139
|
return;
|
|
110
|
-
if (entry.name.startsWith(".") || entry.name === "node_modules")
|
|
111
|
-
continue;
|
|
112
140
|
const full = join(dir, entry.name);
|
|
113
|
-
|
|
141
|
+
const isDir = entry.isDirectory();
|
|
142
|
+
if (corpusSkips(declared, here, full, entry.name, isDir))
|
|
143
|
+
continue;
|
|
144
|
+
if (isDir) {
|
|
114
145
|
if (isExcluded(full)) {
|
|
115
146
|
excludedDirs += 1;
|
|
116
147
|
continue;
|
|
117
148
|
}
|
|
118
|
-
await walk(full, depth + 1);
|
|
149
|
+
await walk(full, depth + 1, here);
|
|
119
150
|
}
|
|
120
151
|
else if (entry.isFile())
|
|
121
152
|
await scanFile(full);
|
|
@@ -125,7 +156,7 @@ export async function runSearch(req) {
|
|
|
125
156
|
if (req.single !== null)
|
|
126
157
|
await scanFile(req.single);
|
|
127
158
|
else
|
|
128
|
-
await walk(req.root, 0);
|
|
159
|
+
await walk(req.root, 0, rootLayer === null ? [] : [rootLayer]);
|
|
129
160
|
}
|
|
130
161
|
catch (err) {
|
|
131
162
|
return { token: req.token, matches, totalMatches, filesSeen, skippedFiles, multiLink, unreadableDirs, excludedDirs, stopped, stoppedAt, error: err.message };
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@vincemakes/kiso-tools-node",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.39.0",
|
|
4
4
|
"description": "kiso coding tools for Node hosts \u2014 read file, list directory, search text, write/edit file, shell command.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"license": "MIT",
|
|
@@ -25,7 +25,8 @@
|
|
|
25
25
|
"test": "vitest run"
|
|
26
26
|
},
|
|
27
27
|
"dependencies": {
|
|
28
|
-
"@vincemakes/kiso-core": "0.
|
|
28
|
+
"@vincemakes/kiso-core": "0.39.0",
|
|
29
|
+
"ignore": "^7.0.9"
|
|
29
30
|
},
|
|
30
31
|
"devDependencies": {
|
|
31
32
|
"@types/node": "^26.1.2",
|