@vincemakes/kiso-tools-node 0.36.0 → 0.38.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/corpus.d.ts +96 -0
- package/dist/corpus.js +218 -0
- package/dist/index.d.ts +1 -0
- package/dist/index.js +168 -24
- package/dist/search-miss.d.ts +24 -0
- package/dist/search-miss.js +94 -0
- package/dist/search-worker.d.ts +6 -0
- package/dist/search-worker.js +42 -8
- package/package.json +3 -2
package/dist/corpus.d.ts
ADDED
|
@@ -0,0 +1,96 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* THE SEARCH CORPUS — one definition of "which files a search may see",
|
|
3
|
+
* shared by `search_text` and `list_dir`'s glob.
|
|
4
|
+
*
|
|
5
|
+
* They disagreed before this existed: `list_dir` listed every dot entry
|
|
6
|
+
* and `search_text` skipped all of them. Two walkers is how that happens,
|
|
7
|
+
* so there is one.
|
|
8
|
+
*
|
|
9
|
+
* The rule is the user's own declaration. A `.gitignore` FILE at the
|
|
10
|
+
* workspace root switches the corpus on — not the presence of `.git`, and
|
|
11
|
+
* the walk never goes up. A `.gitignore` is the declaration wherever the
|
|
12
|
+
* repository root happens to be, which is what lets a monorepo
|
|
13
|
+
* subdirectory carrying its own file be treated properly instead of being
|
|
14
|
+
* penalised for not being a repository root; and `.git` says nothing the
|
|
15
|
+
* file does not. With no declaration there is nothing to trust, so the
|
|
16
|
+
* old conservative rule stands: every dot entry skipped.
|
|
17
|
+
*
|
|
18
|
+
* Two things are never searched, declaration or not:
|
|
19
|
+
*
|
|
20
|
+
* - `.git`, which is machinery, not content.
|
|
21
|
+
* - the CREDENTIAL SET below — files whose conventional purpose is to
|
|
22
|
+
* hold credentials. A committed file is not a secret by the user's own
|
|
23
|
+
* declaration, so `.github/` and `.eslintrc` ARE searchable; but an
|
|
24
|
+
* incidental hit from a search for "KEY" is exposure without intent,
|
|
25
|
+
* and that is what this prevents. An explicit `read_file` of any of
|
|
26
|
+
* them is unchanged.
|
|
27
|
+
*
|
|
28
|
+
* `node_modules` is skipped unconditionally in both modes, as before. It
|
|
29
|
+
* is not a declaration question: DC-54 measured 296,924 files under a home
|
|
30
|
+
* directory, and a corpus that can reach a dependency tree is the freeze
|
|
31
|
+
* that finding was about.
|
|
32
|
+
*/
|
|
33
|
+
import { type Ignore } from "ignore";
|
|
34
|
+
/** The walk stops here. `search_text` has always used 8; the glob inherits
|
|
35
|
+
* it so a pattern cannot reach deeper than a search can. */
|
|
36
|
+
export declare const CORPUS_MAX_DEPTH = 8;
|
|
37
|
+
export declare function isCredentialName(name: string): boolean;
|
|
38
|
+
export interface CorpusOptions {
|
|
39
|
+
workspaceRoot: string;
|
|
40
|
+
/** Default CORPUS_MAX_DEPTH. */
|
|
41
|
+
maxDepth?: number;
|
|
42
|
+
/** Stop after this many files. Default: no cap (search has its own
|
|
43
|
+
* budget; the glob passes 200). */
|
|
44
|
+
maxEntries?: number;
|
|
45
|
+
/** DC-49 exclude roots — the user's own knob, unchanged. */
|
|
46
|
+
isExcluded?: (fullPath: string) => boolean;
|
|
47
|
+
/** Keep only these. Applied BEFORE the cap, so `maxEntries` bounds the
|
|
48
|
+
* files returned and not the files walked past — capping the walk
|
|
49
|
+
* first and filtering after would silently return fewer matches than
|
|
50
|
+
* exist, which is the defect the cap note exists to prevent. */
|
|
51
|
+
accept?: (workspaceRelative: string) => boolean;
|
|
52
|
+
/** Start the walk here instead of at the root. The DECLARATION is still
|
|
53
|
+
* read at `workspaceRoot` and paths are still relative to it. */
|
|
54
|
+
walkFrom?: string;
|
|
55
|
+
}
|
|
56
|
+
export interface CorpusWalk {
|
|
57
|
+
/** Workspace-relative, POSIX-separated. */
|
|
58
|
+
files: string[];
|
|
59
|
+
/** The walk stopped descending somewhere. DIFFERENT from the cap: the
|
|
60
|
+
* remedy is "the file is deeper than a search reaches", not "narrow
|
|
61
|
+
* the directory". Reported separately, and never claimed when it did
|
|
62
|
+
* not happen — a walk that silently returns less is the same defect
|
|
63
|
+
* as a read that silently returns 200 lines. */
|
|
64
|
+
cutByDepth: boolean;
|
|
65
|
+
/** The entry cap was reached. */
|
|
66
|
+
cutByCap: boolean;
|
|
67
|
+
}
|
|
68
|
+
/** One directory's `.gitignore`, and the directory it is relative to. */
|
|
69
|
+
export interface Layer {
|
|
70
|
+
dir: string;
|
|
71
|
+
matcher: Ignore;
|
|
72
|
+
}
|
|
73
|
+
export declare function readLayer(dir: string): Layer | null;
|
|
74
|
+
/** Ignored by ANY ancestor's file, each tested against the path relative
|
|
75
|
+
* to the directory that declared it — which is what makes a nested
|
|
76
|
+
* `.gitignore` apply to its own subtree and not above it. */
|
|
77
|
+
export declare function ignoredBy(layers: readonly Layer[], full: string, isDir: boolean): boolean;
|
|
78
|
+
/** The SHARED predicate. `search_text` walks its own tree (its walk is
|
|
79
|
+
* interleaved with the budget and the matcher) and `walkCorpus` walks
|
|
80
|
+
* here, but both ask THIS — which is the whole point of there being one
|
|
81
|
+
* corpus rather than two walkers that agree by coincidence. */
|
|
82
|
+
export declare function corpusSkips(declared: boolean, layers: readonly Layer[], full: string, name: string, isDir: boolean): boolean;
|
|
83
|
+
/** The layers in force inside `dir`, given its parent's. */
|
|
84
|
+
export declare function layersEntering(dir: string, parent: readonly Layer[]): readonly Layer[];
|
|
85
|
+
export declare function walkCorpus(opts: CorpusOptions): CorpusWalk;
|
|
86
|
+
/** A minimal glob → RegExp, anchored, over workspace-relative POSIX paths.
|
|
87
|
+
*
|
|
88
|
+
* `*` any run except `/`, `?` one character except `/`, `**` any run
|
|
89
|
+
* including `/`. A leading `** /` (without the space) matches zero or
|
|
90
|
+
* more directories, so `**\/*.ts` finds a root-level `a.ts` — the
|
|
91
|
+
* behaviour people expect and the one a naive translation gets wrong by
|
|
92
|
+
* requiring at least one directory.
|
|
93
|
+
*
|
|
94
|
+
* Everything else is literal, including the regex metacharacters a path
|
|
95
|
+
* can legally contain. */
|
|
96
|
+
export declare function globToRegExp(pattern: string): RegExp;
|
package/dist/corpus.js
ADDED
|
@@ -0,0 +1,218 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* THE SEARCH CORPUS — one definition of "which files a search may see",
|
|
3
|
+
* shared by `search_text` and `list_dir`'s glob.
|
|
4
|
+
*
|
|
5
|
+
* They disagreed before this existed: `list_dir` listed every dot entry
|
|
6
|
+
* and `search_text` skipped all of them. Two walkers is how that happens,
|
|
7
|
+
* so there is one.
|
|
8
|
+
*
|
|
9
|
+
* The rule is the user's own declaration. A `.gitignore` FILE at the
|
|
10
|
+
* workspace root switches the corpus on — not the presence of `.git`, and
|
|
11
|
+
* the walk never goes up. A `.gitignore` is the declaration wherever the
|
|
12
|
+
* repository root happens to be, which is what lets a monorepo
|
|
13
|
+
* subdirectory carrying its own file be treated properly instead of being
|
|
14
|
+
* penalised for not being a repository root; and `.git` says nothing the
|
|
15
|
+
* file does not. With no declaration there is nothing to trust, so the
|
|
16
|
+
* old conservative rule stands: every dot entry skipped.
|
|
17
|
+
*
|
|
18
|
+
* Two things are never searched, declaration or not:
|
|
19
|
+
*
|
|
20
|
+
* - `.git`, which is machinery, not content.
|
|
21
|
+
* - the CREDENTIAL SET below — files whose conventional purpose is to
|
|
22
|
+
* hold credentials. A committed file is not a secret by the user's own
|
|
23
|
+
* declaration, so `.github/` and `.eslintrc` ARE searchable; but an
|
|
24
|
+
* incidental hit from a search for "KEY" is exposure without intent,
|
|
25
|
+
* and that is what this prevents. An explicit `read_file` of any of
|
|
26
|
+
* them is unchanged.
|
|
27
|
+
*
|
|
28
|
+
* `node_modules` is skipped unconditionally in both modes, as before. It
|
|
29
|
+
* is not a declaration question: DC-54 measured 296,924 files under a home
|
|
30
|
+
* directory, and a corpus that can reach a dependency tree is the freeze
|
|
31
|
+
* that finding was about.
|
|
32
|
+
*/
|
|
33
|
+
import { readFileSync, readdirSync } from "node:fs";
|
|
34
|
+
import { join, relative, sep } from "node:path";
|
|
35
|
+
import ignore, {} from "ignore";
|
|
36
|
+
/** The walk stops here. `search_text` has always used 8; the glob inherits
|
|
37
|
+
* it so a pattern cannot reach deeper than a search can. */
|
|
38
|
+
export const CORPUS_MAX_DEPTH = 8;
|
|
39
|
+
/** THE CREDENTIAL SET — files whose CONVENTIONAL PURPOSE is to hold
|
|
40
|
+
* credentials. That class is the rule; the list is its application, and
|
|
41
|
+
* the next candidate is judged by the class rather than by resemblance.
|
|
42
|
+
* Changes to this list are RULINGS, not edits.
|
|
43
|
+
*
|
|
44
|
+
* Enumerated rather than matched by prefix: `.env*` as a prefix would
|
|
45
|
+
* also swallow `.environment` and `.envoy.yaml`, which are ordinary
|
|
46
|
+
* files that happen to start the same way.
|
|
47
|
+
*
|
|
48
|
+
* IN, and why:
|
|
49
|
+
* `.env`, `.env.*` the convention itself
|
|
50
|
+
* `.envrc` direnv — routinely `export AWS_SECRET_...`
|
|
51
|
+
* `.netrc` machine credentials, by definition
|
|
52
|
+
* id_rsa, id_dsa, private keys, by name. Their `.pub` counterparts
|
|
53
|
+
* id_ecdsa, are different names and stay searchable, which is
|
|
54
|
+
* id_ed25519 correct: a public key is public.
|
|
55
|
+
* *.pem the same, by extension
|
|
56
|
+
*
|
|
57
|
+
* OUT, deliberately, so the omissions are decisions and not oversights:
|
|
58
|
+
* `.npmrc` a config file by convention; tokens in it are normally
|
|
59
|
+
* `${VAR}` placeholders, so excluding it would cost more
|
|
60
|
+
* than it protects
|
|
61
|
+
* `*.key` too many non-secret uses to be a credential by name
|
|
62
|
+
*
|
|
63
|
+
* An explicit `read_file` of ANY of these is unchanged. Reading on
|
|
64
|
+
* purpose is the user's model doing what it was told; an incidental hit
|
|
65
|
+
* from a search for "KEY" is exposure without intent, and only the
|
|
66
|
+
* second is what this prevents. */
|
|
67
|
+
const ENV_TEMPLATES = new Set([".env.example", ".env.sample", ".env.template"]);
|
|
68
|
+
const CREDENTIAL_NAMES = new Set([".envrc", ".netrc", "id_rsa", "id_dsa", "id_ecdsa", "id_ed25519"]);
|
|
69
|
+
export function isCredentialName(name) {
|
|
70
|
+
if (ENV_TEMPLATES.has(name))
|
|
71
|
+
return false;
|
|
72
|
+
if (name === ".env" || name.startsWith(".env."))
|
|
73
|
+
return true;
|
|
74
|
+
if (CREDENTIAL_NAMES.has(name))
|
|
75
|
+
return true;
|
|
76
|
+
return name.endsWith(".pem");
|
|
77
|
+
}
|
|
78
|
+
export function readLayer(dir) {
|
|
79
|
+
let text;
|
|
80
|
+
try {
|
|
81
|
+
text = readFileSync(join(dir, ".gitignore"), "utf8");
|
|
82
|
+
}
|
|
83
|
+
catch {
|
|
84
|
+
return null;
|
|
85
|
+
}
|
|
86
|
+
return { dir, matcher: ignore().add(text) };
|
|
87
|
+
}
|
|
88
|
+
/** Ignored by ANY ancestor's file, each tested against the path relative
|
|
89
|
+
* to the directory that declared it — which is what makes a nested
|
|
90
|
+
* `.gitignore` apply to its own subtree and not above it. */
|
|
91
|
+
export function ignoredBy(layers, full, isDir) {
|
|
92
|
+
for (const layer of layers) {
|
|
93
|
+
const rel = relative(layer.dir, full).split(sep).join("/");
|
|
94
|
+
if (rel === "" || rel.startsWith("../"))
|
|
95
|
+
continue;
|
|
96
|
+
if (layer.matcher.ignores(isDir ? `${rel}/` : rel))
|
|
97
|
+
return true;
|
|
98
|
+
}
|
|
99
|
+
return false;
|
|
100
|
+
}
|
|
101
|
+
/** The SHARED predicate. `search_text` walks its own tree (its walk is
|
|
102
|
+
* interleaved with the budget and the matcher) and `walkCorpus` walks
|
|
103
|
+
* here, but both ask THIS — which is the whole point of there being one
|
|
104
|
+
* corpus rather than two walkers that agree by coincidence. */
|
|
105
|
+
export function corpusSkips(declared, layers, full, name, isDir) {
|
|
106
|
+
if (name === "node_modules" || name === ".git")
|
|
107
|
+
return true;
|
|
108
|
+
if (isCredentialName(name))
|
|
109
|
+
return true;
|
|
110
|
+
if (!declared)
|
|
111
|
+
return name.startsWith(".");
|
|
112
|
+
return ignoredBy(layers, full, isDir);
|
|
113
|
+
}
|
|
114
|
+
/** The layers in force inside `dir`, given its parent's. */
|
|
115
|
+
export function layersEntering(dir, parent) {
|
|
116
|
+
const own = readLayer(dir);
|
|
117
|
+
return own === null ? parent : [...parent, own];
|
|
118
|
+
}
|
|
119
|
+
export function walkCorpus(opts) {
|
|
120
|
+
const root = opts.workspaceRoot;
|
|
121
|
+
const maxDepth = opts.maxDepth ?? CORPUS_MAX_DEPTH;
|
|
122
|
+
const maxEntries = opts.maxEntries ?? Number.POSITIVE_INFINITY;
|
|
123
|
+
const rootLayer = readLayer(root);
|
|
124
|
+
/** No declaration, nothing to trust: the pre-corpus rule, unchanged. */
|
|
125
|
+
const declared = rootLayer !== null;
|
|
126
|
+
const files = [];
|
|
127
|
+
let cutByDepth = false;
|
|
128
|
+
let cutByCap = false;
|
|
129
|
+
const walk = (dir, depth, layers) => {
|
|
130
|
+
if (cutByCap)
|
|
131
|
+
return;
|
|
132
|
+
let entries;
|
|
133
|
+
try {
|
|
134
|
+
entries = readdirSync(dir, { withFileTypes: true });
|
|
135
|
+
}
|
|
136
|
+
catch {
|
|
137
|
+
return; // unreadable directory: skipped, as before
|
|
138
|
+
}
|
|
139
|
+
const here = depth === 0 ? layers : layersEntering(dir, layers);
|
|
140
|
+
for (const entry of entries) {
|
|
141
|
+
if (cutByCap)
|
|
142
|
+
return;
|
|
143
|
+
const name = entry.name;
|
|
144
|
+
const full = join(dir, name);
|
|
145
|
+
const isDir = entry.isDirectory();
|
|
146
|
+
if (corpusSkips(declared, here, full, name, isDir))
|
|
147
|
+
continue;
|
|
148
|
+
if (isDir) {
|
|
149
|
+
if (opts.isExcluded?.(full) === true)
|
|
150
|
+
continue;
|
|
151
|
+
if (depth + 1 > maxDepth) {
|
|
152
|
+
cutByDepth = true;
|
|
153
|
+
continue;
|
|
154
|
+
}
|
|
155
|
+
walk(full, depth + 1, here);
|
|
156
|
+
continue;
|
|
157
|
+
}
|
|
158
|
+
const rel = relative(root, full).split(sep).join("/");
|
|
159
|
+
if (opts.accept !== undefined && !opts.accept(rel))
|
|
160
|
+
continue;
|
|
161
|
+
if (files.length >= maxEntries) {
|
|
162
|
+
cutByCap = true;
|
|
163
|
+
return;
|
|
164
|
+
}
|
|
165
|
+
files.push(rel);
|
|
166
|
+
}
|
|
167
|
+
};
|
|
168
|
+
const start = opts.walkFrom ?? root;
|
|
169
|
+
// entering a subtree, the layers between root and it still apply
|
|
170
|
+
let startLayers = rootLayer === null ? [] : [rootLayer];
|
|
171
|
+
if (start !== root) {
|
|
172
|
+
let cur = root;
|
|
173
|
+
for (const part of relative(root, start).split(sep).filter(Boolean)) {
|
|
174
|
+
cur = join(cur, part);
|
|
175
|
+
startLayers = layersEntering(cur, startLayers);
|
|
176
|
+
}
|
|
177
|
+
}
|
|
178
|
+
walk(start, 0, startLayers);
|
|
179
|
+
return { files, cutByDepth, cutByCap };
|
|
180
|
+
}
|
|
181
|
+
/** A minimal glob → RegExp, anchored, over workspace-relative POSIX paths.
|
|
182
|
+
*
|
|
183
|
+
* `*` any run except `/`, `?` one character except `/`, `**` any run
|
|
184
|
+
* including `/`. A leading `** /` (without the space) matches zero or
|
|
185
|
+
* more directories, so `**\/*.ts` finds a root-level `a.ts` — the
|
|
186
|
+
* behaviour people expect and the one a naive translation gets wrong by
|
|
187
|
+
* requiring at least one directory.
|
|
188
|
+
*
|
|
189
|
+
* Everything else is literal, including the regex metacharacters a path
|
|
190
|
+
* can legally contain. */
|
|
191
|
+
export function globToRegExp(pattern) {
|
|
192
|
+
let out = "";
|
|
193
|
+
for (let i = 0; i < pattern.length; i += 1) {
|
|
194
|
+
const c = pattern[i];
|
|
195
|
+
if (c === "*") {
|
|
196
|
+
if (pattern[i + 1] === "*") {
|
|
197
|
+
// `**/` spans zero or more directories; a bare `**` spans anything
|
|
198
|
+
if (pattern[i + 2] === "/") {
|
|
199
|
+
out += "(?:.*/)?";
|
|
200
|
+
i += 2;
|
|
201
|
+
}
|
|
202
|
+
else {
|
|
203
|
+
out += ".*";
|
|
204
|
+
i += 1;
|
|
205
|
+
}
|
|
206
|
+
}
|
|
207
|
+
else
|
|
208
|
+
out += "[^/]*";
|
|
209
|
+
continue;
|
|
210
|
+
}
|
|
211
|
+
if (c === "?") {
|
|
212
|
+
out += "[^/]";
|
|
213
|
+
continue;
|
|
214
|
+
}
|
|
215
|
+
out += c.replace(/[.+^${}()|[\]\\]/g, "\\$&");
|
|
216
|
+
}
|
|
217
|
+
return new RegExp(`^${out}$`);
|
|
218
|
+
}
|
package/dist/index.d.ts
CHANGED
|
@@ -145,6 +145,7 @@ export declare function readFileTool(opts: WorkspaceToolsOptions): Tool<{
|
|
|
145
145
|
}>;
|
|
146
146
|
export declare function listDirTool(opts: WorkspaceToolsOptions): Tool<{
|
|
147
147
|
path?: string;
|
|
148
|
+
glob?: string;
|
|
148
149
|
}>;
|
|
149
150
|
/** The instrument behind gate (d): live workers and queue depth. */
|
|
150
151
|
export declare function searchWorkerStats(): {
|
package/dist/index.js
CHANGED
|
@@ -24,11 +24,13 @@ import { Worker } from "node:worker_threads";
|
|
|
24
24
|
import { fileURLToPath } from "node:url";
|
|
25
25
|
import { createHash } from "node:crypto";
|
|
26
26
|
import { tmpdir } from "node:os";
|
|
27
|
-
import { basename, dirname, isAbsolute, join, relative, resolve } from "node:path";
|
|
27
|
+
import { basename, dirname, isAbsolute, join, relative, resolve, sep } from "node:path";
|
|
28
28
|
import { defineTool } from "@vincemakes/kiso-core";
|
|
29
29
|
// WR-1/WR-1A — the revision-guard primitives (unit-tested in wr1a-coda):
|
|
30
30
|
import { strippedShellEnv } from "./secret-env.js";
|
|
31
31
|
import { contentRevision, normalizeRevision, postEffectEscape, precondition, publishNewFile, revalidateBeforeRename } from "./wr1.js";
|
|
32
|
+
import { CORPUS_MAX_DEPTH, globToRegExp, walkCorpus } from "./corpus.js";
|
|
33
|
+
import { describeSearchMiss } from "./search-miss.js";
|
|
32
34
|
/**
|
|
33
35
|
* TUI2-R1 (C) — THE SHELL PROGRESS SIDECAR.
|
|
34
36
|
*
|
|
@@ -68,6 +70,11 @@ export function shellProgressPath(sessionId, command) {
|
|
|
68
70
|
const key = createHash("sha256").update(`${sessionId ?? ""} ${command}`).digest("hex").slice(0, 16);
|
|
69
71
|
return join(SHELL_PROGRESS_DIR, `${key}.log`);
|
|
70
72
|
}
|
|
73
|
+
/** ACI-2: at most this many DISTINCT match lines are named in a refusal. */
|
|
74
|
+
const ACI2_LINES_SHOWN = 5;
|
|
75
|
+
/** Match offsets kept for that report. Past this the refusal says "more"
|
|
76
|
+
* without a number rather than one it cannot stand behind. */
|
|
77
|
+
const ACI2_OFFSETS_KEPT = 500;
|
|
71
78
|
const OUTPUT_CAP = 100_000; // chars of output a tool result may carry
|
|
72
79
|
const DEFAULT_SHELL_TIMEOUT_MS = 30_000;
|
|
73
80
|
// the token round: the scoped-read defaults — read_file shows the head 200 lines
|
|
@@ -76,6 +83,12 @@ const DEFAULT_SHELL_TIMEOUT_MS = 30_000;
|
|
|
76
83
|
// red line: every truncation names its continuation — the model always
|
|
77
84
|
// has a path to the full content.
|
|
78
85
|
const DEFAULT_READ_LINES = 200;
|
|
86
|
+
/** The default window's SECOND bound, and the one lines cannot express: 200
|
|
87
|
+
* lines of minified source is megabytes, 200 lines of prose is a few KB.
|
|
88
|
+
* Whichever binds first wins, and the cut is always at a line boundary. An
|
|
89
|
+
* explicit `limit` is the caller saying what they want and is not capped
|
|
90
|
+
* here — the 100k output cap still applies to it. */
|
|
91
|
+
const DEFAULT_READ_CHARS = 16_000;
|
|
79
92
|
/** R3: how many files a search may read before it hands the event loop
|
|
80
93
|
* back. Small enough that the 200ms motion cadence never misses a beat,
|
|
81
94
|
* large enough that the yield costs nothing on a small tree. */
|
|
@@ -305,13 +318,17 @@ const INODE_SCAN_MS = 2_000;
|
|
|
305
318
|
/** The "… N more lines" note — the actionable continuation: the exact
|
|
306
319
|
* line the next read must start at, so the model can always reach the
|
|
307
320
|
* full content in ranges (the red line). */
|
|
308
|
-
function moreLinesNote(nextOffset, remaining) {
|
|
309
|
-
|
|
321
|
+
function moreLinesNote(nextOffset, remaining, limit) {
|
|
322
|
+
// BOTH parameters. Naming only `offset` was an instruction to read the
|
|
323
|
+
// rest of the file: with `limit` absent the read runs to EOF, so a model
|
|
324
|
+
// following its own continuation note defeated the window from the second
|
|
325
|
+
// read onward. Measured before the fix: 7.3% of real reads took that path.
|
|
326
|
+
return `\n… ${remaining} more ${remaining === 1 ? "line" : "lines"} (call again with offset=${nextOffset} limit=${limit})`;
|
|
310
327
|
}
|
|
311
328
|
export function readFileTool(opts) {
|
|
312
329
|
return defineTool({
|
|
313
330
|
name: "read_file",
|
|
314
|
-
description: "Read a workspace file or a range
|
|
331
|
+
description: "Read a workspace file or a range. Without `limit`: 200 lines or 16000 chars from `offset` (default 1), whichever binds, then a note naming the next offset and limit. The final [rev:X] line identifies the version read.",
|
|
315
332
|
parameters: {
|
|
316
333
|
type: "object",
|
|
317
334
|
properties: {
|
|
@@ -398,21 +415,44 @@ export function readFileTool(opts) {
|
|
|
398
415
|
errorKind: "invalid_input",
|
|
399
416
|
};
|
|
400
417
|
}
|
|
401
|
-
|
|
402
|
-
//
|
|
403
|
-
//
|
|
404
|
-
//
|
|
418
|
+
// THE DEFAULT WINDOW applies whenever `limit` is ABSENT, from
|
|
419
|
+
// `offset ?? 1`. It used to apply only when BOTH were absent,
|
|
420
|
+
// so an offset alone read to the end of the file — and since
|
|
421
|
+
// the note named only `offset`, a model following its own
|
|
422
|
+
// continuation note left the window behind after one read.
|
|
405
423
|
let text;
|
|
406
424
|
let note = "";
|
|
407
|
-
if (
|
|
408
|
-
|
|
409
|
-
|
|
410
|
-
|
|
425
|
+
if (limit === undefined) {
|
|
426
|
+
const lastLine = Math.min(start + DEFAULT_READ_LINES - 1, total);
|
|
427
|
+
const slice = parts.slice(start - 1, lastLine);
|
|
428
|
+
let body = slice.join("\n");
|
|
429
|
+
let shown = slice.length;
|
|
430
|
+
if (body.length > DEFAULT_READ_CHARS) {
|
|
431
|
+
const cut = body.lastIndexOf("\n", DEFAULT_READ_CHARS);
|
|
432
|
+
if (cut > 0) {
|
|
433
|
+
body = body.slice(0, cut);
|
|
434
|
+
shown = body.split("\n").length;
|
|
435
|
+
}
|
|
436
|
+
else {
|
|
437
|
+
// No newline inside the budget: the first line alone
|
|
438
|
+
// is over it. One WHOLE line is the smallest honest
|
|
439
|
+
// answer — a cut mid-line is a lie about the file.
|
|
440
|
+
body = slice[0] ?? "";
|
|
441
|
+
shown = 1;
|
|
442
|
+
}
|
|
443
|
+
}
|
|
444
|
+
// The whole file from line 1 is returned VERBATIM, trailing
|
|
445
|
+
// newline included: `parts.join` would drop it.
|
|
446
|
+
text = start === 1 && shown === total ? content : body;
|
|
447
|
+
const next = start + shown;
|
|
448
|
+
if (next <= total)
|
|
449
|
+
note = moreLinesNote(next, total - next + 1, DEFAULT_READ_LINES);
|
|
411
450
|
}
|
|
412
451
|
else {
|
|
452
|
+
const end = Math.min(start + limit - 1, total);
|
|
413
453
|
text = parts.slice(start - 1, end).join("\n");
|
|
414
454
|
if (end < total)
|
|
415
|
-
note = moreLinesNote(end + 1, total - end);
|
|
455
|
+
note = moreLinesNote(end + 1, total - end, limit);
|
|
416
456
|
}
|
|
417
457
|
// The output cap's cut must STAY actionable: cut at a line
|
|
418
458
|
// boundary and name the exact next offset (the generic cap()
|
|
@@ -449,10 +489,13 @@ export function readFileTool(opts) {
|
|
|
449
489
|
export function listDirTool(opts) {
|
|
450
490
|
return defineTool({
|
|
451
491
|
name: "list_dir",
|
|
452
|
-
description: "List the entries of a directory
|
|
492
|
+
description: "List the entries of a directory, or with glob, search the tree recursively for workspace-relative paths matching it. Capped at 200 with an overflow note.",
|
|
453
493
|
parameters: {
|
|
454
494
|
type: "object",
|
|
455
|
-
properties: {
|
|
495
|
+
properties: {
|
|
496
|
+
path: { type: "string", description: "Workspace-relative directory to list" },
|
|
497
|
+
glob: { type: "string", description: "Recursive search: * and ? within a segment, ** across them; matched against workspace-relative paths" },
|
|
498
|
+
},
|
|
456
499
|
additionalProperties: false,
|
|
457
500
|
},
|
|
458
501
|
idempotent: true,
|
|
@@ -460,9 +503,35 @@ export function listDirTool(opts) {
|
|
|
460
503
|
effects: { precommitSafe: true, concurrency: "shared" },
|
|
461
504
|
promptSnippet: "list_dir — directory entries (the workspace ls)",
|
|
462
505
|
promptGuidelines: ["narrow to a subdirectory when the listing caps at 200 entries"],
|
|
463
|
-
execute: async ({ path }) => {
|
|
506
|
+
execute: async ({ path, glob }) => {
|
|
464
507
|
try {
|
|
465
508
|
const dir = resolveWithinRoot(opts.workspaceRoot, path ?? ".");
|
|
509
|
+
// ACI-8: with a pattern this is a recursive search over the
|
|
510
|
+
// SAME corpus `search_text` uses — one definition, so the two
|
|
511
|
+
// cannot drift the way they had.
|
|
512
|
+
if (glob !== undefined) {
|
|
513
|
+
const re = globToRegExp(glob);
|
|
514
|
+
const walk = walkCorpus({
|
|
515
|
+
workspaceRoot: opts.workspaceRoot,
|
|
516
|
+
walkFrom: dir,
|
|
517
|
+
maxEntries: MAX_DIR_ENTRIES,
|
|
518
|
+
accept: (rel) => re.test(rel),
|
|
519
|
+
isExcluded: (full) => {
|
|
520
|
+
const r = relative(opts.workspaceRoot, full).split(sep).join("/");
|
|
521
|
+
return (opts.excludeRoots ?? []).some((ex) => r === ex || r.startsWith(`${ex}/`));
|
|
522
|
+
},
|
|
523
|
+
});
|
|
524
|
+
const body = walk.files.length ? cap(walk.files.join("\n")) : `(no match for ${glob})`;
|
|
525
|
+
// TWO truncations, different remedies, and neither claimed
|
|
526
|
+
// when it did not happen: a walk that silently returns less
|
|
527
|
+
// is the defect a note exists to prevent.
|
|
528
|
+
const notes = [];
|
|
529
|
+
if (walk.cutByCap)
|
|
530
|
+
notes.push(`${MAX_DIR_ENTRIES} shown (narrow the pattern for more)`);
|
|
531
|
+
if (walk.cutByDepth)
|
|
532
|
+
notes.push(`the walk stopped at depth ${CORPUS_MAX_DEPTH} — anything deeper is not listed`);
|
|
533
|
+
return { content: notes.length ? `${body}\n… ${notes.join("; ")}` : body, isError: false };
|
|
534
|
+
}
|
|
466
535
|
// DC-54 — TRUNCATED BEFORE IT BUILDS. It was a `.map` over
|
|
467
536
|
// EVERY entry with the 200-entry slice only after: a directory
|
|
468
537
|
// of 200,000 entries built 200,000 strings to show 200 of
|
|
@@ -663,7 +732,12 @@ export function searchTextTool(opts) {
|
|
|
663
732
|
// CX-1 F4 (audit F4): the walk-and-match runs on its OWN thread, which
|
|
664
733
|
// the deadline and the abort both TERMINATE. A catastrophic regex used
|
|
665
734
|
// to block this loop — no budget check, timer or abort could run.
|
|
666
|
-
const outcome = await runSearchWorker(
|
|
735
|
+
const outcome = await runSearchWorker(
|
|
736
|
+
// The workspace root is realpath'd with the SAME helper the search
|
|
737
|
+
// root uses: `full` is walked from a realpath'd root, and making
|
|
738
|
+
// a path relative between a resolved and an unresolved base
|
|
739
|
+
// yields `../..` the moment a symlink sits between them.
|
|
740
|
+
{ token: 0, root: searchRootReal, workspaceRoot: realOrSelf(opts.workspaceRoot), single, pattern, flags, excluded, maxFileBytes, maxFiles, deadline, maxMatches: MAX_SEARCH_MATCHES, sniffBytes: BINARY_SNIFF_BYTES }, deadline, ctx.signal);
|
|
667
741
|
if (outcome.kind === "aborted")
|
|
668
742
|
return { content: "search_text aborted", isError: true, errorKind: "fatal" };
|
|
669
743
|
if (outcome.kind === "error")
|
|
@@ -838,10 +912,62 @@ export function writeFileTool(opts) {
|
|
|
838
912
|
},
|
|
839
913
|
});
|
|
840
914
|
}
|
|
915
|
+
/** ACI-2: every offset where `search` occurs, OVERLAPPING ones included —
|
|
916
|
+
* "aa" occurs twice in "aaa", and the two resolutions differ, so the call
|
|
917
|
+
* is ambiguous. A scan that steps past each match by its own length
|
|
918
|
+
* reports one and edits blind. This is the ONLY scan: `count === 0` is the
|
|
919
|
+
* missing pattern and `offsets[0]` is where a unique one resolved. */
|
|
920
|
+
function occurrencesOf(text, search) {
|
|
921
|
+
const offsets = [];
|
|
922
|
+
let count = 0;
|
|
923
|
+
for (let at = text.indexOf(search); at !== -1;) {
|
|
924
|
+
count += 1;
|
|
925
|
+
if (offsets.length < ACI2_OFFSETS_KEPT)
|
|
926
|
+
offsets.push(at);
|
|
927
|
+
// The cursor must ADVANCE. indexOf("", n) clamps n to the string
|
|
928
|
+
// length and then returns the same offset forever — a synchronous
|
|
929
|
+
// spin no test timeout can interrupt. The single indexOf this
|
|
930
|
+
// replaced could not hang; a loop has to earn that.
|
|
931
|
+
const next = text.indexOf(search, at + 1);
|
|
932
|
+
if (next <= at)
|
|
933
|
+
break;
|
|
934
|
+
at = next;
|
|
935
|
+
}
|
|
936
|
+
return { count, offsets };
|
|
937
|
+
}
|
|
938
|
+
/** The distinct 1-based lines the (ASCENDING) offsets fall on, in ONE pass —
|
|
939
|
+
* a line lookup per offset re-walks the file and a common pattern has
|
|
940
|
+
* hundreds of them. Ascending order is what lets adjacent-dedupe stand in
|
|
941
|
+
* for a set. */
|
|
942
|
+
function linesOfOffsets(text, offsets) {
|
|
943
|
+
const lines = [];
|
|
944
|
+
let line = 1;
|
|
945
|
+
let i = 0;
|
|
946
|
+
for (const at of offsets) {
|
|
947
|
+
for (; i < at; i += 1)
|
|
948
|
+
if (text.charCodeAt(i) === 10)
|
|
949
|
+
line += 1;
|
|
950
|
+
if (lines[lines.length - 1] !== line)
|
|
951
|
+
lines.push(line);
|
|
952
|
+
}
|
|
953
|
+
return lines;
|
|
954
|
+
}
|
|
955
|
+
/** "line 4" / "lines 1, 2" / "lines 1, 2, 3, 4, 5 and 35 more" — the tail
|
|
956
|
+
* counts LINES not yet named, never matches: a pattern hit ten times on
|
|
957
|
+
* one line reads "line 1", because "and 9 more" would send the caller
|
|
958
|
+
* looking for nine lines that are not there. `exhaustive` false means the
|
|
959
|
+
* offsets themselves were capped, so the tail carries no number at all. */
|
|
960
|
+
function describeMatchLines(text, offsets, exhaustive) {
|
|
961
|
+
const lines = linesOfOffsets(text, offsets);
|
|
962
|
+
const shown = lines.slice(0, ACI2_LINES_SHOWN);
|
|
963
|
+
const hidden = lines.length - shown.length;
|
|
964
|
+
const tail = !exhaustive ? " and more" : hidden > 0 ? ` and ${hidden} more` : "";
|
|
965
|
+
return `${shown.length === 1 && tail === "" ? "line" : "lines"} ${shown.join(", ")}${tail}`;
|
|
966
|
+
}
|
|
841
967
|
export function editFileTool(opts) {
|
|
842
968
|
return defineTool({
|
|
843
969
|
name: "edit_file",
|
|
844
|
-
description: "Edit a workspace file at its latest revision (expectedRevision). ONE of: search+replace (
|
|
970
|
+
description: "Edit a workspace file at its latest revision (expectedRevision). ONE of: search+replace (must match exactly once), or edits (1-32 disjoint hunks resolved against the same snapshot, applied atomically).",
|
|
845
971
|
parameters: {
|
|
846
972
|
type: "object",
|
|
847
973
|
properties: {
|
|
@@ -936,19 +1062,37 @@ export function editFileTool(opts) {
|
|
|
936
1062
|
const text = bytes.toString("utf8");
|
|
937
1063
|
// WR-1E2: EVERY hunk resolves against THIS snapshot — never the
|
|
938
1064
|
// output of an earlier hunk. All spans are known before any
|
|
939
|
-
// staging; overlaps refuse
|
|
940
|
-
//
|
|
1065
|
+
// staging; overlaps refuse. Since ACI-2 a non-unique search is
|
|
1066
|
+
// already refused above, so the overlap left to catch is two
|
|
1067
|
+
// hunks aimed at the same unique text — never retargeted.
|
|
941
1068
|
const spans = [];
|
|
942
1069
|
for (let i = 0; i < hunks.length; i += 1) {
|
|
943
1070
|
const h = hunks[i];
|
|
944
|
-
const
|
|
945
|
-
if (
|
|
1071
|
+
const { count, offsets } = occurrencesOf(text, h.search);
|
|
1072
|
+
if (count === 0) {
|
|
946
1073
|
// WR-1A ④: the WORLD lacks the pattern (the input is
|
|
947
1074
|
// fine) and nothing ran — precondition; the note never
|
|
948
1075
|
// rides an edit that wrote nothing.
|
|
949
|
-
|
|
1076
|
+
// The headline says WHAT failed; the detail says WHERE.
|
|
1077
|
+
// A refusal that names the divergence costs one line
|
|
1078
|
+
// here and saves a whole file read at the caller.
|
|
1079
|
+
const headline = hunks.length === 1 && edits === undefined
|
|
1080
|
+
? `edit_file: pattern not found in ${path}`
|
|
1081
|
+
: `edit_file: pattern not found in ${path} (hunk ${i + 1})`;
|
|
1082
|
+
const detail = describeSearchMiss(text, h.search);
|
|
1083
|
+
return precondition(detail ? `${headline}\n${detail}` : headline);
|
|
1084
|
+
}
|
|
1085
|
+
// ACI-2: more than one resolution is a QUESTION, not an edit.
|
|
1086
|
+
// Taking the first one wrote the wrong place and reported
|
|
1087
|
+
// success — the one failure shape a mutation tool must not
|
|
1088
|
+
// have. The refusal carries the count and the lines so the
|
|
1089
|
+
// call can be fixed without reading the file again.
|
|
1090
|
+
if (count > 1) {
|
|
1091
|
+
const lines = describeMatchLines(text, offsets, count <= ACI2_OFFSETS_KEPT);
|
|
1092
|
+
const where = hunks.length === 1 && edits === undefined ? lines : `hunk ${i + 1}, ${lines}`;
|
|
1093
|
+
return precondition(`edit_file: pattern matches ${count} places in ${path} (${where}) — include enough surrounding text to make it unique`);
|
|
950
1094
|
}
|
|
951
|
-
spans.push({ start:
|
|
1095
|
+
spans.push({ start: offsets[0], end: offsets[0] + h.search.length, replace: h.replace });
|
|
952
1096
|
}
|
|
953
1097
|
const bySpan = [...spans].sort((a, b) => a.start - b.start);
|
|
954
1098
|
for (let i = 1; i < bySpan.length; i += 1) {
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* When a search does not match, say WHERE it stopped matching.
|
|
3
|
+
*
|
|
4
|
+
* `edit_file: pattern not found in src/report.js (hunk 2)` is true and
|
|
5
|
+
* useless. The tool has already established, by failing, that the text is
|
|
6
|
+
* not there — but it also knows, or can cheaply find out, how much of the
|
|
7
|
+
* search DID match and what the file has instead. Withholding that leaves
|
|
8
|
+
* one recourse: read the whole file again.
|
|
9
|
+
*
|
|
10
|
+
* This is not a guess about what callers need. Ten refused edits in one
|
|
11
|
+
* measured session were all `pattern not found`, none stale, none
|
|
12
|
+
* overlapping, and in every one of them a long prefix matched before the
|
|
13
|
+
* search ran into text the caller had not written yet — 93 of 223
|
|
14
|
+
* characters, 94 of 286, 306 of 913. Four more searched for an import
|
|
15
|
+
* line with the new symbol ALREADY IN IT. The failure has one shape: the
|
|
16
|
+
* search describes the file as it will be, not as it is. A message that
|
|
17
|
+
* names the divergence answers that in one line; the current one costs a
|
|
18
|
+
* whole file read to discover.
|
|
19
|
+
*/
|
|
20
|
+
/**
|
|
21
|
+
* The detail lines for a failed search, or "" when there is nothing useful
|
|
22
|
+
* to say. The caller owns the headline; this is what follows it.
|
|
23
|
+
*/
|
|
24
|
+
export declare function describeSearchMiss(text: string, search: string): string;
|
|
@@ -0,0 +1,94 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* When a search does not match, say WHERE it stopped matching.
|
|
3
|
+
*
|
|
4
|
+
* `edit_file: pattern not found in src/report.js (hunk 2)` is true and
|
|
5
|
+
* useless. The tool has already established, by failing, that the text is
|
|
6
|
+
* not there — but it also knows, or can cheaply find out, how much of the
|
|
7
|
+
* search DID match and what the file has instead. Withholding that leaves
|
|
8
|
+
* one recourse: read the whole file again.
|
|
9
|
+
*
|
|
10
|
+
* This is not a guess about what callers need. Ten refused edits in one
|
|
11
|
+
* measured session were all `pattern not found`, none stale, none
|
|
12
|
+
* overlapping, and in every one of them a long prefix matched before the
|
|
13
|
+
* search ran into text the caller had not written yet — 93 of 223
|
|
14
|
+
* characters, 94 of 286, 306 of 913. Four more searched for an import
|
|
15
|
+
* line with the new symbol ALREADY IN IT. The failure has one shape: the
|
|
16
|
+
* search describes the file as it will be, not as it is. A message that
|
|
17
|
+
* names the divergence answers that in one line; the current one costs a
|
|
18
|
+
* whole file read to discover.
|
|
19
|
+
*/
|
|
20
|
+
/** The longest prefix of `needle` that occurs in `hay`, by length.
|
|
21
|
+
*
|
|
22
|
+
* Monotone — if a prefix occurs then so does every shorter one — so this
|
|
23
|
+
* binary-searches instead of walking. A linear walk is O(m) substring
|
|
24
|
+
* searches, which on a large file and a long search is the kind of cost
|
|
25
|
+
* that turns a better error message into a worse tool.
|
|
26
|
+
*/
|
|
27
|
+
function longestMatchingPrefix(hay, needle) {
|
|
28
|
+
let lo = 0;
|
|
29
|
+
let hi = needle.length;
|
|
30
|
+
while (lo < hi) {
|
|
31
|
+
const mid = (lo + hi + 1) >> 1;
|
|
32
|
+
if (hay.includes(needle.slice(0, mid)))
|
|
33
|
+
lo = mid;
|
|
34
|
+
else
|
|
35
|
+
hi = mid - 1;
|
|
36
|
+
}
|
|
37
|
+
return lo;
|
|
38
|
+
}
|
|
39
|
+
/** 1-based line number of an offset. */
|
|
40
|
+
function lineAt(text, offset) {
|
|
41
|
+
let n = 1;
|
|
42
|
+
for (let i = 0; i < offset && i < text.length; i += 1)
|
|
43
|
+
if (text.charCodeAt(i) === 10)
|
|
44
|
+
n += 1;
|
|
45
|
+
return n;
|
|
46
|
+
}
|
|
47
|
+
/** A fragment for a one-line message: escaped, and bounded. */
|
|
48
|
+
function fragment(s, max = 60) {
|
|
49
|
+
const cut = s.slice(0, max);
|
|
50
|
+
const shown = JSON.stringify(cut).slice(1, -1); // drop the quotes, keep \n and \t visible
|
|
51
|
+
return s.length > max ? `${shown}…` : shown;
|
|
52
|
+
}
|
|
53
|
+
/**
|
|
54
|
+
* The detail lines for a failed search, or "" when there is nothing useful
|
|
55
|
+
* to say. The caller owns the headline; this is what follows it.
|
|
56
|
+
*/
|
|
57
|
+
export function describeSearchMiss(text, search) {
|
|
58
|
+
if (search.length === 0)
|
|
59
|
+
return "";
|
|
60
|
+
const matched = longestMatchingPrefix(text, search);
|
|
61
|
+
if (matched === 0) {
|
|
62
|
+
const firstLine = search.split("\n", 1)[0] ?? "";
|
|
63
|
+
return ` no part of it appears in the file — it begins "${fragment(firstLine)}"`;
|
|
64
|
+
}
|
|
65
|
+
// The FIRST place the prefix appears. Since ACI-2 the tool no longer
|
|
66
|
+
// has first-occurrence semantics to match, so the reason is now this
|
|
67
|
+
// function's own: a miss report needs one location and the earliest is
|
|
68
|
+
// the deterministic choice. A prefix occurring in several places is
|
|
69
|
+
// reported at the earliest of them, which can be further from where the
|
|
70
|
+
// caller was aiming than the report admits.
|
|
71
|
+
const at = text.indexOf(search.slice(0, matched));
|
|
72
|
+
const endOfMatch = at + matched;
|
|
73
|
+
const line = lineAt(text, at);
|
|
74
|
+
const endLine = lineAt(text, endOfMatch);
|
|
75
|
+
const head = ` ${matched} of ${search.length} characters matched, from line ${line} to line ${endLine}`;
|
|
76
|
+
const rest = search.slice(matched);
|
|
77
|
+
// RUNNING PAST THE END IS THE COMMON CASE AND ITS OWN SENTENCE. Four of
|
|
78
|
+
// the ten refusals in the measured session ended exactly here, and
|
|
79
|
+
// rendering that as `the file then has: ""` buries the one fact worth
|
|
80
|
+
// having: there is no more file. A caller that appended what it meant
|
|
81
|
+
// to ADD onto the end of what it meant to FIND reads its own mistake
|
|
82
|
+
// off this line.
|
|
83
|
+
if (endOfMatch >= text.length) {
|
|
84
|
+
return [
|
|
85
|
+
head,
|
|
86
|
+
` the file ENDS there — your search continues for ${rest.length} more characters: "${fragment(rest)}"`,
|
|
87
|
+
].join("\n");
|
|
88
|
+
}
|
|
89
|
+
return [
|
|
90
|
+
head,
|
|
91
|
+
` the file then has: "${fragment(text.slice(endOfMatch))}"`,
|
|
92
|
+
` your search wanted: "${fragment(rest)}"`,
|
|
93
|
+
].join("\n");
|
|
94
|
+
}
|
package/dist/search-worker.d.ts
CHANGED
|
@@ -18,6 +18,12 @@
|
|
|
18
18
|
export interface SearchRequest {
|
|
19
19
|
readonly token: number;
|
|
20
20
|
readonly root: string;
|
|
21
|
+
/** The WORKSPACE root, which is not always the search root: a search under
|
|
22
|
+
* `packages/runtime` must still name `packages/runtime/src/run.ts` so the
|
|
23
|
+
* result can be handed to `read_file` unchanged. Realpath'd by the
|
|
24
|
+
* caller, because `full` is walked from a realpath'd root and a mixed
|
|
25
|
+
* pair produces `../..` the moment a symlink is involved. */
|
|
26
|
+
readonly workspaceRoot: string;
|
|
21
27
|
/** a single file to scan instead of walking `root` */
|
|
22
28
|
readonly single: string | null;
|
|
23
29
|
readonly pattern: string;
|
package/dist/search-worker.js
CHANGED
|
@@ -16,8 +16,33 @@
|
|
|
16
16
|
* message from a superseded worker is ignored.
|
|
17
17
|
*/
|
|
18
18
|
import { open, readdir } from "node:fs/promises";
|
|
19
|
-
import { join, relative } from "node:path";
|
|
19
|
+
import { basename, join, relative } from "node:path";
|
|
20
|
+
import { corpusSkips, layersEntering, readLayer } from "./corpus.js";
|
|
20
21
|
import { isMainThread, parentPort } from "node:worker_threads";
|
|
22
|
+
/** ACI-5 — the excerpt WINDOWS THE MATCH instead of taking the line's head.
|
|
23
|
+
*
|
|
24
|
+
* `line.trim().slice(0, 160)` answers "what does this line start with",
|
|
25
|
+
* and the model asked "where is my pattern". Measured over 171 real
|
|
26
|
+
* search results and 1,931 excerpt lines: 11.7% hit the 160-char cut and
|
|
27
|
+
* 6.1% did not contain the pattern they matched — a hit the model cannot
|
|
28
|
+
* act on without spending a read to find out what it found.
|
|
29
|
+
*
|
|
30
|
+
* A short line is returned exactly as before, markers and all absent, so
|
|
31
|
+
* the common case is byte-identical. */
|
|
32
|
+
const EXCERPT_RADIUS = 80;
|
|
33
|
+
function excerptAround(line, regex) {
|
|
34
|
+
const trimmed = line.trim();
|
|
35
|
+
if (trimmed.length <= EXCERPT_RADIUS * 2)
|
|
36
|
+
return trimmed;
|
|
37
|
+
// A fresh non-global copy: `lastIndex` on a shared /g regex would make
|
|
38
|
+
// the excerpt depend on which line was scanned before it.
|
|
39
|
+
const found = new RegExp(regex.source, regex.flags.replace("g", "")).exec(trimmed);
|
|
40
|
+
const at = found ? found.index : 0;
|
|
41
|
+
const hit = found ? found[0].length : 0;
|
|
42
|
+
const start = Math.max(0, at - EXCERPT_RADIUS);
|
|
43
|
+
const end = Math.min(trimmed.length, at + hit + EXCERPT_RADIUS);
|
|
44
|
+
return `${start > 0 ? "…" : ""}${trimmed.slice(start, end)}${end < trimmed.length ? "…" : ""}`;
|
|
45
|
+
}
|
|
21
46
|
export async function runSearch(req) {
|
|
22
47
|
const regex = new RegExp(req.pattern, req.flags);
|
|
23
48
|
const matches = [];
|
|
@@ -77,8 +102,11 @@ export async function runSearch(req) {
|
|
|
77
102
|
for (const [i, line] of text.split("\n").entries()) {
|
|
78
103
|
if (regex.test(line)) {
|
|
79
104
|
totalMatches += 1;
|
|
105
|
+
// WORKSPACE-RELATIVE, not absolute: `read_file` refuses an
|
|
106
|
+
// absolute path, so an absolute hit here is a result the
|
|
107
|
+
// model cannot feed back without rewriting it by hand.
|
|
80
108
|
if (matches.length < req.maxMatches)
|
|
81
|
-
matches.push(`${full}:${i + 1}: ${line
|
|
109
|
+
matches.push(`${relative(req.workspaceRoot, full) || basename(full)}:${i + 1}: ${excerptAround(line, regex)}`);
|
|
82
110
|
}
|
|
83
111
|
}
|
|
84
112
|
}
|
|
@@ -86,7 +114,11 @@ export async function runSearch(req) {
|
|
|
86
114
|
// unreadable file: skipped, like before
|
|
87
115
|
}
|
|
88
116
|
};
|
|
89
|
-
|
|
117
|
+
// ACI-4/ACI-8: the corpus is declared by a `.gitignore` FILE at the
|
|
118
|
+
// workspace root — not by `.git`, and the walk never goes up.
|
|
119
|
+
const rootLayer = readLayer(req.root);
|
|
120
|
+
const declared = rootLayer !== null;
|
|
121
|
+
const walk = async (dir, depth, layers) => {
|
|
90
122
|
if (depth > 8 || outOfBudget())
|
|
91
123
|
return;
|
|
92
124
|
let entries;
|
|
@@ -101,18 +133,20 @@ export async function runSearch(req) {
|
|
|
101
133
|
}
|
|
102
134
|
throw err;
|
|
103
135
|
}
|
|
136
|
+
const here = depth === 0 ? layers : layersEntering(dir, layers);
|
|
104
137
|
for (const entry of entries) {
|
|
105
138
|
if (outOfBudget())
|
|
106
139
|
return;
|
|
107
|
-
if (entry.name.startsWith(".") || entry.name === "node_modules")
|
|
108
|
-
continue;
|
|
109
140
|
const full = join(dir, entry.name);
|
|
110
|
-
|
|
141
|
+
const isDir = entry.isDirectory();
|
|
142
|
+
if (corpusSkips(declared, here, full, entry.name, isDir))
|
|
143
|
+
continue;
|
|
144
|
+
if (isDir) {
|
|
111
145
|
if (isExcluded(full)) {
|
|
112
146
|
excludedDirs += 1;
|
|
113
147
|
continue;
|
|
114
148
|
}
|
|
115
|
-
await walk(full, depth + 1);
|
|
149
|
+
await walk(full, depth + 1, here);
|
|
116
150
|
}
|
|
117
151
|
else if (entry.isFile())
|
|
118
152
|
await scanFile(full);
|
|
@@ -122,7 +156,7 @@ export async function runSearch(req) {
|
|
|
122
156
|
if (req.single !== null)
|
|
123
157
|
await scanFile(req.single);
|
|
124
158
|
else
|
|
125
|
-
await walk(req.root, 0);
|
|
159
|
+
await walk(req.root, 0, rootLayer === null ? [] : [rootLayer]);
|
|
126
160
|
}
|
|
127
161
|
catch (err) {
|
|
128
162
|
return { token: req.token, matches, totalMatches, filesSeen, skippedFiles, multiLink, unreadableDirs, excludedDirs, stopped, stoppedAt, error: err.message };
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@vincemakes/kiso-tools-node",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.38.0",
|
|
4
4
|
"description": "kiso coding tools for Node hosts \u2014 read file, list directory, search text, write/edit file, shell command.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"license": "MIT",
|
|
@@ -25,7 +25,8 @@
|
|
|
25
25
|
"test": "vitest run"
|
|
26
26
|
},
|
|
27
27
|
"dependencies": {
|
|
28
|
-
"@vincemakes/kiso-core": "0.
|
|
28
|
+
"@vincemakes/kiso-core": "0.38.0",
|
|
29
|
+
"ignore": "^7.0.9"
|
|
29
30
|
},
|
|
30
31
|
"devDependencies": {
|
|
31
32
|
"@types/node": "^26.1.2",
|