knodin 0.10.7 → 0.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/bin/cli.js +11 -4
- package/dist/src/compact-structural.js +8 -3
- package/dist/src/credential-patterns.js +14 -1
- package/dist/src/engine/git-layout.js +53 -0
- package/dist/src/engine/index.js +6 -0
- package/dist/src/engine/parse-pool.js +21 -4
- package/dist/src/engine/seal.js +122 -3
- package/dist/src/engine/sealed-open.js +125 -8
- package/dist/src/engine/text-matches.js +121 -79
- package/dist/src/init.js +28 -3
- package/dist/src/mcp-worker-supervisor.js +10 -2
- package/dist/src/structural-fast-path.js +64 -1
- package/dist/src/tools/knodin-tools.js +16 -9
- package/docs/REPOSITORIES-AND-WORKTREES.md +30 -0
- package/docs/releases/0.10.8.md +165 -0
- package/docs/releases/0.11.0.md +227 -0
- package/package.json +4 -2
|
@@ -25,6 +25,24 @@ import { walkRepoFiles } from "./file-walker.js";
|
|
|
25
25
|
const WORD_CHARACTER = /[\p{L}\p{N}_]/u;
|
|
26
26
|
/** Upper bound on `TextMatch.enclosingText`, so one row cannot flood a preview. */
|
|
27
27
|
const ENCLOSING_TEXT_LIMIT = 120;
|
|
28
|
+
/**
|
|
29
|
+
* Largest file this search will read, in bytes.
|
|
30
|
+
*
|
|
31
|
+
* R1 deliberately drops the source-extension filter, so the walk reaches
|
|
32
|
+
* lockfiles, NDJSON dumps, fixtures and generated JSON as well as prose. Four
|
|
33
|
+
* mebibytes clears every hand-written source or document file by a wide margin
|
|
34
|
+
* while excluding the generated artefacts that would otherwise decide the
|
|
35
|
+
* process's peak memory. Oversized files are reported in `uncovered`, never
|
|
36
|
+
* dropped.
|
|
37
|
+
*/
|
|
38
|
+
export const MAX_SEARCHABLE_FILE_BYTES = 4 * 1024 * 1024;
|
|
39
|
+
/**
|
|
40
|
+
* Largest number of matches this search will accumulate.
|
|
41
|
+
*
|
|
42
|
+
* Past this the result is no longer reviewable and the match objects alone
|
|
43
|
+
* dominate memory; the honest answer is a bounded list plus `truncated: true`.
|
|
44
|
+
*/
|
|
45
|
+
export const MAX_TEXT_MATCHES = 5_000;
|
|
28
46
|
function isWordCharacter(value) {
|
|
29
47
|
return value !== undefined && WORD_CHARACTER.test(value);
|
|
30
48
|
}
|
|
@@ -164,86 +182,101 @@ function classifyAgainstTree(parser, content, raw) {
|
|
|
164
182
|
* over a large repository touches enough files to reproduce it exactly.
|
|
165
183
|
*/
|
|
166
184
|
export async function searchFiles(files, term, loadLanguage, newParser, options = {}) {
|
|
167
|
-
const
|
|
168
|
-
const matches = [];
|
|
169
|
-
const uncovered = [];
|
|
170
|
-
let filesSearched = 0;
|
|
185
|
+
const state = newSearchState();
|
|
171
186
|
for (const { file, content } of files) {
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
|
|
176
|
-
|
|
177
|
-
|
|
187
|
+
await searchOneFile(state, file, content, term, loadLanguage, newParser, options);
|
|
188
|
+
if (state.truncated)
|
|
189
|
+
break;
|
|
190
|
+
}
|
|
191
|
+
return { ...state, searched: true };
|
|
192
|
+
}
|
|
193
|
+
function newSearchState() {
|
|
194
|
+
return { matches: [], uncovered: [], filesSearched: 0, truncated: false };
|
|
195
|
+
}
|
|
196
|
+
/**
|
|
197
|
+
* Search and classify one file, appending to `state`.
|
|
198
|
+
*
|
|
199
|
+
* Takes a single file's content rather than a list so that a repository-wide
|
|
200
|
+
* search never has to hold more than one file in memory at once (KNODIN-24).
|
|
201
|
+
* Sets `state.truncated` when the match cap is reached; the caller is
|
|
202
|
+
* responsible for stopping there.
|
|
203
|
+
*/
|
|
204
|
+
async function searchOneFile(state, file, content, term, loadLanguage, newParser, options) {
|
|
205
|
+
const caseSensitive = options.caseSensitive ?? true;
|
|
206
|
+
const raw = findOccurrences(content, term, caseSensitive);
|
|
207
|
+
state.filesSearched++;
|
|
208
|
+
if (raw.length === 0)
|
|
209
|
+
return;
|
|
210
|
+
let classified = null;
|
|
211
|
+
let language = null;
|
|
212
|
+
try {
|
|
213
|
+
language = await loadLanguage(file);
|
|
214
|
+
}
|
|
215
|
+
catch {
|
|
216
|
+
// A grammar that will not load costs classification, not the match.
|
|
217
|
+
language = null;
|
|
218
|
+
}
|
|
219
|
+
if (language) {
|
|
220
|
+
const parser = newParser();
|
|
178
221
|
try {
|
|
179
|
-
language
|
|
222
|
+
parser.setLanguage(language);
|
|
223
|
+
classified = classifyAgainstTree(parser, content, raw);
|
|
180
224
|
}
|
|
181
|
-
catch {
|
|
182
|
-
|
|
183
|
-
|
|
225
|
+
catch (error) {
|
|
226
|
+
if (isWasmAbort(error))
|
|
227
|
+
throw error;
|
|
228
|
+
// Parsing failed on this file only. The matches are still real and
|
|
229
|
+
// still reported — just without node evidence.
|
|
230
|
+
classified = null;
|
|
231
|
+
state.uncovered.push({
|
|
232
|
+
file,
|
|
233
|
+
reason: `matched but could not be parsed for classification: ${error instanceof Error ? error.message : String(error)}`,
|
|
234
|
+
});
|
|
184
235
|
}
|
|
185
|
-
|
|
186
|
-
const parser = newParser();
|
|
236
|
+
finally {
|
|
187
237
|
try {
|
|
188
|
-
parser.
|
|
189
|
-
classified = classifyAgainstTree(parser, content, raw);
|
|
238
|
+
parser.delete();
|
|
190
239
|
}
|
|
191
|
-
catch
|
|
192
|
-
|
|
193
|
-
throw error;
|
|
194
|
-
// Parsing failed on this file only. The matches are still real and
|
|
195
|
-
// still reported — just without node evidence.
|
|
196
|
-
classified = null;
|
|
197
|
-
uncovered.push({
|
|
198
|
-
file,
|
|
199
|
-
reason: `matched but could not be parsed for classification: ${error instanceof Error ? error.message : String(error)}`,
|
|
200
|
-
});
|
|
201
|
-
}
|
|
202
|
-
finally {
|
|
203
|
-
try {
|
|
204
|
-
parser.delete();
|
|
205
|
-
}
|
|
206
|
-
catch {
|
|
207
|
-
/* a dead module has nothing to reclaim */
|
|
208
|
-
}
|
|
240
|
+
catch {
|
|
241
|
+
/* a dead module has nothing to reclaim */
|
|
209
242
|
}
|
|
210
243
|
}
|
|
211
|
-
|
|
212
|
-
|
|
213
|
-
|
|
214
|
-
|
|
215
|
-
|
|
216
|
-
|
|
217
|
-
|
|
218
|
-
|
|
219
|
-
|
|
220
|
-
|
|
221
|
-
|
|
222
|
-
|
|
223
|
-
|
|
224
|
-
|
|
225
|
-
|
|
226
|
-
|
|
227
|
-
|
|
228
|
-
|
|
229
|
-
|
|
230
|
-
|
|
231
|
-
|
|
232
|
-
|
|
244
|
+
}
|
|
245
|
+
const starts = lineStarts(content);
|
|
246
|
+
for (const [position, match] of raw.entries()) {
|
|
247
|
+
if (state.matches.length >= MAX_TEXT_MATCHES) {
|
|
248
|
+
state.truncated = true;
|
|
249
|
+
return;
|
|
250
|
+
}
|
|
251
|
+
const lineIndex = lineIndexFor(starts, match.startIndex);
|
|
252
|
+
const lineStart = starts[lineIndex];
|
|
253
|
+
const nextStart = starts[lineIndex + 1] ?? content.length + 1;
|
|
254
|
+
const node = classified?.[position] ?? null;
|
|
255
|
+
state.matches.push({
|
|
256
|
+
file,
|
|
257
|
+
line: lineIndex + 1,
|
|
258
|
+
column: match.startIndex - lineStart + 1,
|
|
259
|
+
startIndex: match.startIndex,
|
|
260
|
+
endIndex: match.endIndex,
|
|
261
|
+
lineText: content.slice(lineStart, Math.max(lineStart, nextStart - 1)).replace(/\r$/, ""),
|
|
262
|
+
matchedText: match.matchedText,
|
|
263
|
+
caseMatchesQuery: match.matchedText === term,
|
|
264
|
+
withinLargerWord: isWordCharacter(content[match.startIndex - 1]) || isWordCharacter(content[match.endIndex]),
|
|
265
|
+
classification: node ? node.classification : classifyByFileType(file),
|
|
266
|
+
basis: node ? "parse-node" : language ? "none" : "file-type",
|
|
267
|
+
nodeType: node ? node.nodeType : null,
|
|
268
|
+
enclosingText: node ? node.enclosingText : null,
|
|
233
269
|
});
|
|
234
270
|
}
|
|
235
|
-
return { matches, uncovered, filesSearched, searched: true };
|
|
236
271
|
}
|
|
237
|
-
|
|
238
|
-
* Read a file for searching, or say why it could not be.
|
|
239
|
-
*
|
|
240
|
-
* Binary detection is a NUL-byte sniff over the head of the file rather than an
|
|
241
|
-
* extension list: the extensions that matter here are open-ended, and a
|
|
242
|
-
* misjudged binary read produces garbage matches rather than an honest skip.
|
|
243
|
-
*/
|
|
244
|
-
function readSearchable(absolute) {
|
|
272
|
+
export function readSearchable(absolute) {
|
|
245
273
|
let buffer;
|
|
246
274
|
try {
|
|
275
|
+
const stats = fs.statSync(absolute);
|
|
276
|
+
if (stats.size > MAX_SEARCHABLE_FILE_BYTES)
|
|
277
|
+
return {
|
|
278
|
+
reason: `too large to search: ${stats.size} bytes exceeds the ${MAX_SEARCHABLE_FILE_BYTES}-byte limit`,
|
|
279
|
+
};
|
|
247
280
|
buffer = fs.readFileSync(absolute);
|
|
248
281
|
}
|
|
249
282
|
catch (error) {
|
|
@@ -264,8 +297,15 @@ function readSearchable(absolute) {
|
|
|
264
297
|
* Every file that could not be searched is returned in `uncovered` (R5). A
|
|
265
298
|
* rename that silently skipped files would leave a half-renamed repository
|
|
266
299
|
* looking finished, which is the worst available outcome for this operation.
|
|
300
|
+
*
|
|
301
|
+
* Files are read and searched one at a time. Reading them all up front held the
|
|
302
|
+
* entire non-pruned working tree in memory at once, which on a tree carrying
|
|
303
|
+
* generated dumps or vendored data is gigabytes for a single call (KNODIN-24).
|
|
304
|
+
*
|
|
305
|
+
* `readFile` is injected only so a test can observe that reads interleave with
|
|
306
|
+
* searching rather than preceding it.
|
|
267
307
|
*/
|
|
268
|
-
export async function searchRepoText(repoPath, term, loadLanguage, newParser, options = {}) {
|
|
308
|
+
export async function searchRepoText(repoPath, term, loadLanguage, newParser, options = {}, readFile = readSearchable) {
|
|
269
309
|
let candidates;
|
|
270
310
|
try {
|
|
271
311
|
// Checked explicitly because `walkRepoFiles` swallows a missing directory
|
|
@@ -291,19 +331,21 @@ export async function searchRepoText(repoPath, term, loadLanguage, newParser, op
|
|
|
291
331
|
],
|
|
292
332
|
filesSearched: 0,
|
|
293
333
|
searched: false,
|
|
334
|
+
truncated: false,
|
|
294
335
|
};
|
|
295
336
|
}
|
|
296
|
-
const
|
|
297
|
-
const uncovered = [];
|
|
298
|
-
for (const file of candidates) {
|
|
299
|
-
const outcome = readSearchable(path.join(repoPath, file));
|
|
300
|
-
if ("reason" in outcome)
|
|
301
|
-
uncovered.push({ file, reason: outcome.reason });
|
|
302
|
-
else
|
|
303
|
-
readable.push({ file, content: outcome.content });
|
|
304
|
-
}
|
|
305
|
-
const result = await searchFiles(readable, term, loadLanguage, newParser, options);
|
|
337
|
+
const state = newSearchState();
|
|
306
338
|
// Files skipped at read time and files that failed classification are both
|
|
307
339
|
// gaps in the same claim, so they are reported through one list.
|
|
308
|
-
|
|
340
|
+
for (const file of candidates) {
|
|
341
|
+
const outcome = readFile(path.join(repoPath, file));
|
|
342
|
+
if ("reason" in outcome) {
|
|
343
|
+
state.uncovered.push({ file, reason: outcome.reason });
|
|
344
|
+
continue;
|
|
345
|
+
}
|
|
346
|
+
await searchOneFile(state, file, outcome.content, term, loadLanguage, newParser, options);
|
|
347
|
+
if (state.truncated)
|
|
348
|
+
break;
|
|
349
|
+
}
|
|
350
|
+
return { ...state, searched: true };
|
|
309
351
|
}
|
package/dist/src/init.js
CHANGED
|
@@ -478,8 +478,15 @@ if ! mkdir "$LOCK_DIR" 2>/dev/null; then
|
|
|
478
478
|
if [ -f "$LOCK_DIR/pid" ]; then
|
|
479
479
|
LOCK_PID="$(cat "$LOCK_DIR/pid" 2>/dev/null || true)"
|
|
480
480
|
if [ -n "$LOCK_PID" ] && ! kill -0 "$LOCK_PID" 2>/dev/null; then
|
|
481
|
-
rm -rf "
|
|
482
|
-
|
|
481
|
+
# Claim the stale lock by RENAMING it: "rm -rf then mkdir" leaves a window
|
|
482
|
+
# in which two processes that both saw the same dead pid can each succeed.
|
|
483
|
+
# Only one rename can win, because the loser's source no longer exists.
|
|
484
|
+
if mv "$LOCK_DIR" "$LOCK_DIR.stale.$$" 2>/dev/null; then
|
|
485
|
+
rm -rf "$LOCK_DIR.stale.$$"
|
|
486
|
+
mkdir "$LOCK_DIR" 2>/dev/null || exit 0
|
|
487
|
+
else
|
|
488
|
+
exit 0
|
|
489
|
+
fi
|
|
483
490
|
else
|
|
484
491
|
exit 0
|
|
485
492
|
fi
|
|
@@ -488,7 +495,25 @@ if ! mkdir "$LOCK_DIR" 2>/dev/null; then
|
|
|
488
495
|
fi
|
|
489
496
|
fi
|
|
490
497
|
printf '%s\n' "$$" > "$LOCK_DIR/pid"
|
|
491
|
-
|
|
498
|
+
# Release only a lock this process still owns. The previous handler removed
|
|
499
|
+
# $LOCK_DIR unconditionally, so a script that had already lost the lock deleted
|
|
500
|
+
# its SUCCESSOR's — and init sends SIGTERM here on every drain that outruns its
|
|
501
|
+
# 30s spawnSync timeout, which made that the common case rather than the rare
|
|
502
|
+
# one (KNODIN-23).
|
|
503
|
+
knodin_release_lock() {
|
|
504
|
+
if [ "$(cat "$LOCK_DIR/pid" 2>/dev/null || true)" = "$$" ]; then
|
|
505
|
+
rm -rf "$LOCK_DIR"
|
|
506
|
+
fi
|
|
507
|
+
rm -f "$FAILURE_TMP"
|
|
508
|
+
}
|
|
509
|
+
trap 'knodin_release_lock' EXIT
|
|
510
|
+
# Each signal exits explicitly. POSIX sh defers a trap until the foreground
|
|
511
|
+
# command finishes and then RESUMES where it left off, so a handler that only
|
|
512
|
+
# cleaned up dropped the lock and kept draining — two processors at once, and
|
|
513
|
+
# Ctrl-C failed to cancel a foreground run at all.
|
|
514
|
+
trap 'knodin_release_lock; trap - EXIT; exit 129' HUP
|
|
515
|
+
trap 'knodin_release_lock; trap - EXIT; exit 130' INT
|
|
516
|
+
trap 'knodin_release_lock; trap - EXIT; exit 143' TERM
|
|
492
517
|
LOG_PATH="$REPO_ROOT/.knodin/indexer.log"
|
|
493
518
|
if [ -f "$LOG_PATH" ]; then
|
|
494
519
|
LOG_BYTES="$(wc -c < "$LOG_PATH" 2>/dev/null | tr -d '[:space:]')"
|
|
@@ -214,9 +214,17 @@ export class RepositoryWorker extends EventEmitter {
|
|
|
214
214
|
if (signal.aborted)
|
|
215
215
|
throw codedError("KNODIN_CLIENT_DISCONNECTED", "MCP client disconnected before worker dispatch");
|
|
216
216
|
if (!this.child) {
|
|
217
|
+
// Prune HERE as well as in `exited`. Once this gate throws, no child is
|
|
218
|
+
// spawned, so no exit event ever fires to prune the array — filtering
|
|
219
|
+
// only on exit turned the 60s rate window into a permanent per-repository
|
|
220
|
+
// latch that outlived whatever caused the crashes, and the cached
|
|
221
|
+
// RepositoryWorker is never evicted, so only a gateway restart cleared
|
|
222
|
+
// it (KNODIN-17).
|
|
223
|
+
const startedAt = Date.now();
|
|
224
|
+
this.restartTimes = this.restartTimes.filter((at) => startedAt - at < RESTART_WINDOW_MS);
|
|
217
225
|
if (this.restartTimes.length >= MAX_RESTARTS)
|
|
218
|
-
throw codedError("KNODIN_RESTART_LIMIT",
|
|
219
|
-
this.restartTimes.push(
|
|
226
|
+
throw codedError("KNODIN_RESTART_LIMIT", `graph worker restarted ${MAX_RESTARTS} times within ${RESTART_WINDOW_MS / 1000}s; it will be retried once that window elapses`);
|
|
227
|
+
this.restartTimes.push(startedAt);
|
|
220
228
|
}
|
|
221
229
|
const startup = this.start();
|
|
222
230
|
const remainingForStartup = Math.max(1, deadlineAt - Date.now());
|
|
@@ -169,6 +169,55 @@ function directFile(repo, filePath) {
|
|
|
169
169
|
symbols: directSymbols(filePath, content),
|
|
170
170
|
};
|
|
171
171
|
}
|
|
172
|
+
function directoryOf(filePath) {
|
|
173
|
+
const slash = filePath.lastIndexOf("/");
|
|
174
|
+
return slash === -1 ? "." : filePath.slice(0, slash);
|
|
175
|
+
}
|
|
176
|
+
function extensionOf(name) {
|
|
177
|
+
const dot = name.lastIndexOf(".");
|
|
178
|
+
return dot > 0 ? name.slice(dot) : null;
|
|
179
|
+
}
|
|
180
|
+
/**
|
|
181
|
+
* True when every directory the aggregate covers holds only files the snapshot
|
|
182
|
+
* already knows about.
|
|
183
|
+
*
|
|
184
|
+
* The indexable extensions are taken from the snapshot's own file list rather
|
|
185
|
+
* than hardcoded, so this stays calibrated to whatever the walker actually
|
|
186
|
+
* indexed. Membership is tested against the WHOLE snapshot, not the filtered
|
|
187
|
+
* subset, so a sibling that is indexed but outside the caller's target does not
|
|
188
|
+
* read as an omission.
|
|
189
|
+
*/
|
|
190
|
+
function snapshotCoversDirectories(repo, snapshot, covered) {
|
|
191
|
+
const indexed = new Set((snapshot?.files ?? []).map((file) => file.path));
|
|
192
|
+
if (indexed.size === 0)
|
|
193
|
+
return false;
|
|
194
|
+
const extensions = new Set();
|
|
195
|
+
for (const known of indexed) {
|
|
196
|
+
const extension = extensionOf(known.slice(known.lastIndexOf("/") + 1));
|
|
197
|
+
if (extension)
|
|
198
|
+
extensions.add(extension);
|
|
199
|
+
}
|
|
200
|
+
for (const directory of new Set(covered.map((file) => directoryOf(file.path)))) {
|
|
201
|
+
let entries;
|
|
202
|
+
try {
|
|
203
|
+
entries = fs.readdirSync(path.resolve(repo, directory), { withFileTypes: true });
|
|
204
|
+
}
|
|
205
|
+
catch {
|
|
206
|
+
return false;
|
|
207
|
+
}
|
|
208
|
+
for (const entry of entries) {
|
|
209
|
+
if (!entry.isFile())
|
|
210
|
+
continue;
|
|
211
|
+
const extension = extensionOf(entry.name);
|
|
212
|
+
if (!extension || !extensions.has(extension))
|
|
213
|
+
continue;
|
|
214
|
+
const relative = directory === "." ? entry.name : `${directory}/${entry.name}`;
|
|
215
|
+
if (!indexed.has(relative))
|
|
216
|
+
return false;
|
|
217
|
+
}
|
|
218
|
+
}
|
|
219
|
+
return true;
|
|
220
|
+
}
|
|
172
221
|
export function runStructuralFastPath(repo, pattern, target, limit = 100) {
|
|
173
222
|
const snapshot = readStructuralSnapshot(repo);
|
|
174
223
|
if (!snapshot && pattern !== "file_summary")
|
|
@@ -195,7 +244,7 @@ export function runStructuralFastPath(repo, pattern, target, limit = 100) {
|
|
|
195
244
|
let fingerprint;
|
|
196
245
|
let fresh = false;
|
|
197
246
|
if (pattern === "project_overview") {
|
|
198
|
-
|
|
247
|
+
const unchanged = matching.every((file) => {
|
|
199
248
|
try {
|
|
200
249
|
const stat = fs.statSync(path.resolve(repo, file.path));
|
|
201
250
|
return stat.size === file.sizeBytes && stat.mtimeMs === file.mtimeMs;
|
|
@@ -204,6 +253,20 @@ export function runStructuralFastPath(repo, pattern, target, limit = 100) {
|
|
|
204
253
|
return false;
|
|
205
254
|
}
|
|
206
255
|
});
|
|
256
|
+
// A size+mtime sweep can only ever prove that the files the snapshot
|
|
257
|
+
// ALREADY knows about are unchanged. A file added since the snapshot is not
|
|
258
|
+
// in that set, so it was never checked, never counted, and never mentioned
|
|
259
|
+
// — while the overview still reported itself "fresh" (KNODIN-25). So also
|
|
260
|
+
// read back the directories being aggregated and look for an indexable
|
|
261
|
+
// sibling the snapshot has never seen.
|
|
262
|
+
//
|
|
263
|
+
// Known bound: this sees additions to directories the snapshot already
|
|
264
|
+
// covers, which is where source files are overwhelmingly added. A brand new
|
|
265
|
+
// top-level directory is still missed, because recognising one would mean
|
|
266
|
+
// re-deriving the walker's prune rules here and re-walking the tree, which
|
|
267
|
+
// is the cost this fast path exists to avoid. `upgrade` already points at
|
|
268
|
+
// the full `architecture_overview` query for callers that need certainty.
|
|
269
|
+
fresh = unchanged && snapshotCoversDirectories(repo, snapshot, matching);
|
|
207
270
|
const directories = new Map();
|
|
208
271
|
for (const file of matching) {
|
|
209
272
|
const directory = file.path.includes("/") ? file.path.split("/", 1)[0] : ".";
|
|
@@ -70,10 +70,8 @@ export async function closeKnodinToolEngine() {
|
|
|
70
70
|
if (initializedEngine)
|
|
71
71
|
await initializedEngine.close();
|
|
72
72
|
initializedEngine = null;
|
|
73
|
-
compactReadyRepos.clear();
|
|
74
73
|
}
|
|
75
74
|
const localTelemetry = [];
|
|
76
|
-
const compactReadyRepos = new Set();
|
|
77
75
|
let cachedSchemaTokens;
|
|
78
76
|
function gatewaySchemaTokens() {
|
|
79
77
|
cachedSchemaTokens ??= countOutputTokens(JSON.stringify(buildDocumentedKnodinTools()[0]?.inputSchema ?? {}));
|
|
@@ -82,15 +80,24 @@ function gatewaySchemaTokens() {
|
|
|
82
80
|
function inspectGatewayGraphHealth(repo) {
|
|
83
81
|
return inspectGraphQueryHealth(repo, async (target) => attachLifecycleHealth(target, await engine.status(target, { audit: "cached" })));
|
|
84
82
|
}
|
|
83
|
+
/**
|
|
84
|
+
* Probe on EVERY compact call, exactly as the non-compact path does.
|
|
85
|
+
*
|
|
86
|
+
* This used to latch a repository as ready after one successful probe and never
|
|
87
|
+
* look again for the life of the process. Nothing invalidated that latch —
|
|
88
|
+
* `closeKnodinToolEngine` clears it but has no production caller — so once
|
|
89
|
+
* `.knodin/` was deleted or re-inited from another terminal, a long-lived MCP
|
|
90
|
+
* server kept answering compact queries against a database the open path had
|
|
91
|
+
* silently recreated as an empty schema (`CREATE TABLE IF NOT EXISTS`). The
|
|
92
|
+
* result was "no matches": indistinguishable from a true negative, and the exact
|
|
93
|
+
* confident-but-wrong answer this product exists to avoid (KNODIN-18).
|
|
94
|
+
*
|
|
95
|
+
* The probe is cheap — `engine.status(..., { audit: "cached" })` — so the latch
|
|
96
|
+
* was buying very little in exchange for that failure mode.
|
|
97
|
+
*/
|
|
85
98
|
async function compactGraphUnavailable(repo) {
|
|
86
|
-
const key = nodePath.resolve(repo);
|
|
87
|
-
if (compactReadyRepos.has(key))
|
|
88
|
-
return null;
|
|
89
99
|
const health = await inspectGatewayGraphHealth(repo);
|
|
90
|
-
|
|
91
|
-
return health;
|
|
92
|
-
compactReadyRepos.add(key);
|
|
93
|
-
return null;
|
|
100
|
+
return health.available ? null : health;
|
|
94
101
|
}
|
|
95
102
|
/** Runs `fn`, converting any thrown error into a structured `{ error }` result
|
|
96
103
|
* instead of letting it escape into the stdio transport. Extracted from the
|
|
@@ -24,6 +24,36 @@ Results distinguish:
|
|
|
24
24
|
- stale linked-worktree metadata;
|
|
25
25
|
- an unrelated repository under the same discovery root.
|
|
26
26
|
|
|
27
|
+
### Graphs are per-worktree
|
|
28
|
+
|
|
29
|
+
A linked worktree carries its own `.knodin` with its own graph, `indexedHead`,
|
|
30
|
+
and freshness. A healthy main checkout does not cover it, so a worktree created
|
|
31
|
+
by `git worktree add` has **no graph until `knodin init` runs inside it**.
|
|
32
|
+
|
|
33
|
+
`git worktree add` creates no `.knodin` of its own — the directory is knodin's,
|
|
34
|
+
written by `init` and by the managed lifecycle hooks. Where those hooks are
|
|
35
|
+
already installed in the repository, a new worktree can therefore appear to have
|
|
36
|
+
a `.knodin` while holding no graph at all, which is the more confusing shape of
|
|
37
|
+
this: the directory exists, and it is empty of anything that can answer.
|
|
38
|
+
|
|
39
|
+
That does not mean a new worktree pays for a full index. `init` selects the
|
|
40
|
+
already-indexed sibling worktree closest in commit distance, reflink-copies its
|
|
41
|
+
baseline where the filesystem supports it, reconciles only the paths that differ
|
|
42
|
+
between that sibling's `indexedHead` and this HEAD, deep-audits the result, and
|
|
43
|
+
promotes it only if it is healthy — falling back to a full index when no
|
|
44
|
+
schema- and build-compatible sibling exists. `KNODIN_DISABLE_WORKTREE_SEED=1`
|
|
45
|
+
forces the full path.
|
|
46
|
+
|
|
47
|
+
Seeding is on the `init` path only, which is why an uninitialized worktree is
|
|
48
|
+
steered to `init` rather than `repair`: both produce a correct index, but
|
|
49
|
+
`repair` rebuilds from scratch.
|
|
50
|
+
|
|
51
|
+
Nothing silently answers from the wrong graph in that state: structural queries
|
|
52
|
+
refuse with `available: false`, `state: not-initialized`, and exit code 1 rather
|
|
53
|
+
than returning an empty result, and status leads its remediation with the
|
|
54
|
+
per-worktree model and `knodin init`. Read a refusal as "this checkout was never
|
|
55
|
+
indexed", never as "there is nothing here to find".
|
|
56
|
+
|
|
27
57
|
### Opt-in repository signals
|
|
28
58
|
|
|
29
59
|
`repos discover --json --signals` adds a deterministic `signals` object to
|
|
@@ -0,0 +1,165 @@
|
|
|
1
|
+
# knodin 0.10.8
|
|
2
|
+
|
|
3
|
+
One change, prompted by a field report from an agent session that built a
|
|
4
|
+
duplicate implementation of a hook that already existed. Most of that report
|
|
5
|
+
described behaviour knodin already has; the part it got right is fixed here.
|
|
6
|
+
|
|
7
|
+
## A linked worktree with no graph now says so in its remediation
|
|
8
|
+
|
|
9
|
+
`git worktree add` gives you a checkout with no graph in it. Graphs are
|
|
10
|
+
**per-worktree**, so a perfectly healthy main checkout tells you nothing about
|
|
11
|
+
the worktree you are standing in — and that makes the missing graph *more*
|
|
12
|
+
surprising, not less.
|
|
13
|
+
|
|
14
|
+
Git creates no `.knodin`; that directory is knodin's, written by `init` and by
|
|
15
|
+
the managed lifecycle hooks. Where those hooks are already installed, a fresh
|
|
16
|
+
worktree can present a `.knodin` that holds no graph — which is the more
|
|
17
|
+
confusing shape of this, and the one the original report hit.
|
|
18
|
+
|
|
19
|
+
Until now, status in that state produced generic remediation:
|
|
20
|
+
|
|
21
|
+
```
|
|
22
|
+
remediation:
|
|
23
|
+
- Run `knodin repair` to create the local index.
|
|
24
|
+
- Run `knodin status` again to verify health.
|
|
25
|
+
```
|
|
26
|
+
|
|
27
|
+
That guidance works — `repair` does build the index from cold, and then tells
|
|
28
|
+
you to run `init` for lifecycle routing. What it does not do is explain *why*
|
|
29
|
+
there was no index in a repository whose main checkout is fine. A reader who
|
|
30
|
+
does not already know the per-worktree model has no way to tell "this checkout
|
|
31
|
+
was never indexed" from "there is nothing here to find", and the second reading
|
|
32
|
+
is the one that quietly confirms whatever hypothesis sent them looking.
|
|
33
|
+
|
|
34
|
+
A linked worktree now leads with the model and the fix, on both the
|
|
35
|
+
machine-readable `remediation` and the human `status` line:
|
|
36
|
+
|
|
37
|
+
```
|
|
38
|
+
Run `knodin init` here: graphs are per-worktree and this one has none of its
|
|
39
|
+
own, so the main checkout's graph does not cover it. `init` seeds from an
|
|
40
|
+
indexed sibling worktree where one exists rather than rebuilding from scratch.
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
Ordinary checkouts are unchanged. Detection reads the `.git` file's `gitdir:`
|
|
44
|
+
pointer and requires a `commondir` beside it — no subprocess, no registry read,
|
|
45
|
+
and it works on a repository too damaged to answer anything else.
|
|
46
|
+
|
|
47
|
+
A `.git` **file** alone is not enough, and treating it as enough was wrong in
|
|
48
|
+
the first cut of this fix: submodules and `--separate-git-dir` clones use one
|
|
49
|
+
too, and neither has a main checkout whose graph could cover it, so they would
|
|
50
|
+
have been handed an explanation that does not apply to them. `commondir` is
|
|
51
|
+
what actually distinguishes a linked worktree.
|
|
52
|
+
|
|
53
|
+
The check lives in a new `src/engine/git-layout.ts` rather than in the engine,
|
|
54
|
+
for a reason worth stating. `engine/index.ts` carries a pinned budget on direct
|
|
55
|
+
`fs` reads, because a sealed artifact answers with no working tree and a reader
|
|
56
|
+
that touches the filesystem there gets ENOENT — which the nearest guard turns
|
|
57
|
+
into `""` or `null`, and which reads downstream as "this symbol has no body"
|
|
58
|
+
rather than "this was not covered". Git *metadata* is categorically outside
|
|
59
|
+
that concern: a sealed artifact has no `.git` at all, so routing these reads
|
|
60
|
+
through the sealed resolver would add a branch that can never execute. Saying
|
|
61
|
+
so in the layout is more honest than spending budget the engine reserves for
|
|
62
|
+
source reads — and it makes the check directly unit-testable, which it was not
|
|
63
|
+
as an engine-private helper.
|
|
64
|
+
|
|
65
|
+
Its spec covers every branch (100% of statements and branches): a plain
|
|
66
|
+
directory, an ordinary checkout, a real linked worktree, a real submodule, a
|
|
67
|
+
real `--separate-git-dir` clone, a malformed `.git` file naming no gitdir, and a
|
|
68
|
+
dangling pointer whose administrative directory was pruned.
|
|
69
|
+
|
|
70
|
+
### Why `init` and not `repair` here
|
|
71
|
+
|
|
72
|
+
`repair` does build the index from cold, and for an ordinary checkout it stays
|
|
73
|
+
the right advice. In a linked worktree it is the *expensive* answer: seeding
|
|
74
|
+
lives on the `init` path only (`indexOrSeed`), so `repair` rebuilds from scratch
|
|
75
|
+
while `init` reflink-copies an indexed sibling's baseline, reconciles only the
|
|
76
|
+
paths that differ between that sibling's `indexedHead` and this HEAD,
|
|
77
|
+
deep-audits the candidate, and promotes it only if it passes — falling back to a
|
|
78
|
+
full index when no compatible sibling exists.
|
|
79
|
+
|
|
80
|
+
The human `status` renderer had been computing its own one-line advice and
|
|
81
|
+
ignoring `repairSteps` entirely, so the first fix reached the JSON surface and
|
|
82
|
+
not the line a person actually reads. It now defers to the step the engine
|
|
83
|
+
already wrote. Measured on a 20-file fixture, `init` in a fresh worktree with an
|
|
84
|
+
indexed sibling completes in about three seconds.
|
|
85
|
+
|
|
86
|
+
Worth stating plainly, because the surprise runs the other way from the cost:
|
|
87
|
+
the per-worktree model does **not** mean a new worktree pays for a full index.
|
|
88
|
+
|
|
89
|
+
## What the same report claimed that measurement did not support
|
|
90
|
+
|
|
91
|
+
Recorded because acting on any of these would have been a regression, and
|
|
92
|
+
because each was reported in good faith from accumulated notes rather than from
|
|
93
|
+
a run against this release.
|
|
94
|
+
|
|
95
|
+
**"Structural queries return empty rather than erroring on an uninitialized
|
|
96
|
+
graph."** They do not, and have not for some time. Against a committed
|
|
97
|
+
repository with no index, `knodin query callers_of <symbol>` returns
|
|
98
|
+
`available: false`, `status: unavailable`, `state: not-initialized`, the full
|
|
99
|
+
status envelope, and **exit code 1**. The same holds for `impact`, `dead_code`,
|
|
100
|
+
`tests_for`, and `file_summary`, on both the CLI and the MCP gateway — both
|
|
101
|
+
route through the same `inspectGraphQueryHealth` gate before and after the
|
|
102
|
+
operation runs. There was no empty result set to fix.
|
|
103
|
+
|
|
104
|
+
This one matters most, because an empty structural result that agrees with your
|
|
105
|
+
hypothesis is the worst failure this tool could have. It is worth restating that
|
|
106
|
+
the gate is there, and that it fails loudly on both surfaces.
|
|
107
|
+
|
|
108
|
+
**"`knodin init --scope personal` exits 0 having done nothing."** It exits **1**.
|
|
109
|
+
The refusal is thrown, and every non-`agent-event` throw reaches the top-level
|
|
110
|
+
handler, which prints to stderr and exits 1. Reproduced against a repository with
|
|
111
|
+
tracked team integration active.
|
|
112
|
+
|
|
113
|
+
**"A backgrounded `knodin repair` produces no progress output."** Repair emits
|
|
114
|
+
ordered per-phase progress on stderr — audit, planning, file reconciliation,
|
|
115
|
+
symbol identities, embeddings, orphan cleanup, verification — and finishes with a
|
|
116
|
+
verified summary naming indexed file and symbol counts.
|
|
117
|
+
|
|
118
|
+
Both of these turned out to have the same cause, since confirmed by the
|
|
119
|
+
reporter: the commands were run through `| tail`. A pipeline's exit status is the
|
|
120
|
+
last command's, so `init`'s 1 was recorded as `tail`'s 0; and `tail` flushes
|
|
121
|
+
nothing until the stream closes, so a capture file read mid-run looked empty
|
|
122
|
+
while repair was in fact reporting every phase. Neither surface changed. Worth
|
|
123
|
+
naming because one habit produced two bug reports against the wrong component.
|
|
124
|
+
|
|
125
|
+
## Not changed here: semantic search ranking
|
|
126
|
+
|
|
127
|
+
A follow-up from the same session measured something real and unaddressed: on a
|
|
128
|
+
healthy graph, `"hands-free voice conversation mode hook with barge-in"` ranked a
|
|
129
|
+
constant list (`BARGE_IN_PHRASES`, 0.588) and the caller's own minutes-old
|
|
130
|
+
duplicate (0.552) above the canonical 513-line implementation, which did not
|
|
131
|
+
appear in the top hits at all. Mechanism-shaped phrasing
|
|
132
|
+
(`"voice pipeline auto restart mic after TTS"`) put the canonical hook at #2,
|
|
133
|
+
0.517.
|
|
134
|
+
|
|
135
|
+
So feature-shaped phrasing — the phrasing a developer uses when asking "does this
|
|
136
|
+
already exist?" — ranked worst for exactly that question, and surfaced the new
|
|
137
|
+
duplicate as apparent confirmation. Centrality data the fix would need
|
|
138
|
+
(in-degree, community membership) is already in the graph and already in the
|
|
139
|
+
output; it is not weighted into ranking.
|
|
140
|
+
|
|
141
|
+
That is a ranking change with its own evaluation burden and it is not bundled
|
|
142
|
+
into a remediation-string fix. It is recorded here so it is not rediscovered from
|
|
143
|
+
scratch.
|
|
144
|
+
|
|
145
|
+
## The pre-push test gate was not gating
|
|
146
|
+
|
|
147
|
+
Found because this release's own broken commit sailed through it.
|
|
148
|
+
|
|
149
|
+
`lefthook.yml`'s `quality-gates` block ran `bun run test:coverage` and then
|
|
150
|
+
`sh scripts/run-sonar-scan.sh` as two lines of one shell block with no
|
|
151
|
+
`set -e`. A shell block's exit status is its **last** command's, so a failing
|
|
152
|
+
suite followed by a passing Sonar scan reported success, and the push was
|
|
153
|
+
allowed. The suite printed `1 failed | 2058 passed` and
|
|
154
|
+
`script "test:coverage" exited with code 1`, and the push completed anyway.
|
|
155
|
+
|
|
156
|
+
`set -e` now leads the block. Coverage thresholds and Sonar were unaffected;
|
|
157
|
+
what was broken is that a red suite could not stop a push.
|
|
158
|
+
|
|
159
|
+
## Verification
|
|
160
|
+
|
|
161
|
+
`npm test`, `npm run lint`, and `npm run typecheck` pass. The new behaviour is
|
|
162
|
+
pinned by a test in `src/__tests__/unit/index-health.spec.ts` that builds a real
|
|
163
|
+
linked worktree with `git worktree add`, asserts the per-worktree line leads its
|
|
164
|
+
remediation, and asserts the ordinary checkout keeps repair-first guidance with
|
|
165
|
+
no per-worktree line.
|