knodin 0.10.7 → 0.11.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -25,6 +25,24 @@ import { walkRepoFiles } from "./file-walker.js";
25
25
  const WORD_CHARACTER = /[\p{L}\p{N}_]/u;
26
26
  /** Upper bound on `TextMatch.enclosingText`, so one row cannot flood a preview. */
27
27
  const ENCLOSING_TEXT_LIMIT = 120;
28
+ /**
29
+ * Largest file this search will read, in bytes.
30
+ *
31
+ * R1 deliberately drops the source-extension filter, so the walk reaches
32
+ * lockfiles, NDJSON dumps, fixtures and generated JSON as well as prose. Four
33
+ * mebibytes clears every hand-written source or document file by a wide margin
34
+ * while excluding the generated artefacts that would otherwise decide the
35
+ * process's peak memory. Oversized files are reported in `uncovered`, never
36
+ * dropped.
37
+ */
38
+ export const MAX_SEARCHABLE_FILE_BYTES = 4 * 1024 * 1024;
39
+ /**
40
+ * Largest number of matches this search will accumulate.
41
+ *
42
+ * Past this the result is no longer reviewable and the match objects alone
43
+ * dominate memory; the honest answer is a bounded list plus `truncated: true`.
44
+ */
45
+ export const MAX_TEXT_MATCHES = 5_000;
28
46
  function isWordCharacter(value) {
29
47
  return value !== undefined && WORD_CHARACTER.test(value);
30
48
  }
@@ -164,86 +182,101 @@ function classifyAgainstTree(parser, content, raw) {
164
182
  * over a large repository touches enough files to reproduce it exactly.
165
183
  */
166
184
  export async function searchFiles(files, term, loadLanguage, newParser, options = {}) {
167
- const caseSensitive = options.caseSensitive ?? true;
168
- const matches = [];
169
- const uncovered = [];
170
- let filesSearched = 0;
185
+ const state = newSearchState();
171
186
  for (const { file, content } of files) {
172
- const raw = findOccurrences(content, term, caseSensitive);
173
- filesSearched++;
174
- if (raw.length === 0)
175
- continue;
176
- let classified = null;
177
- let language = null;
187
+ await searchOneFile(state, file, content, term, loadLanguage, newParser, options);
188
+ if (state.truncated)
189
+ break;
190
+ }
191
+ return { ...state, searched: true };
192
+ }
193
+ function newSearchState() {
194
+ return { matches: [], uncovered: [], filesSearched: 0, truncated: false };
195
+ }
196
+ /**
197
+ * Search and classify one file, appending to `state`.
198
+ *
199
+ * Takes a single file's content rather than a list so that a repository-wide
200
+ * search never has to hold more than one file in memory at once (KNODIN-24).
201
+ * Sets `state.truncated` when the match cap is reached; the caller is
202
+ * responsible for stopping there.
203
+ */
204
+ async function searchOneFile(state, file, content, term, loadLanguage, newParser, options) {
205
+ const caseSensitive = options.caseSensitive ?? true;
206
+ const raw = findOccurrences(content, term, caseSensitive);
207
+ state.filesSearched++;
208
+ if (raw.length === 0)
209
+ return;
210
+ let classified = null;
211
+ let language = null;
212
+ try {
213
+ language = await loadLanguage(file);
214
+ }
215
+ catch {
216
+ // A grammar that will not load costs classification, not the match.
217
+ language = null;
218
+ }
219
+ if (language) {
220
+ const parser = newParser();
178
221
  try {
179
- language = await loadLanguage(file);
222
+ parser.setLanguage(language);
223
+ classified = classifyAgainstTree(parser, content, raw);
180
224
  }
181
- catch {
182
- // A grammar that will not load costs classification, not the match.
183
- language = null;
225
+ catch (error) {
226
+ if (isWasmAbort(error))
227
+ throw error;
228
+ // Parsing failed on this file only. The matches are still real and
229
+ // still reported — just without node evidence.
230
+ classified = null;
231
+ state.uncovered.push({
232
+ file,
233
+ reason: `matched but could not be parsed for classification: ${error instanceof Error ? error.message : String(error)}`,
234
+ });
184
235
  }
185
- if (language) {
186
- const parser = newParser();
236
+ finally {
187
237
  try {
188
- parser.setLanguage(language);
189
- classified = classifyAgainstTree(parser, content, raw);
238
+ parser.delete();
190
239
  }
191
- catch (error) {
192
- if (isWasmAbort(error))
193
- throw error;
194
- // Parsing failed on this file only. The matches are still real and
195
- // still reported — just without node evidence.
196
- classified = null;
197
- uncovered.push({
198
- file,
199
- reason: `matched but could not be parsed for classification: ${error instanceof Error ? error.message : String(error)}`,
200
- });
201
- }
202
- finally {
203
- try {
204
- parser.delete();
205
- }
206
- catch {
207
- /* a dead module has nothing to reclaim */
208
- }
240
+ catch {
241
+ /* a dead module has nothing to reclaim */
209
242
  }
210
243
  }
211
- const starts = lineStarts(content);
212
- raw.forEach((match, position) => {
213
- const lineIndex = lineIndexFor(starts, match.startIndex);
214
- const lineStart = starts[lineIndex];
215
- const nextStart = starts[lineIndex + 1] ?? content.length + 1;
216
- const node = classified?.[position] ?? null;
217
- matches.push({
218
- file,
219
- line: lineIndex + 1,
220
- column: match.startIndex - lineStart + 1,
221
- startIndex: match.startIndex,
222
- endIndex: match.endIndex,
223
- lineText: content.slice(lineStart, Math.max(lineStart, nextStart - 1)).replace(/\r$/, ""),
224
- matchedText: match.matchedText,
225
- caseMatchesQuery: match.matchedText === term,
226
- withinLargerWord: isWordCharacter(content[match.startIndex - 1]) ||
227
- isWordCharacter(content[match.endIndex]),
228
- classification: node ? node.classification : classifyByFileType(file),
229
- basis: node ? "parse-node" : language ? "none" : "file-type",
230
- nodeType: node ? node.nodeType : null,
231
- enclosingText: node ? node.enclosingText : null,
232
- });
244
+ }
245
+ const starts = lineStarts(content);
246
+ for (const [position, match] of raw.entries()) {
247
+ if (state.matches.length >= MAX_TEXT_MATCHES) {
248
+ state.truncated = true;
249
+ return;
250
+ }
251
+ const lineIndex = lineIndexFor(starts, match.startIndex);
252
+ const lineStart = starts[lineIndex];
253
+ const nextStart = starts[lineIndex + 1] ?? content.length + 1;
254
+ const node = classified?.[position] ?? null;
255
+ state.matches.push({
256
+ file,
257
+ line: lineIndex + 1,
258
+ column: match.startIndex - lineStart + 1,
259
+ startIndex: match.startIndex,
260
+ endIndex: match.endIndex,
261
+ lineText: content.slice(lineStart, Math.max(lineStart, nextStart - 1)).replace(/\r$/, ""),
262
+ matchedText: match.matchedText,
263
+ caseMatchesQuery: match.matchedText === term,
264
+ withinLargerWord: isWordCharacter(content[match.startIndex - 1]) || isWordCharacter(content[match.endIndex]),
265
+ classification: node ? node.classification : classifyByFileType(file),
266
+ basis: node ? "parse-node" : language ? "none" : "file-type",
267
+ nodeType: node ? node.nodeType : null,
268
+ enclosingText: node ? node.enclosingText : null,
233
269
  });
234
270
  }
235
- return { matches, uncovered, filesSearched, searched: true };
236
271
  }
237
- /**
238
- * Read a file for searching, or say why it could not be.
239
- *
240
- * Binary detection is a NUL-byte sniff over the head of the file rather than an
241
- * extension list: the extensions that matter here are open-ended, and a
242
- * misjudged binary read produces garbage matches rather than an honest skip.
243
- */
244
- function readSearchable(absolute) {
272
+ export function readSearchable(absolute) {
245
273
  let buffer;
246
274
  try {
275
+ const stats = fs.statSync(absolute);
276
+ if (stats.size > MAX_SEARCHABLE_FILE_BYTES)
277
+ return {
278
+ reason: `too large to search: ${stats.size} bytes exceeds the ${MAX_SEARCHABLE_FILE_BYTES}-byte limit`,
279
+ };
247
280
  buffer = fs.readFileSync(absolute);
248
281
  }
249
282
  catch (error) {
@@ -264,8 +297,15 @@ function readSearchable(absolute) {
264
297
  * Every file that could not be searched is returned in `uncovered` (R5). A
265
298
  * rename that silently skipped files would leave a half-renamed repository
266
299
  * looking finished, which is the worst available outcome for this operation.
300
+ *
301
+ * Files are read and searched one at a time. Reading them all up front held the
302
+ * entire non-pruned working tree in memory at once, which on a tree carrying
303
+ * generated dumps or vendored data is gigabytes for a single call (KNODIN-24).
304
+ *
305
+ * `readFile` is injected only so a test can observe that reads interleave with
306
+ * searching rather than preceding it.
267
307
  */
268
- export async function searchRepoText(repoPath, term, loadLanguage, newParser, options = {}) {
308
+ export async function searchRepoText(repoPath, term, loadLanguage, newParser, options = {}, readFile = readSearchable) {
269
309
  let candidates;
270
310
  try {
271
311
  // Checked explicitly because `walkRepoFiles` swallows a missing directory
@@ -291,19 +331,21 @@ export async function searchRepoText(repoPath, term, loadLanguage, newParser, op
291
331
  ],
292
332
  filesSearched: 0,
293
333
  searched: false,
334
+ truncated: false,
294
335
  };
295
336
  }
296
- const readable = [];
297
- const uncovered = [];
298
- for (const file of candidates) {
299
- const outcome = readSearchable(path.join(repoPath, file));
300
- if ("reason" in outcome)
301
- uncovered.push({ file, reason: outcome.reason });
302
- else
303
- readable.push({ file, content: outcome.content });
304
- }
305
- const result = await searchFiles(readable, term, loadLanguage, newParser, options);
337
+ const state = newSearchState();
306
338
  // Files skipped at read time and files that failed classification are both
307
339
  // gaps in the same claim, so they are reported through one list.
308
- return { ...result, uncovered: [...uncovered, ...result.uncovered] };
340
+ for (const file of candidates) {
341
+ const outcome = readFile(path.join(repoPath, file));
342
+ if ("reason" in outcome) {
343
+ state.uncovered.push({ file, reason: outcome.reason });
344
+ continue;
345
+ }
346
+ await searchOneFile(state, file, outcome.content, term, loadLanguage, newParser, options);
347
+ if (state.truncated)
348
+ break;
349
+ }
350
+ return { ...state, searched: true };
309
351
  }
package/dist/src/init.js CHANGED
@@ -478,8 +478,15 @@ if ! mkdir "$LOCK_DIR" 2>/dev/null; then
478
478
  if [ -f "$LOCK_DIR/pid" ]; then
479
479
  LOCK_PID="$(cat "$LOCK_DIR/pid" 2>/dev/null || true)"
480
480
  if [ -n "$LOCK_PID" ] && ! kill -0 "$LOCK_PID" 2>/dev/null; then
481
- rm -rf "$LOCK_DIR"
482
- mkdir "$LOCK_DIR" 2>/dev/null || exit 0
481
+ # Claim the stale lock by RENAMING it: "rm -rf then mkdir" leaves a window
482
+ # in which two processes that both saw the same dead pid can each succeed.
483
+ # Only one rename can win, because the loser's source no longer exists.
484
+ if mv "$LOCK_DIR" "$LOCK_DIR.stale.$$" 2>/dev/null; then
485
+ rm -rf "$LOCK_DIR.stale.$$"
486
+ mkdir "$LOCK_DIR" 2>/dev/null || exit 0
487
+ else
488
+ exit 0
489
+ fi
483
490
  else
484
491
  exit 0
485
492
  fi
@@ -488,7 +495,25 @@ if ! mkdir "$LOCK_DIR" 2>/dev/null; then
488
495
  fi
489
496
  fi
490
497
  printf '%s\n' "$$" > "$LOCK_DIR/pid"
491
- trap 'rm -rf "$LOCK_DIR"; rm -f "$FAILURE_TMP"' EXIT HUP INT TERM
498
+ # Release only a lock this process still owns. The previous handler removed
499
+ # $LOCK_DIR unconditionally, so a script that had already lost the lock deleted
500
+ # its SUCCESSOR's — and init sends SIGTERM here on every drain that outruns its
501
+ # 30s spawnSync timeout, which made that the common case rather than the rare
502
+ # one (KNODIN-23).
503
+ knodin_release_lock() {
504
+ if [ "$(cat "$LOCK_DIR/pid" 2>/dev/null || true)" = "$$" ]; then
505
+ rm -rf "$LOCK_DIR"
506
+ fi
507
+ rm -f "$FAILURE_TMP"
508
+ }
509
+ trap 'knodin_release_lock' EXIT
510
+ # Each signal exits explicitly. POSIX sh defers a trap until the foreground
511
+ # command finishes and then RESUMES where it left off, so a handler that only
512
+ # cleaned up dropped the lock and kept draining — two processors at once, and
513
+ # Ctrl-C failed to cancel a foreground run at all.
514
+ trap 'knodin_release_lock; trap - EXIT; exit 129' HUP
515
+ trap 'knodin_release_lock; trap - EXIT; exit 130' INT
516
+ trap 'knodin_release_lock; trap - EXIT; exit 143' TERM
492
517
  LOG_PATH="$REPO_ROOT/.knodin/indexer.log"
493
518
  if [ -f "$LOG_PATH" ]; then
494
519
  LOG_BYTES="$(wc -c < "$LOG_PATH" 2>/dev/null | tr -d '[:space:]')"
@@ -214,9 +214,17 @@ export class RepositoryWorker extends EventEmitter {
214
214
  if (signal.aborted)
215
215
  throw codedError("KNODIN_CLIENT_DISCONNECTED", "MCP client disconnected before worker dispatch");
216
216
  if (!this.child) {
217
+ // Prune HERE as well as in `exited`. Once this gate throws, no child is
218
+ // spawned, so no exit event ever fires to prune the array — filtering
219
+ // only on exit turned the 60s rate window into a permanent per-repository
220
+ // latch that outlived whatever caused the crashes, and the cached
221
+ // RepositoryWorker is never evicted, so only a gateway restart cleared
222
+ // it (KNODIN-17).
223
+ const startedAt = Date.now();
224
+ this.restartTimes = this.restartTimes.filter((at) => startedAt - at < RESTART_WINDOW_MS);
217
225
  if (this.restartTimes.length >= MAX_RESTARTS)
218
- throw codedError("KNODIN_RESTART_LIMIT", "graph worker restart limit reached");
219
- this.restartTimes.push(Date.now());
226
+ throw codedError("KNODIN_RESTART_LIMIT", `graph worker restarted ${MAX_RESTARTS} times within ${RESTART_WINDOW_MS / 1000}s; it will be retried once that window elapses`);
227
+ this.restartTimes.push(startedAt);
220
228
  }
221
229
  const startup = this.start();
222
230
  const remainingForStartup = Math.max(1, deadlineAt - Date.now());
@@ -169,6 +169,55 @@ function directFile(repo, filePath) {
169
169
  symbols: directSymbols(filePath, content),
170
170
  };
171
171
  }
172
+ function directoryOf(filePath) {
173
+ const slash = filePath.lastIndexOf("/");
174
+ return slash === -1 ? "." : filePath.slice(0, slash);
175
+ }
176
+ function extensionOf(name) {
177
+ const dot = name.lastIndexOf(".");
178
+ return dot > 0 ? name.slice(dot) : null;
179
+ }
180
+ /**
181
+ * True when every directory the aggregate covers holds only files the snapshot
182
+ * already knows about.
183
+ *
184
+ * The indexable extensions are taken from the snapshot's own file list rather
185
+ * than hardcoded, so this stays calibrated to whatever the walker actually
186
+ * indexed. Membership is tested against the WHOLE snapshot, not the filtered
187
+ * subset, so a sibling that is indexed but outside the caller's target does not
188
+ * read as an omission.
189
+ */
190
+ function snapshotCoversDirectories(repo, snapshot, covered) {
191
+ const indexed = new Set((snapshot?.files ?? []).map((file) => file.path));
192
+ if (indexed.size === 0)
193
+ return false;
194
+ const extensions = new Set();
195
+ for (const known of indexed) {
196
+ const extension = extensionOf(known.slice(known.lastIndexOf("/") + 1));
197
+ if (extension)
198
+ extensions.add(extension);
199
+ }
200
+ for (const directory of new Set(covered.map((file) => directoryOf(file.path)))) {
201
+ let entries;
202
+ try {
203
+ entries = fs.readdirSync(path.resolve(repo, directory), { withFileTypes: true });
204
+ }
205
+ catch {
206
+ return false;
207
+ }
208
+ for (const entry of entries) {
209
+ if (!entry.isFile())
210
+ continue;
211
+ const extension = extensionOf(entry.name);
212
+ if (!extension || !extensions.has(extension))
213
+ continue;
214
+ const relative = directory === "." ? entry.name : `${directory}/${entry.name}`;
215
+ if (!indexed.has(relative))
216
+ return false;
217
+ }
218
+ }
219
+ return true;
220
+ }
172
221
  export function runStructuralFastPath(repo, pattern, target, limit = 100) {
173
222
  const snapshot = readStructuralSnapshot(repo);
174
223
  if (!snapshot && pattern !== "file_summary")
@@ -195,7 +244,7 @@ export function runStructuralFastPath(repo, pattern, target, limit = 100) {
195
244
  let fingerprint;
196
245
  let fresh = false;
197
246
  if (pattern === "project_overview") {
198
- fresh = matching.every((file) => {
247
+ const unchanged = matching.every((file) => {
199
248
  try {
200
249
  const stat = fs.statSync(path.resolve(repo, file.path));
201
250
  return stat.size === file.sizeBytes && stat.mtimeMs === file.mtimeMs;
@@ -204,6 +253,20 @@ export function runStructuralFastPath(repo, pattern, target, limit = 100) {
204
253
  return false;
205
254
  }
206
255
  });
256
+ // A size+mtime sweep can only ever prove that the files the snapshot
257
+ // ALREADY knows about are unchanged. A file added since the snapshot is not
258
+ // in that set, so it was never checked, never counted, and never mentioned
259
+ // — while the overview still reported itself "fresh" (KNODIN-25). So also
260
+ // read back the directories being aggregated and look for an indexable
261
+ // sibling the snapshot has never seen.
262
+ //
263
+ // Known bound: this sees additions to directories the snapshot already
264
+ // covers, which is where source files are overwhelmingly added. A brand new
265
+ // top-level directory is still missed, because recognising one would mean
266
+ // re-deriving the walker's prune rules here and re-walking the tree, which
267
+ // is the cost this fast path exists to avoid. `upgrade` already points at
268
+ // the full `architecture_overview` query for callers that need certainty.
269
+ fresh = unchanged && snapshotCoversDirectories(repo, snapshot, matching);
207
270
  const directories = new Map();
208
271
  for (const file of matching) {
209
272
  const directory = file.path.includes("/") ? file.path.split("/", 1)[0] : ".";
@@ -70,10 +70,8 @@ export async function closeKnodinToolEngine() {
70
70
  if (initializedEngine)
71
71
  await initializedEngine.close();
72
72
  initializedEngine = null;
73
- compactReadyRepos.clear();
74
73
  }
75
74
  const localTelemetry = [];
76
- const compactReadyRepos = new Set();
77
75
  let cachedSchemaTokens;
78
76
  function gatewaySchemaTokens() {
79
77
  cachedSchemaTokens ??= countOutputTokens(JSON.stringify(buildDocumentedKnodinTools()[0]?.inputSchema ?? {}));
@@ -82,15 +80,24 @@ function gatewaySchemaTokens() {
82
80
  function inspectGatewayGraphHealth(repo) {
83
81
  return inspectGraphQueryHealth(repo, async (target) => attachLifecycleHealth(target, await engine.status(target, { audit: "cached" })));
84
82
  }
83
+ /**
84
+ * Probe on EVERY compact call, exactly as the non-compact path does.
85
+ *
86
+ * This used to latch a repository as ready after one successful probe and never
87
+ * look again for the life of the process. Nothing invalidated that latch —
88
+ * `closeKnodinToolEngine` clears it but has no production caller — so once
89
+ * `.knodin/` was deleted or re-inited from another terminal, a long-lived MCP
90
+ * server kept answering compact queries against a database the open path had
91
+ * silently recreated as an empty schema (`CREATE TABLE IF NOT EXISTS`). The
92
+ * result was "no matches": indistinguishable from a true negative, and the exact
93
+ * confident-but-wrong answer this product exists to avoid (KNODIN-18).
94
+ *
95
+ * The probe is cheap — `engine.status(..., { audit: "cached" })` — so the latch
96
+ * was buying very little in exchange for that failure mode.
97
+ */
85
98
  async function compactGraphUnavailable(repo) {
86
- const key = nodePath.resolve(repo);
87
- if (compactReadyRepos.has(key))
88
- return null;
89
99
  const health = await inspectGatewayGraphHealth(repo);
90
- if (!health.available)
91
- return health;
92
- compactReadyRepos.add(key);
93
- return null;
100
+ return health.available ? null : health;
94
101
  }
95
102
  /** Runs `fn`, converting any thrown error into a structured `{ error }` result
96
103
  * instead of letting it escape into the stdio transport. Extracted from the
@@ -24,6 +24,36 @@ Results distinguish:
24
24
  - stale linked-worktree metadata;
25
25
  - an unrelated repository under the same discovery root.
26
26
 
27
+ ### Graphs are per-worktree
28
+
29
+ A linked worktree carries its own `.knodin` with its own graph, `indexedHead`,
30
+ and freshness. A healthy main checkout does not cover it, so a worktree created
31
+ by `git worktree add` has **no graph until `knodin init` runs inside it**.
32
+
33
+ `git worktree add` creates no `.knodin` of its own — the directory is knodin's,
34
+ written by `init` and by the managed lifecycle hooks. Where those hooks are
35
+ already installed in the repository, a new worktree can therefore appear to have
36
+ a `.knodin` while holding no graph at all, which is the more confusing shape of
37
+ this: the directory exists, and it is empty of anything that can answer.
38
+
39
+ That does not mean a new worktree pays for a full index. `init` selects the
40
+ already-indexed sibling worktree closest in commit distance, reflink-copies its
41
+ baseline where the filesystem supports it, reconciles only the paths that differ
42
+ between that sibling's `indexedHead` and this HEAD, deep-audits the result, and
43
+ promotes it only if it is healthy — falling back to a full index when no
44
+ schema- and build-compatible sibling exists. `KNODIN_DISABLE_WORKTREE_SEED=1`
45
+ forces the full path.
46
+
47
+ Seeding is on the `init` path only, which is why an uninitialized worktree is
48
+ steered to `init` rather than `repair`: both produce a correct index, but
49
+ `repair` rebuilds from scratch.
50
+
51
+ Nothing silently answers from the wrong graph in that state: structural queries
52
+ refuse with `available: false`, `state: not-initialized`, and exit code 1 rather
53
+ than returning an empty result, and status leads its remediation with the
54
+ per-worktree model and `knodin init`. Read a refusal as "this checkout was never
55
+ indexed", never as "there is nothing here to find".
56
+
27
57
  ### Opt-in repository signals
28
58
 
29
59
  `repos discover --json --signals` adds a deterministic `signals` object to
@@ -0,0 +1,165 @@
1
+ # knodin 0.10.8
2
+
3
+ One change, prompted by a field report from an agent session that built a
4
+ duplicate implementation of a hook that already existed. Most of that report
5
+ described behaviour knodin already has; the part it got right is fixed here.
6
+
7
+ ## A linked worktree with no graph now says so in its remediation
8
+
9
+ `git worktree add` gives you a checkout with no graph in it. Graphs are
10
+ **per-worktree**, so a perfectly healthy main checkout tells you nothing about
11
+ the worktree you are standing in — and that makes the missing graph *more*
12
+ surprising, not less.
13
+
14
+ Git creates no `.knodin`; that directory is knodin's, written by `init` and by
15
+ the managed lifecycle hooks. Where those hooks are already installed, a fresh
16
+ worktree can present a `.knodin` that holds no graph — which is the more
17
+ confusing shape of this, and the one the original report hit.
18
+
19
+ Until now, status in that state produced generic remediation:
20
+
21
+ ```
22
+ remediation:
23
+ - Run `knodin repair` to create the local index.
24
+ - Run `knodin status` again to verify health.
25
+ ```
26
+
27
+ That guidance works — `repair` does build the index from cold, and then tells
28
+ you to run `init` for lifecycle routing. What it does not do is explain *why*
29
+ there was no index in a repository whose main checkout is fine. A reader who
30
+ does not already know the per-worktree model has no way to tell "this checkout
31
+ was never indexed" from "there is nothing here to find", and the second reading
32
+ is the one that quietly confirms whatever hypothesis sent them looking.
33
+
34
+ A linked worktree now leads with the model and the fix, on both the
35
+ machine-readable `remediation` and the human `status` line:
36
+
37
+ ```
38
+ Run `knodin init` here: graphs are per-worktree and this one has none of its
39
+ own, so the main checkout's graph does not cover it. `init` seeds from an
40
+ indexed sibling worktree where one exists rather than rebuilding from scratch.
41
+ ```
42
+
43
+ Ordinary checkouts are unchanged. Detection reads the `.git` file's `gitdir:`
44
+ pointer and requires a `commondir` beside it — no subprocess, no registry read,
45
+ and it works on a repository too damaged to answer anything else.
46
+
47
+ A `.git` **file** alone is not enough, and treating it as enough was wrong in
48
+ the first cut of this fix: submodules and `--separate-git-dir` clones use one
49
+ too, and neither has a main checkout whose graph could cover it, so they would
50
+ have been handed an explanation that does not apply to them. `commondir` is
51
+ what actually distinguishes a linked worktree.
52
+
53
+ The check lives in a new `src/engine/git-layout.ts` rather than in the engine,
54
+ for a reason worth stating. `engine/index.ts` carries a pinned budget on direct
55
+ `fs` reads, because a sealed artifact answers with no working tree and a reader
56
+ that touches the filesystem there gets ENOENT — which the nearest guard turns
57
+ into `""` or `null`, and which reads downstream as "this symbol has no body"
58
+ rather than "this was not covered". Git *metadata* is categorically outside
59
+ that concern: a sealed artifact has no `.git` at all, so routing these reads
60
+ through the sealed resolver would add a branch that can never execute. Saying
61
+ so in the layout is more honest than spending budget the engine reserves for
62
+ source reads — and it makes the check directly unit-testable, which it was not
63
+ as an engine-private helper.
64
+
65
+ Its spec covers every branch (100% of statements and branches): a plain
66
+ directory, an ordinary checkout, a real linked worktree, a real submodule, a
67
+ real `--separate-git-dir` clone, a malformed `.git` file naming no gitdir, and a
68
+ dangling pointer whose administrative directory was pruned.
69
+
70
+ ### Why `init` and not `repair` here
71
+
72
+ `repair` does build the index from cold, and for an ordinary checkout it stays
73
+ the right advice. In a linked worktree it is the *expensive* answer: seeding
74
+ lives on the `init` path only (`indexOrSeed`), so `repair` rebuilds from scratch
75
+ while `init` reflink-copies an indexed sibling's baseline, reconciles only the
76
+ paths that differ between that sibling's `indexedHead` and this HEAD,
77
+ deep-audits the candidate, and promotes it only if it passes — falling back to a
78
+ full index when no compatible sibling exists.
79
+
80
+ The human `status` renderer had been computing its own one-line advice and
81
+ ignoring `repairSteps` entirely, so the first fix reached the JSON surface and
82
+ not the line a person actually reads. It now defers to the step the engine
83
+ already wrote. Measured on a 20-file fixture, `init` in a fresh worktree with an
84
+ indexed sibling completes in about three seconds.
85
+
86
+ Worth stating plainly, because the surprise runs the other way from the cost:
87
+ the per-worktree model does **not** mean a new worktree pays for a full index.
88
+
89
+ ## What the same report claimed that measurement did not support
90
+
91
+ Recorded because acting on any of these would have been a regression, and
92
+ because each was reported in good faith from accumulated notes rather than from
93
+ a run against this release.
94
+
95
+ **"Structural queries return empty rather than erroring on an uninitialized
96
+ graph."** They do not, and have not for some time. Against a committed
97
+ repository with no index, `knodin query callers_of <symbol>` returns
98
+ `available: false`, `status: unavailable`, `state: not-initialized`, the full
99
+ status envelope, and **exit code 1**. The same holds for `impact`, `dead_code`,
100
+ `tests_for`, and `file_summary`, on both the CLI and the MCP gateway — both
101
+ route through the same `inspectGraphQueryHealth` gate before and after the
102
+ operation runs. There was no empty result set to fix.
103
+
104
+ This one matters most, because an empty structural result that agrees with your
105
+ hypothesis is the worst failure this tool could have. It is worth restating that
106
+ the gate is there, and that it fails loudly on both surfaces.
107
+
108
+ **"`knodin init --scope personal` exits 0 having done nothing."** It exits **1**.
109
+ The refusal is thrown, and every non-`agent-event` throw reaches the top-level
110
+ handler, which prints to stderr and exits 1. Reproduced against a repository with
111
+ tracked team integration active.
112
+
113
+ **"A backgrounded `knodin repair` produces no progress output."** Repair emits
114
+ ordered per-phase progress on stderr — audit, planning, file reconciliation,
115
+ symbol identities, embeddings, orphan cleanup, verification — and finishes with a
116
+ verified summary naming indexed file and symbol counts.
117
+
118
+ Both of these turned out to have the same cause, since confirmed by the
119
+ reporter: the commands were run through `| tail`. A pipeline's exit status is the
120
+ last command's, so `init`'s 1 was recorded as `tail`'s 0; and `tail` flushes
121
+ nothing until the stream closes, so a capture file read mid-run looked empty
122
+ while repair was in fact reporting every phase. Neither surface changed. Worth
123
+ naming because one habit produced two bug reports against the wrong component.
124
+
125
+ ## Not changed here: semantic search ranking
126
+
127
+ A follow-up from the same session measured something real and unaddressed: on a
128
+ healthy graph, `"hands-free voice conversation mode hook with barge-in"` ranked a
129
+ constant list (`BARGE_IN_PHRASES`, 0.588) and the caller's own minutes-old
130
+ duplicate (0.552) above the canonical 513-line implementation, which did not
131
+ appear in the top hits at all. Mechanism-shaped phrasing
132
+ (`"voice pipeline auto restart mic after TTS"`) put the canonical hook at #2,
133
+ 0.517.
134
+
135
+ So feature-shaped phrasing — the phrasing a developer uses when asking "does this
136
+ already exist?" — ranked worst for exactly that question, and surfaced the new
137
+ duplicate as apparent confirmation. Centrality data the fix would need
138
+ (in-degree, community membership) is already in the graph and already in the
139
+ output; it is not weighted into ranking.
140
+
141
+ That is a ranking change with its own evaluation burden and it is not bundled
142
+ into a remediation-string fix. It is recorded here so it is not rediscovered from
143
+ scratch.
144
+
145
+ ## The pre-push test gate was not gating
146
+
147
+ Found because this release's own broken commit sailed through it.
148
+
149
+ `lefthook.yml`'s `quality-gates` block ran `bun run test:coverage` and then
150
+ `sh scripts/run-sonar-scan.sh` as two lines of one shell block with no
151
+ `set -e`. A shell block's exit status is its **last** command's, so a failing
152
+ suite followed by a passing Sonar scan reported success, and the push was
153
+ allowed. The suite printed `1 failed | 2058 passed` and
154
+ `script "test:coverage" exited with code 1`, and the push completed anyway.
155
+
156
+ `set -e` now leads the block. Coverage thresholds and Sonar were unaffected;
157
+ what was broken is that a red suite could not stop a push.
158
+
159
+ ## Verification
160
+
161
+ `npm test`, `npm run lint`, and `npm run typecheck` pass. The new behaviour is
162
+ pinned by a test in `src/__tests__/unit/index-health.spec.ts` that builds a real
163
+ linked worktree with `git worktree add`, asserts the per-worktree line leads its
164
+ remediation, and asserts the ordinary checkout keeps repair-first guidance with
165
+ no per-worktree line.