knodin 0.10.2 → 0.10.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -28,7 +28,7 @@ import { contentFingerprint, writeStructuralSnapshot, } from "../structural-snap
28
28
  import { acquireLifecycleCoordination } from "../update-coordination.js";
29
29
  import { KNODIN_VERSION } from "../version.js";
30
30
  import * as ann from "./ann-hnsw.js";
31
- import { allocateCandidate, assertCandidate, discardCandidateFiles, promoteCandidateFile, recoverInterruptedPromotion, } from "./candidate-database.js";
31
+ import { allocateCandidate, assertCandidate, discardCandidateFiles, listCandidates, promoteCandidateFile, recoverInterruptedPromotion, } from "./candidate-database.js";
32
32
  import { computeSimilarity, generateEmbedding, generateEmbeddings, } from "./embeddings.js";
33
33
  import { walkRepoFiles } from "./file-walker.js";
34
34
  import { clearGitHistorySignalCache, collectGitHistorySignals, } from "./git-history.js";
@@ -46,6 +46,7 @@ import { isIndexableSourcePath } from "./source-policy.js";
46
46
  import { Database } from "./sqlite.js";
47
47
  import { lookupMirror, mayWriteToRepository, resolveDbPath, resolveStateDir, } from "./state-paths.js";
48
48
  import { deleteAllSymbols, deleteSymbolsForFile, deleteSymbolsMatchingPath, ORPHANED_EMBEDDING_PREDICATE, purgeOrphanEmbeddings, } from "./symbol-delete.js";
49
+ import { searchRepoText } from "./text-matches.js";
49
50
  // ES Module resolution
50
51
  const __filename = fileURLToPath(import.meta.url);
51
52
  const __dirname = path.dirname(__filename);
@@ -373,6 +374,36 @@ export function extractMcpToolRegistrations(source, file, root) {
373
374
  }
374
375
  /** `meta` key holding the last index's unparsed-file tally, as a JSON object. */
375
376
  export const COVERAGE_UNPARSED_META_KEY = "coverageUnparsedByExtension";
377
+ /**
378
+ * `meta` key set while a full index is running and cleared when it finishes.
379
+ *
380
+ * The tally is written incrementally so an interrupted index does not lose it,
381
+ * which means a present tally no longer implies a *complete* one. This marker
382
+ * is what keeps the two apart: while it is set, the tally describes however far
383
+ * the run got, and `unparsedUnknown` reports it as a lower bound rather than a
384
+ * measurement.
385
+ *
386
+ * Without it, an index killed at 71% would leave a plausible-looking tally that
387
+ * `status` presents as fact — the precise confusion `unparsedUnknown` exists to
388
+ * prevent, reintroduced through the back door.
389
+ */
390
+ export const COVERAGE_TALLY_IN_PROGRESS_META_KEY = "coverageUnparsedIncomplete";
391
+ /**
392
+ * `meta` keys marking a candidate database as a clean index still in progress.
393
+ *
394
+ * A clean index builds into a candidate and promotes it atomically, discarding
395
+ * it on failure — which is why a partial rebuild can never be observed as the
396
+ * live graph, and why that behaviour is preserved exactly. But a killed process
397
+ * runs no discard, so the candidate simply survives on disk, and until now the
398
+ * next run allocated a fresh one and repeated hours of work.
399
+ *
400
+ * These make the survivor recognisable. The marker says "a build was underway";
401
+ * the head says "of this tree". Both are needed: resuming a build of a tree that
402
+ * has since moved on would promote a graph that never described any single state
403
+ * of the repository, which is worse than starting over.
404
+ */
405
+ export const CLEAN_INDEX_IN_PROGRESS_META_KEY = "cleanIndexInProgress";
406
+ export const CLEAN_INDEX_HEAD_META_KEY = "cleanIndexHead";
376
407
  function repairMetadataFamily(filePath) {
377
408
  const segments = filePath.replaceAll("\\", "/").split("/");
378
409
  const extension = path.extname(filePath).toLowerCase();
@@ -7059,6 +7090,12 @@ function createIndexProgressReporter(onProgress) {
7059
7090
  elapsedMs: emittedAt - startedAt,
7060
7091
  message,
7061
7092
  ...(details.modelFile === undefined ? {} : { modelFile: details.modelFile }),
7093
+ ...(details.phaseBytesCompleted === undefined
7094
+ ? {}
7095
+ : { phaseBytesCompleted: details.phaseBytesCompleted }),
7096
+ ...(details.phaseBytesTotal === undefined
7097
+ ? {}
7098
+ : { phaseBytesTotal: details.phaseBytesTotal }),
7062
7099
  });
7063
7100
  }
7064
7101
  catch {
@@ -7224,6 +7261,75 @@ function sumCounts(counts) {
7224
7261
  *
7225
7262
  * Still never throws: a status call must not fail because a tally is corrupt.
7226
7263
  */
7264
+ /**
7265
+ * A candidate left mid-build that it is safe to continue, or null.
7266
+ *
7267
+ * Deliberately conservative: every uncertainty resolves to "start over", which
7268
+ * costs time, versus resuming onto the wrong tree, which produces a graph that
7269
+ * never described any single state of the repository and then promotes it.
7270
+ *
7271
+ * Stale and unusable candidates are removed as they are found, so a repository
7272
+ * does not accumulate abandoned copies of its own database.
7273
+ */
7274
+ function findResumableCandidate(repoPath) {
7275
+ const head = gitHead(repoPath) ?? "";
7276
+ let resumable = null;
7277
+ for (const candidate of listCandidates(repoPath)) {
7278
+ let usable = false;
7279
+ try {
7280
+ const probe = new Database(candidate.databasePath, { readonly: true });
7281
+ try {
7282
+ usable =
7283
+ getMeta(probe, CLEAN_INDEX_IN_PROGRESS_META_KEY) === "1" &&
7284
+ getMeta(probe, CLEAN_INDEX_HEAD_META_KEY) === head;
7285
+ }
7286
+ finally {
7287
+ probe.close();
7288
+ }
7289
+ }
7290
+ catch {
7291
+ // Unreadable, not a database, or schema too old to query. Not resumable,
7292
+ // and not worth keeping.
7293
+ usable = false;
7294
+ }
7295
+ // Keep only the newest usable one. An older survivor is from an even
7296
+ // earlier interrupted run and has nothing to add.
7297
+ if (usable && resumable === null) {
7298
+ resumable = candidate;
7299
+ continue;
7300
+ }
7301
+ try {
7302
+ discardCandidateFiles(repoPath, candidate);
7303
+ }
7304
+ catch {
7305
+ // Best effort: failing to clean up a stale candidate must not stop the
7306
+ // rebuild that is about to replace it.
7307
+ }
7308
+ }
7309
+ return resumable;
7310
+ }
7311
+ /** Write the coverage tally as it stands. Safe to call repeatedly mid-index. */
7312
+ function persistUnparsedTally(db, tally) {
7313
+ setMeta(db, COVERAGE_UNPARSED_META_KEY, JSON.stringify(Object.fromEntries(tally)));
7314
+ }
7315
+ /**
7316
+ * True when the stored tally came from a run that did not finish.
7317
+ *
7318
+ * Read separately from the tally itself because the two answer different
7319
+ * questions: the tally says what was seen, this says whether that is all there
7320
+ * was to see.
7321
+ */
7322
+ function unparsedTallyIsIncomplete(db) {
7323
+ if (!db)
7324
+ return false;
7325
+ try {
7326
+ return getMeta(db, COVERAGE_TALLY_IN_PROGRESS_META_KEY) === "1";
7327
+ }
7328
+ catch {
7329
+ // Unreadable meta is not a claim that the tally is complete.
7330
+ return true;
7331
+ }
7332
+ }
7227
7333
  function readUnparsedTally(db) {
7228
7334
  if (!db)
7229
7335
  return null;
@@ -7290,6 +7396,10 @@ function worstSemanticReadiness(states) {
7290
7396
  function buildCoverageSkips(skippedByExtension, db) {
7291
7397
  const byExtension = sortedTally(skippedByExtension);
7292
7398
  const unparsed = readUnparsedTally(db);
7399
+ // A tally that exists but came from an unfinished run is as unmeasured as an
7400
+ // absent one: it counts only the files that run happened to reach. Both cases
7401
+ // set `unparsedUnknown`, which downstream already renders as a lower bound.
7402
+ const incomplete = unparsed === null || unparsedTallyIsIncomplete(db);
7293
7403
  return {
7294
7404
  byExtension,
7295
7405
  // Reported as empty when unknown so consumers reading only this field are
@@ -7297,7 +7407,7 @@ function buildCoverageSkips(skippedByExtension, db) {
7297
7407
  // measurement. `total` is then a lower bound, not a count.
7298
7408
  unparsedByExtension: unparsed ?? {},
7299
7409
  total: sumCounts(byExtension) + sumCounts(unparsed ?? {}),
7300
- ...(unparsed === null ? { unparsedUnknown: true } : {}),
7410
+ ...(incomplete ? { unparsedUnknown: true } : {}),
7301
7411
  };
7302
7412
  }
7303
7413
  /**
@@ -7554,13 +7664,23 @@ function buildFreshnessEnvelope(repoPath, db, verifiedAt, stateOverride, gitProb
7554
7664
  * Snapshot a file's on-disk mtime/size into `index_state`, so a later cold start
7555
7665
  * can tell whether it drifted while no knodin process was watching.
7556
7666
  */
7667
+ /**
7668
+ * Record a file's mtime and size, returning the size.
7669
+ *
7670
+ * The size is returned rather than discarded so byte-based progress costs no
7671
+ * extra I/O: this already stats every indexed file, and a second pass over
7672
+ * 900,000 files to learn what it just measured would be pure waste. Returns 0
7673
+ * when the file could not be stat'd, which the caller adds harmlessly.
7674
+ */
7557
7675
  function recordIndexState(db, repoPath, relPath) {
7558
7676
  try {
7559
7677
  const st = fs.statSync(path.join(repoPath, relPath));
7560
7678
  db.run("INSERT INTO index_state(filePath, mtimeMs, size) VALUES (?, ?, ?) ON CONFLICT(filePath) DO UPDATE SET mtimeMs = excluded.mtimeMs, size = excluded.size", [relPath, st.mtimeMs, st.size]);
7679
+ return st.size;
7561
7680
  }
7562
7681
  catch (_) {
7563
7682
  // File vanished between indexing and stat — reconcile-delete handles it.
7683
+ return 0;
7564
7684
  }
7565
7685
  }
7566
7686
  /** Drop a file's `index_state` row (used when a file is deleted). */
@@ -7638,7 +7758,7 @@ function detectDriftByStat(repoPath, db, canonicalFiles) {
7638
7758
  }
7639
7759
  return changed;
7640
7760
  }
7641
- async function reconcileIndex(repoPath, db, progress) {
7761
+ async function reconcileIndex(repoPath, db, progress, skipEmbeddings = false) {
7642
7762
  try {
7643
7763
  progress?.("collecting-files", 0, "Checking existing local graph for changes");
7644
7764
  const changed = new Set();
@@ -7757,18 +7877,26 @@ async function reconcileIndex(repoPath, db, progress) {
7757
7877
  // Re-embed ONLY what changed (indexEmbeddings embeds symbols lacking an
7758
7878
  // embedding), so a no-op reconcile triggers no embedding work at all.
7759
7879
  if (reindexed > 0) {
7760
- progress?.("finalizing", 0, "Refreshing identities and semantic embeddings");
7880
+ progress?.("finalizing", 0, skipEmbeddings
7881
+ ? "Refreshing identities (semantic embeddings deferred)"
7882
+ : "Refreshing identities and semantic embeddings");
7761
7883
  persistSymbolIdentities(db, repoPath, reindexedPaths);
7762
7884
  reconcileTypeScriptDi(db, repoPath);
7763
- await indexEmbeddings(db, repoPath, progress);
7885
+ if (!skipEmbeddings)
7886
+ await indexEmbeddings(db, repoPath, progress);
7764
7887
  indexGeneration++;
7765
7888
  }
7766
- else if (semanticReadinessFor(db) !== "ready") {
7889
+ else if (!skipEmbeddings && semanticReadinessFor(db) !== "ready") {
7767
7890
  // Unchanged files do NOT imply current embeddings: a deferred
7768
7891
  // (`skipEmbeddings`) or interrupted pass leaves symbols unembedded while
7769
7892
  // every file looks reconciled. Without this, the follow-up index a user
7770
7893
  // is told to run would do nothing and semantic search would stay
7771
7894
  // silently short forever.
7895
+ //
7896
+ // Which is also why `skipEmbeddings` has to suppress it: this branch
7897
+ // exists to COMPLETE a deferred pass, so leaving it unguarded would make
7898
+ // `--skip-embeddings` a silent no-op on exactly the warm graphs large
7899
+ // enough to want it — the flag would appear to work and do the opposite.
7772
7900
  progress?.("finalizing", 0, "Completing deferred semantic embeddings");
7773
7901
  await indexEmbeddings(db, repoPath, progress);
7774
7902
  indexGeneration++;
@@ -8146,7 +8274,17 @@ function stalenessFor(repoPath) {
8146
8274
  return freshnessProbes.get(path.resolve(repoPath))?.staleness ?? "unknown";
8147
8275
  }
8148
8276
  /** Recursively indexes all matching files within the repository. */
8149
- async function indexRepo(repoPath, db, progress, skipEmbeddings = false, publishProcessState = true) {
8277
+ async function indexRepo(repoPath, db, progress, skipEmbeddings = false, publishProcessState = true,
8278
+ /**
8279
+ * Continue a build already in this database instead of starting over.
8280
+ *
8281
+ * Skips the wipe below and skips files whose `index_state` row still matches
8282
+ * disk. Only ever set for a candidate database that a previous run left
8283
+ * mid-build: the live graph is never resumed into, because resuming implies
8284
+ * partially-populated intermediate state and the live graph must never be
8285
+ * observable in that condition.
8286
+ */
8287
+ resume = false) {
8150
8288
  progress?.("collecting-files", 0, "Discovering indexable files");
8151
8289
  const collected = measurePerfPhaseSync("file_collection", () => collectRepoFilesWithCoverage(repoPath));
8152
8290
  const files = collected.files;
@@ -8154,15 +8292,82 @@ async function indexRepo(repoPath, db, progress, skipEmbeddings = false, publish
8154
8292
  // during the parse. Persisted below so `status` can report the gap without
8155
8293
  // re-indexing the repository to rediscover it.
8156
8294
  const unparsedByExtension = new Map();
8295
+ // Claim the tally as in-progress BEFORE the first file. An index killed
8296
+ // partway used to leave the previous run's tally in place and reported as
8297
+ // fact; now whatever is stored is flagged a lower bound until this run
8298
+ // finishes and clears the marker.
8299
+ setMeta(db, COVERAGE_TALLY_IN_PROGRESS_META_KEY, "1");
8300
+ // One stat pass to learn the weight of the work before starting it. Counted
8301
+ // separately from files because the two diverge sharply: on a real Salesforce
8302
+ // checkout 0.05% of the files hold 78% of the bytes, so a files-only estimate
8303
+ // is confidently wrong rather than merely rough.
8304
+ //
8305
+ // `size` is a floor on cost, not a proxy for it — a large XML file is not
8306
+ // exactly proportional to a large TypeScript one — but it tracks the actual
8307
+ // shape of the work far better than a file count, which treats a 7 MB profile
8308
+ // and a 200-byte translation as equal.
8309
+ let totalBytes = 0;
8310
+ for (const file of files) {
8311
+ try {
8312
+ totalBytes += fs.statSync(path.join(repoPath, file)).size;
8313
+ }
8314
+ catch {
8315
+ // Unreadable or vanished: it will fail in the loop below too, where the
8316
+ // failure is reported. Excluding it here only makes the total a floor.
8317
+ }
8318
+ }
8319
+ let completedBytes = 0;
8157
8320
  progress?.("indexing-files", 0, `Indexing ${files.length.toLocaleString()} files`, {
8158
8321
  phaseTotal: files.length,
8322
+ phaseBytesCompleted: 0,
8323
+ phaseBytesTotal: totalBytes,
8159
8324
  });
8160
- // Clean out existing data to ensure consistency on full re-index
8161
- deleteAllSymbols(db);
8162
- db.run('DELETE FROM "references";');
8163
- db.run("DELETE FROM dependencies;");
8164
- db.run("DELETE FROM mcp_tools;");
8165
- db.run("DELETE FROM index_state;");
8325
+ // Clean out existing data to ensure consistency on full re-index.
8326
+ //
8327
+ // Skipped when resuming, which is the whole point: this wipe is why an
8328
+ // interrupted clean index used to lose everything it had done. Resuming into
8329
+ // a database it had just emptied would be indistinguishable from starting
8330
+ // over.
8331
+ if (!resume) {
8332
+ deleteAllSymbols(db);
8333
+ db.run('DELETE FROM "references";');
8334
+ db.run("DELETE FROM dependencies;");
8335
+ db.run("DELETE FROM mcp_tools;");
8336
+ db.run("DELETE FROM index_state;");
8337
+ }
8338
+ // On a resume, everything already recorded and still matching disk is done.
8339
+ // Counted toward progress rather than dropped from it, so the totals stay
8340
+ // whole-repository and the run visibly picks up where it stopped instead of
8341
+ // appearing to start a smaller job.
8342
+ //
8343
+ // Sizes come from the stored `index_state` rows, not fresh stats: the drift
8344
+ // check just proved they still match, so re-measuring 600,000 files to learn
8345
+ // what the database already knows would be the expensive way to be no more
8346
+ // correct.
8347
+ let workList = files;
8348
+ let resumedFiles = 0;
8349
+ let resumedBytes = 0;
8350
+ if (resume) {
8351
+ const pending = [];
8352
+ for (const file of files) {
8353
+ if (fileDriftedFromIndexState(db, repoPath, file)) {
8354
+ pending.push(file);
8355
+ continue;
8356
+ }
8357
+ resumedFiles++;
8358
+ resumedBytes +=
8359
+ db
8360
+ .query("SELECT size FROM index_state WHERE filePath = ?")
8361
+ .get(file)?.size ?? 0;
8362
+ }
8363
+ workList = pending;
8364
+ completedBytes = resumedBytes;
8365
+ progress?.("indexing-files", resumedFiles, `Resuming: ${resumedFiles.toLocaleString()} of ${files.length.toLocaleString()} files already indexed`, {
8366
+ phaseTotal: files.length,
8367
+ phaseBytesCompleted: resumedBytes,
8368
+ phaseBytesTotal: totalBytes,
8369
+ });
8370
+ }
8166
8371
  // Process files sequentially to ensure thread-safe SQLite transactions.
8167
8372
  // Extraction (buildFileIndexResult, pure) still happens one file at a
8168
8373
  // time here; only the write is batched, via `pendingWrites`, so the
@@ -8183,14 +8388,21 @@ async function indexRepo(repoPath, db, progress, skipEmbeddings = false, publish
8183
8388
  };
8184
8389
  // Progress is reported against a single counter so the pool and the
8185
8390
  // sequential path can interleave without the count going backwards.
8186
- let processed = 0;
8391
+ let processed = resumedFiles;
8187
8392
  const afterFile = async (file) => {
8188
8393
  if (pendingWrites.length >= INDEX_WRITE_BATCH_SIZE)
8189
8394
  flushPendingWrites();
8190
- recordIndexState(db, repoPath, file);
8395
+ completedBytes += recordIndexState(db, repoPath, file);
8191
8396
  processed++;
8397
+ // Persisted on the same cadence as the write batches rather than only at
8398
+ // the end, so an interrupted index keeps what it learned. Cheap: one meta
8399
+ // row per 200 files, not per file.
8400
+ if (processed % INDEX_WRITE_BATCH_SIZE === 0)
8401
+ persistUnparsedTally(db, unparsedByExtension);
8192
8402
  progress?.("indexing-files", processed, `Indexing ${files.length.toLocaleString()} files`, {
8193
8403
  phaseTotal: files.length,
8404
+ phaseBytesCompleted: completedBytes,
8405
+ phaseBytesTotal: totalBytes,
8194
8406
  });
8195
8407
  if (Date.now() - lastYieldAt >= INDEX_EVENT_LOOP_YIELD_MS) {
8196
8408
  await yieldToIndexEventLoop();
@@ -8215,7 +8427,7 @@ async function indexRepo(repoPath, db, progress, skipEmbeddings = false, publish
8215
8427
  // cannot disagree about where a file belongs.
8216
8428
  const eligible = [];
8217
8429
  const inline = [];
8218
- for (const file of files) {
8430
+ for (const file of workList) {
8219
8431
  if (isWorkerEligibleFile(path.join(repoPath, file), file))
8220
8432
  eligible.push({ absolutePath: path.join(repoPath, file), relativePath: file, repoPath });
8221
8433
  else
@@ -8246,7 +8458,7 @@ async function indexRepo(repoPath, db, progress, skipEmbeddings = false, publish
8246
8458
  }
8247
8459
  }
8248
8460
  else {
8249
- for (const file of files) {
8461
+ for (const file of workList) {
8250
8462
  await indexOneInline(file);
8251
8463
  await afterFile(file);
8252
8464
  }
@@ -8270,7 +8482,11 @@ async function indexRepo(repoPath, db, progress, skipEmbeddings = false, publish
8270
8482
  setMeta(db, "knodinVersion", KNODIN_VERSION);
8271
8483
  setMeta(db, "lastSuccessfulReconciliation", new Date().toISOString());
8272
8484
  setMeta(db, "mcpBackfillVersion", "17");
8273
- setMeta(db, COVERAGE_UNPARSED_META_KEY, JSON.stringify(Object.fromEntries(unparsedByExtension)));
8485
+ persistUnparsedTally(db, unparsedByExtension);
8486
+ // Only now is the tally a measurement rather than a lower bound. Clearing the
8487
+ // marker LAST, after the final write, means any failure above leaves it set
8488
+ // and the tally honestly flagged incomplete.
8489
+ setMeta(db, COVERAGE_TALLY_IN_PROGRESS_META_KEY, "");
8274
8490
  recordFreshnessBaseline(repoPath, db);
8275
8491
  if (publishProcessState)
8276
8492
  indexGeneration++;
@@ -11340,6 +11556,13 @@ function withStaleness(engine, claimRepository, openDb, openPolicy) {
11340
11556
  row.staleness = staleness;
11341
11557
  return page;
11342
11558
  },
11559
+ searchText(term, repoPath, options) {
11560
+ claimRepository(repoPath);
11561
+ // No staleness annotation: this reads the working tree directly rather
11562
+ // than the graph, so its answers are current by construction and
11563
+ // stamping them with the index's freshness would misreport them.
11564
+ return engine.searchText(term, repoPath, options);
11565
+ },
11343
11566
  async query(pattern, target, repoPath, to, limit, depth, detailLevel, selector, impactOptions, options) {
11344
11567
  claimRepository(repoPath);
11345
11568
  const result = await engine.query(pattern, target, repoPath, to, limit, depth, detailLevel, selector, impactOptions, options);
@@ -11654,7 +11877,18 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
11654
11877
  if (!db)
11655
11878
  throw new Error("candidate database did not open");
11656
11879
  if (!changes) {
11657
- await indexRepo(resolved, db, options?.onProgress, options?.skipEmbeddings === true, false);
11880
+ // Claim the build before touching anything, and record which tree it
11881
+ // is for. A candidate carrying this marker is one a previous run left
11882
+ // mid-build; without the commit, a resume could silently continue a
11883
+ // build of a tree that has since moved on and promote a graph that
11884
+ // never described any single state of the repository.
11885
+ setMeta(db, CLEAN_INDEX_IN_PROGRESS_META_KEY, "1");
11886
+ setMeta(db, CLEAN_INDEX_HEAD_META_KEY, gitHead(resolved) ?? "");
11887
+ await indexRepo(resolved, db, options?.onProgress, options?.skipEmbeddings === true, false, options?.resume === true);
11888
+ // Cleared only after indexRepo returns. Anything that fails or is
11889
+ // killed above leaves it set, which is exactly what makes the
11890
+ // candidate recognisable as resumable rather than abandoned.
11891
+ setMeta(db, CLEAN_INDEX_IN_PROGRESS_META_KEY, "");
11658
11892
  return { reconciled: collectRepoFiles(resolved) };
11659
11893
  }
11660
11894
  const reconciled = [...new Set(changes)]
@@ -13491,6 +13725,27 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
13491
13725
  },
13492
13726
  async index(repoPath, files, clean = false, options) {
13493
13727
  const resolved = path.resolve(repoPath);
13728
+ // Expand `--under` before anything else, so everything downstream sees an
13729
+ // ordinary explicit-file index and no path can route a scoped rebuild
13730
+ // into the destructive full-index branch.
13731
+ const scopedUnder = options?.under;
13732
+ if (scopedUnder !== undefined && (!files || files.length === 0)) {
13733
+ const prefix = path
13734
+ .relative(resolved, path.resolve(resolved, scopedUnder))
13735
+ .split(path.sep)
13736
+ .join("/");
13737
+ if (prefix.startsWith("..") || path.isAbsolute(prefix))
13738
+ throw new Error(`knodin index --under: ${scopedUnder} is outside the repository`);
13739
+ const collected = collectRepoFilesWithCoverage(resolved);
13740
+ files =
13741
+ prefix === ""
13742
+ ? collected.files
13743
+ : collected.files.filter((file) => file === prefix || file.startsWith(`${prefix}/`));
13744
+ // Failing loudly rather than indexing nothing and reporting success:
13745
+ // a silent no-op here would look identical to a completed rebuild.
13746
+ if (files.length === 0)
13747
+ throw new Error(`knodin index --under: no indexable files under ${scopedUnder || "the repository root"}`);
13748
+ }
13494
13749
  const indexCoordination = acquireLifecycleCoordination({
13495
13750
  command: "index",
13496
13751
  repository: resolved,
@@ -13503,13 +13758,27 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
13503
13758
  !process.env.VITEST &&
13504
13759
  process.env.NODE_ENV !== "test") {
13505
13760
  const progress = createIndexProgressReporter(options?.onProgress);
13506
- progress("starting", 0, "Creating isolated clean-index candidate");
13507
- const candidate = await engine.createCandidate(resolved);
13761
+ // Look for a build a previous run left unfinished before starting a
13762
+ // new one. A candidate survives only when the process was killed —
13763
+ // a failure discards it — so a survivor is precisely the case worth
13764
+ // continuing.
13765
+ const resumable = findResumableCandidate(resolved);
13766
+ // Adopt it into the engine's registry. Ownership is what
13767
+ // reconcile/audit/promote check, and a candidate recovered from disk
13768
+ // was never registered because the process that created it is gone.
13769
+ if (resumable)
13770
+ candidates.set(resumable.id, resumable);
13771
+ const candidate = resumable ?? (await engine.createCandidate(resolved));
13772
+ if (resumable)
13773
+ progress("starting", 0, "Resuming the interrupted clean index (previous progress kept)");
13774
+ else
13775
+ progress("starting", 0, "Creating isolated clean-index candidate");
13508
13776
  let promoted = false;
13509
13777
  try {
13510
13778
  const reconciliation = await engine.reconcileCandidate(candidate, resolved, undefined, {
13511
13779
  onProgress: progress,
13512
13780
  skipEmbeddings: options?.skipEmbeddings,
13781
+ resume: resumable !== null,
13513
13782
  });
13514
13783
  progress("verifying", 0, "Deep-auditing clean-index candidate", {
13515
13784
  phaseTotal: 1,
@@ -13650,7 +13919,7 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
13650
13919
  catch (_) { }
13651
13920
  if (!clean && hasSymbols) {
13652
13921
  progress("collecting-files", 0, "Checking existing local graph for changes");
13653
- await reconcileIndex(repoPath, db, progress);
13922
+ await reconcileIndex(repoPath, db, progress, options?.skipEmbeddings === true);
13654
13923
  }
13655
13924
  else {
13656
13925
  await indexRepo(repoPath, db, progress, options?.skipEmbeddings === true);
@@ -13768,9 +14037,26 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
13768
14037
  progress("completed", completed, health.status === "healthy"
13769
14038
  ? completionMessage
13770
14039
  : `${completionMessage}; verification found ${issueCount.toLocaleString()} issue(s)`, { phaseTotal: completed });
14040
+ // A scoped rebuild leaves the stored coverage tally describing a
14041
+ // different run than the graph it now sits beside. Left alone it
14042
+ // would still be present, and therefore still reported as measured
14043
+ // fact — the stale-and-authoritative case. Marking it incomplete
14044
+ // makes `status` report a lower bound until a full index restores a
14045
+ // whole-repository measurement.
14046
+ //
14047
+ // Only for an explicit `--under`, not for ordinary file arguments:
14048
+ // the lifecycle hooks index changed files constantly, and flagging
14049
+ // the tally on every commit would make the signal meaningless noise.
14050
+ if (scopedUnder !== undefined)
14051
+ setMeta(db, COVERAGE_TALLY_IN_PROGRESS_META_KEY, "1");
13771
14052
  return {
13772
14053
  indexed,
13773
14054
  unchanged,
14055
+ // Measured from the graph rather than inferred from the flag: a
14056
+ // run that deferred embeddings and a run that never had any are
14057
+ // the same state to a consumer, and a partially embedded graph is
14058
+ // neither.
14059
+ semanticReadiness: semanticReadinessFor(db),
13774
14060
  scip: scipReport,
13775
14061
  sarif: sarifReport,
13776
14062
  verification: {
@@ -13795,6 +14081,15 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
13795
14081
  indexCoordination.release();
13796
14082
  }
13797
14083
  },
14084
+ async searchText(term, repoPath, options = {}) {
14085
+ if (term.length === 0)
14086
+ throw new Error("knodin searchText: term must not be empty");
14087
+ // Reads the working tree, not the graph, so it needs no open database
14088
+ // and no freshness check. `getLanguageForFile` is the same grammar
14089
+ // loader indexing uses, which is what makes the classification
14090
+ // evidence rather than a heuristic over file extensions.
14091
+ return searchRepoText(path.resolve(repoPath), term, getLanguageForFile, () => new Parser(), options);
14092
+ },
13798
14093
  async search(query, repoPath, limit, options = {}) {
13799
14094
  const resolvedRepoPath = path.resolve(repoPath);
13800
14095
  const offset = options.offset ?? 0;