knodin 0.10.2 → 0.10.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/bin/cli.js +41 -4
- package/dist/src/cli-args.js +4 -0
- package/dist/src/cli-model.js +2 -0
- package/dist/src/engine/candidate-database.js +53 -0
- package/dist/src/engine/index.js +317 -22
- package/dist/src/engine/text-matches.js +309 -0
- package/dist/src/init-progress.js +36 -5
- package/dist/src/init.js +8 -1
- package/dist/src/lifecycle-health.js +36 -0
- package/dist/src/node-runtime.js +33 -0
- package/dist/src/release-compatibility.js +95 -0
- package/dist/src/release-preflight.js +18 -0
- package/dist/src/tools/knodin-tools.js +36 -7
- package/dist/src/update-policy.js +6 -1
- package/dist/src/wait-for-fresh.js +43 -1
- package/docs/releases/0.10.3.md +93 -0
- package/docs/releases/0.10.4.md +161 -0
- package/package.json +4 -2
package/dist/src/engine/index.js
CHANGED
|
@@ -28,7 +28,7 @@ import { contentFingerprint, writeStructuralSnapshot, } from "../structural-snap
|
|
|
28
28
|
import { acquireLifecycleCoordination } from "../update-coordination.js";
|
|
29
29
|
import { KNODIN_VERSION } from "../version.js";
|
|
30
30
|
import * as ann from "./ann-hnsw.js";
|
|
31
|
-
import { allocateCandidate, assertCandidate, discardCandidateFiles, promoteCandidateFile, recoverInterruptedPromotion, } from "./candidate-database.js";
|
|
31
|
+
import { allocateCandidate, assertCandidate, discardCandidateFiles, listCandidates, promoteCandidateFile, recoverInterruptedPromotion, } from "./candidate-database.js";
|
|
32
32
|
import { computeSimilarity, generateEmbedding, generateEmbeddings, } from "./embeddings.js";
|
|
33
33
|
import { walkRepoFiles } from "./file-walker.js";
|
|
34
34
|
import { clearGitHistorySignalCache, collectGitHistorySignals, } from "./git-history.js";
|
|
@@ -46,6 +46,7 @@ import { isIndexableSourcePath } from "./source-policy.js";
|
|
|
46
46
|
import { Database } from "./sqlite.js";
|
|
47
47
|
import { lookupMirror, mayWriteToRepository, resolveDbPath, resolveStateDir, } from "./state-paths.js";
|
|
48
48
|
import { deleteAllSymbols, deleteSymbolsForFile, deleteSymbolsMatchingPath, ORPHANED_EMBEDDING_PREDICATE, purgeOrphanEmbeddings, } from "./symbol-delete.js";
|
|
49
|
+
import { searchRepoText } from "./text-matches.js";
|
|
49
50
|
// ES Module resolution
|
|
50
51
|
const __filename = fileURLToPath(import.meta.url);
|
|
51
52
|
const __dirname = path.dirname(__filename);
|
|
@@ -373,6 +374,36 @@ export function extractMcpToolRegistrations(source, file, root) {
|
|
|
373
374
|
}
|
|
374
375
|
/** `meta` key holding the last index's unparsed-file tally, as a JSON object. */
|
|
375
376
|
export const COVERAGE_UNPARSED_META_KEY = "coverageUnparsedByExtension";
|
|
377
|
+
/**
|
|
378
|
+
* `meta` key set while a full index is running and cleared when it finishes.
|
|
379
|
+
*
|
|
380
|
+
* The tally is written incrementally so an interrupted index does not lose it,
|
|
381
|
+
* which means a present tally no longer implies a *complete* one. This marker
|
|
382
|
+
* is what keeps the two apart: while it is set, the tally describes however far
|
|
383
|
+
* the run got, and `unparsedUnknown` reports it as a lower bound rather than a
|
|
384
|
+
* measurement.
|
|
385
|
+
*
|
|
386
|
+
* Without it, an index killed at 71% would leave a plausible-looking tally that
|
|
387
|
+
* `status` presents as fact — the precise confusion `unparsedUnknown` exists to
|
|
388
|
+
* prevent, reintroduced through the back door.
|
|
389
|
+
*/
|
|
390
|
+
export const COVERAGE_TALLY_IN_PROGRESS_META_KEY = "coverageUnparsedIncomplete";
|
|
391
|
+
/**
|
|
392
|
+
* `meta` keys marking a candidate database as a clean index still in progress.
|
|
393
|
+
*
|
|
394
|
+
* A clean index builds into a candidate and promotes it atomically, discarding
|
|
395
|
+
* it on failure — which is why a partial rebuild can never be observed as the
|
|
396
|
+
* live graph, and why that behaviour is preserved exactly. But a killed process
|
|
397
|
+
* runs no discard, so the candidate simply survives on disk, and until now the
|
|
398
|
+
* next run allocated a fresh one and repeated hours of work.
|
|
399
|
+
*
|
|
400
|
+
* These make the survivor recognisable. The marker says "a build was underway";
|
|
401
|
+
* the head says "of this tree". Both are needed: resuming a build of a tree that
|
|
402
|
+
* has since moved on would promote a graph that never described any single state
|
|
403
|
+
* of the repository, which is worse than starting over.
|
|
404
|
+
*/
|
|
405
|
+
export const CLEAN_INDEX_IN_PROGRESS_META_KEY = "cleanIndexInProgress";
|
|
406
|
+
export const CLEAN_INDEX_HEAD_META_KEY = "cleanIndexHead";
|
|
376
407
|
function repairMetadataFamily(filePath) {
|
|
377
408
|
const segments = filePath.replaceAll("\\", "/").split("/");
|
|
378
409
|
const extension = path.extname(filePath).toLowerCase();
|
|
@@ -7059,6 +7090,12 @@ function createIndexProgressReporter(onProgress) {
|
|
|
7059
7090
|
elapsedMs: emittedAt - startedAt,
|
|
7060
7091
|
message,
|
|
7061
7092
|
...(details.modelFile === undefined ? {} : { modelFile: details.modelFile }),
|
|
7093
|
+
...(details.phaseBytesCompleted === undefined
|
|
7094
|
+
? {}
|
|
7095
|
+
: { phaseBytesCompleted: details.phaseBytesCompleted }),
|
|
7096
|
+
...(details.phaseBytesTotal === undefined
|
|
7097
|
+
? {}
|
|
7098
|
+
: { phaseBytesTotal: details.phaseBytesTotal }),
|
|
7062
7099
|
});
|
|
7063
7100
|
}
|
|
7064
7101
|
catch {
|
|
@@ -7224,6 +7261,75 @@ function sumCounts(counts) {
|
|
|
7224
7261
|
*
|
|
7225
7262
|
* Still never throws: a status call must not fail because a tally is corrupt.
|
|
7226
7263
|
*/
|
|
7264
|
+
/**
|
|
7265
|
+
* A candidate left mid-build that it is safe to continue, or null.
|
|
7266
|
+
*
|
|
7267
|
+
* Deliberately conservative: every uncertainty resolves to "start over", which
|
|
7268
|
+
* costs time, versus resuming onto the wrong tree, which produces a graph that
|
|
7269
|
+
* never described any single state of the repository and then promotes it.
|
|
7270
|
+
*
|
|
7271
|
+
* Stale and unusable candidates are removed as they are found, so a repository
|
|
7272
|
+
* does not accumulate abandoned copies of its own database.
|
|
7273
|
+
*/
|
|
7274
|
+
function findResumableCandidate(repoPath) {
|
|
7275
|
+
const head = gitHead(repoPath) ?? "";
|
|
7276
|
+
let resumable = null;
|
|
7277
|
+
for (const candidate of listCandidates(repoPath)) {
|
|
7278
|
+
let usable = false;
|
|
7279
|
+
try {
|
|
7280
|
+
const probe = new Database(candidate.databasePath, { readonly: true });
|
|
7281
|
+
try {
|
|
7282
|
+
usable =
|
|
7283
|
+
getMeta(probe, CLEAN_INDEX_IN_PROGRESS_META_KEY) === "1" &&
|
|
7284
|
+
getMeta(probe, CLEAN_INDEX_HEAD_META_KEY) === head;
|
|
7285
|
+
}
|
|
7286
|
+
finally {
|
|
7287
|
+
probe.close();
|
|
7288
|
+
}
|
|
7289
|
+
}
|
|
7290
|
+
catch {
|
|
7291
|
+
// Unreadable, not a database, or schema too old to query. Not resumable,
|
|
7292
|
+
// and not worth keeping.
|
|
7293
|
+
usable = false;
|
|
7294
|
+
}
|
|
7295
|
+
// Keep only the newest usable one. An older survivor is from an even
|
|
7296
|
+
// earlier interrupted run and has nothing to add.
|
|
7297
|
+
if (usable && resumable === null) {
|
|
7298
|
+
resumable = candidate;
|
|
7299
|
+
continue;
|
|
7300
|
+
}
|
|
7301
|
+
try {
|
|
7302
|
+
discardCandidateFiles(repoPath, candidate);
|
|
7303
|
+
}
|
|
7304
|
+
catch {
|
|
7305
|
+
// Best effort: failing to clean up a stale candidate must not stop the
|
|
7306
|
+
// rebuild that is about to replace it.
|
|
7307
|
+
}
|
|
7308
|
+
}
|
|
7309
|
+
return resumable;
|
|
7310
|
+
}
|
|
7311
|
+
/** Write the coverage tally as it stands. Safe to call repeatedly mid-index. */
|
|
7312
|
+
function persistUnparsedTally(db, tally) {
|
|
7313
|
+
setMeta(db, COVERAGE_UNPARSED_META_KEY, JSON.stringify(Object.fromEntries(tally)));
|
|
7314
|
+
}
|
|
7315
|
+
/**
|
|
7316
|
+
* True when the stored tally came from a run that did not finish.
|
|
7317
|
+
*
|
|
7318
|
+
* Read separately from the tally itself because the two answer different
|
|
7319
|
+
* questions: the tally says what was seen, this says whether that is all there
|
|
7320
|
+
* was to see.
|
|
7321
|
+
*/
|
|
7322
|
+
function unparsedTallyIsIncomplete(db) {
|
|
7323
|
+
if (!db)
|
|
7324
|
+
return false;
|
|
7325
|
+
try {
|
|
7326
|
+
return getMeta(db, COVERAGE_TALLY_IN_PROGRESS_META_KEY) === "1";
|
|
7327
|
+
}
|
|
7328
|
+
catch {
|
|
7329
|
+
// Unreadable meta is not a claim that the tally is complete.
|
|
7330
|
+
return true;
|
|
7331
|
+
}
|
|
7332
|
+
}
|
|
7227
7333
|
function readUnparsedTally(db) {
|
|
7228
7334
|
if (!db)
|
|
7229
7335
|
return null;
|
|
@@ -7290,6 +7396,10 @@ function worstSemanticReadiness(states) {
|
|
|
7290
7396
|
function buildCoverageSkips(skippedByExtension, db) {
|
|
7291
7397
|
const byExtension = sortedTally(skippedByExtension);
|
|
7292
7398
|
const unparsed = readUnparsedTally(db);
|
|
7399
|
+
// A tally that exists but came from an unfinished run is as unmeasured as an
|
|
7400
|
+
// absent one: it counts only the files that run happened to reach. Both cases
|
|
7401
|
+
// set `unparsedUnknown`, which downstream already renders as a lower bound.
|
|
7402
|
+
const incomplete = unparsed === null || unparsedTallyIsIncomplete(db);
|
|
7293
7403
|
return {
|
|
7294
7404
|
byExtension,
|
|
7295
7405
|
// Reported as empty when unknown so consumers reading only this field are
|
|
@@ -7297,7 +7407,7 @@ function buildCoverageSkips(skippedByExtension, db) {
|
|
|
7297
7407
|
// measurement. `total` is then a lower bound, not a count.
|
|
7298
7408
|
unparsedByExtension: unparsed ?? {},
|
|
7299
7409
|
total: sumCounts(byExtension) + sumCounts(unparsed ?? {}),
|
|
7300
|
-
...(
|
|
7410
|
+
...(incomplete ? { unparsedUnknown: true } : {}),
|
|
7301
7411
|
};
|
|
7302
7412
|
}
|
|
7303
7413
|
/**
|
|
@@ -7554,13 +7664,23 @@ function buildFreshnessEnvelope(repoPath, db, verifiedAt, stateOverride, gitProb
|
|
|
7554
7664
|
* Snapshot a file's on-disk mtime/size into `index_state`, so a later cold start
|
|
7555
7665
|
* can tell whether it drifted while no knodin process was watching.
|
|
7556
7666
|
*/
|
|
7667
|
+
/**
|
|
7668
|
+
* Record a file's mtime and size, returning the size.
|
|
7669
|
+
*
|
|
7670
|
+
* The size is returned rather than discarded so byte-based progress costs no
|
|
7671
|
+
* extra I/O: this already stats every indexed file, and a second pass over
|
|
7672
|
+
* 900,000 files to learn what it just measured would be pure waste. Returns 0
|
|
7673
|
+
* when the file could not be stat'd, which the caller adds harmlessly.
|
|
7674
|
+
*/
|
|
7557
7675
|
function recordIndexState(db, repoPath, relPath) {
|
|
7558
7676
|
try {
|
|
7559
7677
|
const st = fs.statSync(path.join(repoPath, relPath));
|
|
7560
7678
|
db.run("INSERT INTO index_state(filePath, mtimeMs, size) VALUES (?, ?, ?) ON CONFLICT(filePath) DO UPDATE SET mtimeMs = excluded.mtimeMs, size = excluded.size", [relPath, st.mtimeMs, st.size]);
|
|
7679
|
+
return st.size;
|
|
7561
7680
|
}
|
|
7562
7681
|
catch (_) {
|
|
7563
7682
|
// File vanished between indexing and stat — reconcile-delete handles it.
|
|
7683
|
+
return 0;
|
|
7564
7684
|
}
|
|
7565
7685
|
}
|
|
7566
7686
|
/** Drop a file's `index_state` row (used when a file is deleted). */
|
|
@@ -7638,7 +7758,7 @@ function detectDriftByStat(repoPath, db, canonicalFiles) {
|
|
|
7638
7758
|
}
|
|
7639
7759
|
return changed;
|
|
7640
7760
|
}
|
|
7641
|
-
async function reconcileIndex(repoPath, db, progress) {
|
|
7761
|
+
async function reconcileIndex(repoPath, db, progress, skipEmbeddings = false) {
|
|
7642
7762
|
try {
|
|
7643
7763
|
progress?.("collecting-files", 0, "Checking existing local graph for changes");
|
|
7644
7764
|
const changed = new Set();
|
|
@@ -7757,18 +7877,26 @@ async function reconcileIndex(repoPath, db, progress) {
|
|
|
7757
7877
|
// Re-embed ONLY what changed (indexEmbeddings embeds symbols lacking an
|
|
7758
7878
|
// embedding), so a no-op reconcile triggers no embedding work at all.
|
|
7759
7879
|
if (reindexed > 0) {
|
|
7760
|
-
progress?.("finalizing", 0,
|
|
7880
|
+
progress?.("finalizing", 0, skipEmbeddings
|
|
7881
|
+
? "Refreshing identities (semantic embeddings deferred)"
|
|
7882
|
+
: "Refreshing identities and semantic embeddings");
|
|
7761
7883
|
persistSymbolIdentities(db, repoPath, reindexedPaths);
|
|
7762
7884
|
reconcileTypeScriptDi(db, repoPath);
|
|
7763
|
-
|
|
7885
|
+
if (!skipEmbeddings)
|
|
7886
|
+
await indexEmbeddings(db, repoPath, progress);
|
|
7764
7887
|
indexGeneration++;
|
|
7765
7888
|
}
|
|
7766
|
-
else if (semanticReadinessFor(db) !== "ready") {
|
|
7889
|
+
else if (!skipEmbeddings && semanticReadinessFor(db) !== "ready") {
|
|
7767
7890
|
// Unchanged files do NOT imply current embeddings: a deferred
|
|
7768
7891
|
// (`skipEmbeddings`) or interrupted pass leaves symbols unembedded while
|
|
7769
7892
|
// every file looks reconciled. Without this, the follow-up index a user
|
|
7770
7893
|
// is told to run would do nothing and semantic search would stay
|
|
7771
7894
|
// silently short forever.
|
|
7895
|
+
//
|
|
7896
|
+
// Which is also why `skipEmbeddings` has to suppress it: this branch
|
|
7897
|
+
// exists to COMPLETE a deferred pass, so leaving it unguarded would make
|
|
7898
|
+
// `--skip-embeddings` a silent no-op on exactly the warm graphs large
|
|
7899
|
+
// enough to want it — the flag would appear to work and do the opposite.
|
|
7772
7900
|
progress?.("finalizing", 0, "Completing deferred semantic embeddings");
|
|
7773
7901
|
await indexEmbeddings(db, repoPath, progress);
|
|
7774
7902
|
indexGeneration++;
|
|
@@ -8146,7 +8274,17 @@ function stalenessFor(repoPath) {
|
|
|
8146
8274
|
return freshnessProbes.get(path.resolve(repoPath))?.staleness ?? "unknown";
|
|
8147
8275
|
}
|
|
8148
8276
|
/** Recursively indexes all matching files within the repository. */
|
|
8149
|
-
async function indexRepo(repoPath, db, progress, skipEmbeddings = false, publishProcessState = true
|
|
8277
|
+
async function indexRepo(repoPath, db, progress, skipEmbeddings = false, publishProcessState = true,
|
|
8278
|
+
/**
|
|
8279
|
+
* Continue a build already in this database instead of starting over.
|
|
8280
|
+
*
|
|
8281
|
+
* Skips the wipe below and skips files whose `index_state` row still matches
|
|
8282
|
+
* disk. Only ever set for a candidate database that a previous run left
|
|
8283
|
+
* mid-build: the live graph is never resumed into, because resuming implies
|
|
8284
|
+
* partially-populated intermediate state and the live graph must never be
|
|
8285
|
+
* observable in that condition.
|
|
8286
|
+
*/
|
|
8287
|
+
resume = false) {
|
|
8150
8288
|
progress?.("collecting-files", 0, "Discovering indexable files");
|
|
8151
8289
|
const collected = measurePerfPhaseSync("file_collection", () => collectRepoFilesWithCoverage(repoPath));
|
|
8152
8290
|
const files = collected.files;
|
|
@@ -8154,15 +8292,82 @@ async function indexRepo(repoPath, db, progress, skipEmbeddings = false, publish
|
|
|
8154
8292
|
// during the parse. Persisted below so `status` can report the gap without
|
|
8155
8293
|
// re-indexing the repository to rediscover it.
|
|
8156
8294
|
const unparsedByExtension = new Map();
|
|
8295
|
+
// Claim the tally as in-progress BEFORE the first file. An index killed
|
|
8296
|
+
// partway used to leave the previous run's tally in place and reported as
|
|
8297
|
+
// fact; now whatever is stored is flagged a lower bound until this run
|
|
8298
|
+
// finishes and clears the marker.
|
|
8299
|
+
setMeta(db, COVERAGE_TALLY_IN_PROGRESS_META_KEY, "1");
|
|
8300
|
+
// One stat pass to learn the weight of the work before starting it. Counted
|
|
8301
|
+
// separately from files because the two diverge sharply: on a real Salesforce
|
|
8302
|
+
// checkout 0.05% of the files hold 78% of the bytes, so a files-only estimate
|
|
8303
|
+
// is confidently wrong rather than merely rough.
|
|
8304
|
+
//
|
|
8305
|
+
// `size` is a floor on cost, not a proxy for it — a large XML file is not
|
|
8306
|
+
// exactly proportional to a large TypeScript one — but it tracks the actual
|
|
8307
|
+
// shape of the work far better than a file count, which treats a 7 MB profile
|
|
8308
|
+
// and a 200-byte translation as equal.
|
|
8309
|
+
let totalBytes = 0;
|
|
8310
|
+
for (const file of files) {
|
|
8311
|
+
try {
|
|
8312
|
+
totalBytes += fs.statSync(path.join(repoPath, file)).size;
|
|
8313
|
+
}
|
|
8314
|
+
catch {
|
|
8315
|
+
// Unreadable or vanished: it will fail in the loop below too, where the
|
|
8316
|
+
// failure is reported. Excluding it here only makes the total a floor.
|
|
8317
|
+
}
|
|
8318
|
+
}
|
|
8319
|
+
let completedBytes = 0;
|
|
8157
8320
|
progress?.("indexing-files", 0, `Indexing ${files.length.toLocaleString()} files`, {
|
|
8158
8321
|
phaseTotal: files.length,
|
|
8322
|
+
phaseBytesCompleted: 0,
|
|
8323
|
+
phaseBytesTotal: totalBytes,
|
|
8159
8324
|
});
|
|
8160
|
-
// Clean out existing data to ensure consistency on full re-index
|
|
8161
|
-
|
|
8162
|
-
|
|
8163
|
-
|
|
8164
|
-
|
|
8165
|
-
|
|
8325
|
+
// Clean out existing data to ensure consistency on full re-index.
|
|
8326
|
+
//
|
|
8327
|
+
// Skipped when resuming, which is the whole point: this wipe is why an
|
|
8328
|
+
// interrupted clean index used to lose everything it had done. Resuming into
|
|
8329
|
+
// a database it had just emptied would be indistinguishable from starting
|
|
8330
|
+
// over.
|
|
8331
|
+
if (!resume) {
|
|
8332
|
+
deleteAllSymbols(db);
|
|
8333
|
+
db.run('DELETE FROM "references";');
|
|
8334
|
+
db.run("DELETE FROM dependencies;");
|
|
8335
|
+
db.run("DELETE FROM mcp_tools;");
|
|
8336
|
+
db.run("DELETE FROM index_state;");
|
|
8337
|
+
}
|
|
8338
|
+
// On a resume, everything already recorded and still matching disk is done.
|
|
8339
|
+
// Counted toward progress rather than dropped from it, so the totals stay
|
|
8340
|
+
// whole-repository and the run visibly picks up where it stopped instead of
|
|
8341
|
+
// appearing to start a smaller job.
|
|
8342
|
+
//
|
|
8343
|
+
// Sizes come from the stored `index_state` rows, not fresh stats: the drift
|
|
8344
|
+
// check just proved they still match, so re-measuring 600,000 files to learn
|
|
8345
|
+
// what the database already knows would be the expensive way to be no more
|
|
8346
|
+
// correct.
|
|
8347
|
+
let workList = files;
|
|
8348
|
+
let resumedFiles = 0;
|
|
8349
|
+
let resumedBytes = 0;
|
|
8350
|
+
if (resume) {
|
|
8351
|
+
const pending = [];
|
|
8352
|
+
for (const file of files) {
|
|
8353
|
+
if (fileDriftedFromIndexState(db, repoPath, file)) {
|
|
8354
|
+
pending.push(file);
|
|
8355
|
+
continue;
|
|
8356
|
+
}
|
|
8357
|
+
resumedFiles++;
|
|
8358
|
+
resumedBytes +=
|
|
8359
|
+
db
|
|
8360
|
+
.query("SELECT size FROM index_state WHERE filePath = ?")
|
|
8361
|
+
.get(file)?.size ?? 0;
|
|
8362
|
+
}
|
|
8363
|
+
workList = pending;
|
|
8364
|
+
completedBytes = resumedBytes;
|
|
8365
|
+
progress?.("indexing-files", resumedFiles, `Resuming: ${resumedFiles.toLocaleString()} of ${files.length.toLocaleString()} files already indexed`, {
|
|
8366
|
+
phaseTotal: files.length,
|
|
8367
|
+
phaseBytesCompleted: resumedBytes,
|
|
8368
|
+
phaseBytesTotal: totalBytes,
|
|
8369
|
+
});
|
|
8370
|
+
}
|
|
8166
8371
|
// Process files sequentially to ensure thread-safe SQLite transactions.
|
|
8167
8372
|
// Extraction (buildFileIndexResult, pure) still happens one file at a
|
|
8168
8373
|
// time here; only the write is batched, via `pendingWrites`, so the
|
|
@@ -8183,14 +8388,21 @@ async function indexRepo(repoPath, db, progress, skipEmbeddings = false, publish
|
|
|
8183
8388
|
};
|
|
8184
8389
|
// Progress is reported against a single counter so the pool and the
|
|
8185
8390
|
// sequential path can interleave without the count going backwards.
|
|
8186
|
-
let processed =
|
|
8391
|
+
let processed = resumedFiles;
|
|
8187
8392
|
const afterFile = async (file) => {
|
|
8188
8393
|
if (pendingWrites.length >= INDEX_WRITE_BATCH_SIZE)
|
|
8189
8394
|
flushPendingWrites();
|
|
8190
|
-
recordIndexState(db, repoPath, file);
|
|
8395
|
+
completedBytes += recordIndexState(db, repoPath, file);
|
|
8191
8396
|
processed++;
|
|
8397
|
+
// Persisted on the same cadence as the write batches rather than only at
|
|
8398
|
+
// the end, so an interrupted index keeps what it learned. Cheap: one meta
|
|
8399
|
+
// row per 200 files, not per file.
|
|
8400
|
+
if (processed % INDEX_WRITE_BATCH_SIZE === 0)
|
|
8401
|
+
persistUnparsedTally(db, unparsedByExtension);
|
|
8192
8402
|
progress?.("indexing-files", processed, `Indexing ${files.length.toLocaleString()} files`, {
|
|
8193
8403
|
phaseTotal: files.length,
|
|
8404
|
+
phaseBytesCompleted: completedBytes,
|
|
8405
|
+
phaseBytesTotal: totalBytes,
|
|
8194
8406
|
});
|
|
8195
8407
|
if (Date.now() - lastYieldAt >= INDEX_EVENT_LOOP_YIELD_MS) {
|
|
8196
8408
|
await yieldToIndexEventLoop();
|
|
@@ -8215,7 +8427,7 @@ async function indexRepo(repoPath, db, progress, skipEmbeddings = false, publish
|
|
|
8215
8427
|
// cannot disagree about where a file belongs.
|
|
8216
8428
|
const eligible = [];
|
|
8217
8429
|
const inline = [];
|
|
8218
|
-
for (const file of
|
|
8430
|
+
for (const file of workList) {
|
|
8219
8431
|
if (isWorkerEligibleFile(path.join(repoPath, file), file))
|
|
8220
8432
|
eligible.push({ absolutePath: path.join(repoPath, file), relativePath: file, repoPath });
|
|
8221
8433
|
else
|
|
@@ -8246,7 +8458,7 @@ async function indexRepo(repoPath, db, progress, skipEmbeddings = false, publish
|
|
|
8246
8458
|
}
|
|
8247
8459
|
}
|
|
8248
8460
|
else {
|
|
8249
|
-
for (const file of
|
|
8461
|
+
for (const file of workList) {
|
|
8250
8462
|
await indexOneInline(file);
|
|
8251
8463
|
await afterFile(file);
|
|
8252
8464
|
}
|
|
@@ -8270,7 +8482,11 @@ async function indexRepo(repoPath, db, progress, skipEmbeddings = false, publish
|
|
|
8270
8482
|
setMeta(db, "knodinVersion", KNODIN_VERSION);
|
|
8271
8483
|
setMeta(db, "lastSuccessfulReconciliation", new Date().toISOString());
|
|
8272
8484
|
setMeta(db, "mcpBackfillVersion", "17");
|
|
8273
|
-
|
|
8485
|
+
persistUnparsedTally(db, unparsedByExtension);
|
|
8486
|
+
// Only now is the tally a measurement rather than a lower bound. Clearing the
|
|
8487
|
+
// marker LAST, after the final write, means any failure above leaves it set
|
|
8488
|
+
// and the tally honestly flagged incomplete.
|
|
8489
|
+
setMeta(db, COVERAGE_TALLY_IN_PROGRESS_META_KEY, "");
|
|
8274
8490
|
recordFreshnessBaseline(repoPath, db);
|
|
8275
8491
|
if (publishProcessState)
|
|
8276
8492
|
indexGeneration++;
|
|
@@ -11340,6 +11556,13 @@ function withStaleness(engine, claimRepository, openDb, openPolicy) {
|
|
|
11340
11556
|
row.staleness = staleness;
|
|
11341
11557
|
return page;
|
|
11342
11558
|
},
|
|
11559
|
+
searchText(term, repoPath, options) {
|
|
11560
|
+
claimRepository(repoPath);
|
|
11561
|
+
// No staleness annotation: this reads the working tree directly rather
|
|
11562
|
+
// than the graph, so its answers are current by construction and
|
|
11563
|
+
// stamping them with the index's freshness would misreport them.
|
|
11564
|
+
return engine.searchText(term, repoPath, options);
|
|
11565
|
+
},
|
|
11343
11566
|
async query(pattern, target, repoPath, to, limit, depth, detailLevel, selector, impactOptions, options) {
|
|
11344
11567
|
claimRepository(repoPath);
|
|
11345
11568
|
const result = await engine.query(pattern, target, repoPath, to, limit, depth, detailLevel, selector, impactOptions, options);
|
|
@@ -11654,7 +11877,18 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
|
|
|
11654
11877
|
if (!db)
|
|
11655
11878
|
throw new Error("candidate database did not open");
|
|
11656
11879
|
if (!changes) {
|
|
11657
|
-
|
|
11880
|
+
// Claim the build before touching anything, and record which tree it
|
|
11881
|
+
// is for. A candidate carrying this marker is one a previous run left
|
|
11882
|
+
// mid-build; without the commit, a resume could silently continue a
|
|
11883
|
+
// build of a tree that has since moved on and promote a graph that
|
|
11884
|
+
// never described any single state of the repository.
|
|
11885
|
+
setMeta(db, CLEAN_INDEX_IN_PROGRESS_META_KEY, "1");
|
|
11886
|
+
setMeta(db, CLEAN_INDEX_HEAD_META_KEY, gitHead(resolved) ?? "");
|
|
11887
|
+
await indexRepo(resolved, db, options?.onProgress, options?.skipEmbeddings === true, false, options?.resume === true);
|
|
11888
|
+
// Cleared only after indexRepo returns. Anything that fails or is
|
|
11889
|
+
// killed above leaves it set, which is exactly what makes the
|
|
11890
|
+
// candidate recognisable as resumable rather than abandoned.
|
|
11891
|
+
setMeta(db, CLEAN_INDEX_IN_PROGRESS_META_KEY, "");
|
|
11658
11892
|
return { reconciled: collectRepoFiles(resolved) };
|
|
11659
11893
|
}
|
|
11660
11894
|
const reconciled = [...new Set(changes)]
|
|
@@ -13491,6 +13725,27 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
|
|
|
13491
13725
|
},
|
|
13492
13726
|
async index(repoPath, files, clean = false, options) {
|
|
13493
13727
|
const resolved = path.resolve(repoPath);
|
|
13728
|
+
// Expand `--under` before anything else, so everything downstream sees an
|
|
13729
|
+
// ordinary explicit-file index and no path can route a scoped rebuild
|
|
13730
|
+
// into the destructive full-index branch.
|
|
13731
|
+
const scopedUnder = options?.under;
|
|
13732
|
+
if (scopedUnder !== undefined && (!files || files.length === 0)) {
|
|
13733
|
+
const prefix = path
|
|
13734
|
+
.relative(resolved, path.resolve(resolved, scopedUnder))
|
|
13735
|
+
.split(path.sep)
|
|
13736
|
+
.join("/");
|
|
13737
|
+
if (prefix.startsWith("..") || path.isAbsolute(prefix))
|
|
13738
|
+
throw new Error(`knodin index --under: ${scopedUnder} is outside the repository`);
|
|
13739
|
+
const collected = collectRepoFilesWithCoverage(resolved);
|
|
13740
|
+
files =
|
|
13741
|
+
prefix === ""
|
|
13742
|
+
? collected.files
|
|
13743
|
+
: collected.files.filter((file) => file === prefix || file.startsWith(`${prefix}/`));
|
|
13744
|
+
// Failing loudly rather than indexing nothing and reporting success:
|
|
13745
|
+
// a silent no-op here would look identical to a completed rebuild.
|
|
13746
|
+
if (files.length === 0)
|
|
13747
|
+
throw new Error(`knodin index --under: no indexable files under ${scopedUnder || "the repository root"}`);
|
|
13748
|
+
}
|
|
13494
13749
|
const indexCoordination = acquireLifecycleCoordination({
|
|
13495
13750
|
command: "index",
|
|
13496
13751
|
repository: resolved,
|
|
@@ -13503,13 +13758,27 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
|
|
|
13503
13758
|
!process.env.VITEST &&
|
|
13504
13759
|
process.env.NODE_ENV !== "test") {
|
|
13505
13760
|
const progress = createIndexProgressReporter(options?.onProgress);
|
|
13506
|
-
|
|
13507
|
-
|
|
13761
|
+
// Look for a build a previous run left unfinished before starting a
|
|
13762
|
+
// new one. A candidate survives only when the process was killed —
|
|
13763
|
+
// a failure discards it — so a survivor is precisely the case worth
|
|
13764
|
+
// continuing.
|
|
13765
|
+
const resumable = findResumableCandidate(resolved);
|
|
13766
|
+
// Adopt it into the engine's registry. Ownership is what
|
|
13767
|
+
// reconcile/audit/promote check, and a candidate recovered from disk
|
|
13768
|
+
// was never registered because the process that created it is gone.
|
|
13769
|
+
if (resumable)
|
|
13770
|
+
candidates.set(resumable.id, resumable);
|
|
13771
|
+
const candidate = resumable ?? (await engine.createCandidate(resolved));
|
|
13772
|
+
if (resumable)
|
|
13773
|
+
progress("starting", 0, "Resuming the interrupted clean index (previous progress kept)");
|
|
13774
|
+
else
|
|
13775
|
+
progress("starting", 0, "Creating isolated clean-index candidate");
|
|
13508
13776
|
let promoted = false;
|
|
13509
13777
|
try {
|
|
13510
13778
|
const reconciliation = await engine.reconcileCandidate(candidate, resolved, undefined, {
|
|
13511
13779
|
onProgress: progress,
|
|
13512
13780
|
skipEmbeddings: options?.skipEmbeddings,
|
|
13781
|
+
resume: resumable !== null,
|
|
13513
13782
|
});
|
|
13514
13783
|
progress("verifying", 0, "Deep-auditing clean-index candidate", {
|
|
13515
13784
|
phaseTotal: 1,
|
|
@@ -13650,7 +13919,7 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
|
|
|
13650
13919
|
catch (_) { }
|
|
13651
13920
|
if (!clean && hasSymbols) {
|
|
13652
13921
|
progress("collecting-files", 0, "Checking existing local graph for changes");
|
|
13653
|
-
await reconcileIndex(repoPath, db, progress);
|
|
13922
|
+
await reconcileIndex(repoPath, db, progress, options?.skipEmbeddings === true);
|
|
13654
13923
|
}
|
|
13655
13924
|
else {
|
|
13656
13925
|
await indexRepo(repoPath, db, progress, options?.skipEmbeddings === true);
|
|
@@ -13768,9 +14037,26 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
|
|
|
13768
14037
|
progress("completed", completed, health.status === "healthy"
|
|
13769
14038
|
? completionMessage
|
|
13770
14039
|
: `${completionMessage}; verification found ${issueCount.toLocaleString()} issue(s)`, { phaseTotal: completed });
|
|
14040
|
+
// A scoped rebuild leaves the stored coverage tally describing a
|
|
14041
|
+
// different run than the graph it now sits beside. Left alone it
|
|
14042
|
+
// would still be present, and therefore still reported as measured
|
|
14043
|
+
// fact — the stale-and-authoritative case. Marking it incomplete
|
|
14044
|
+
// makes `status` report a lower bound until a full index restores a
|
|
14045
|
+
// whole-repository measurement.
|
|
14046
|
+
//
|
|
14047
|
+
// Only for an explicit `--under`, not for ordinary file arguments:
|
|
14048
|
+
// the lifecycle hooks index changed files constantly, and flagging
|
|
14049
|
+
// the tally on every commit would make the signal meaningless noise.
|
|
14050
|
+
if (scopedUnder !== undefined)
|
|
14051
|
+
setMeta(db, COVERAGE_TALLY_IN_PROGRESS_META_KEY, "1");
|
|
13771
14052
|
return {
|
|
13772
14053
|
indexed,
|
|
13773
14054
|
unchanged,
|
|
14055
|
+
// Measured from the graph rather than inferred from the flag: a
|
|
14056
|
+
// run that deferred embeddings and a run that never had any are
|
|
14057
|
+
// the same state to a consumer, and a partially embedded graph is
|
|
14058
|
+
// neither.
|
|
14059
|
+
semanticReadiness: semanticReadinessFor(db),
|
|
13774
14060
|
scip: scipReport,
|
|
13775
14061
|
sarif: sarifReport,
|
|
13776
14062
|
verification: {
|
|
@@ -13795,6 +14081,15 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
|
|
|
13795
14081
|
indexCoordination.release();
|
|
13796
14082
|
}
|
|
13797
14083
|
},
|
|
14084
|
+
async searchText(term, repoPath, options = {}) {
|
|
14085
|
+
if (term.length === 0)
|
|
14086
|
+
throw new Error("knodin searchText: term must not be empty");
|
|
14087
|
+
// Reads the working tree, not the graph, so it needs no open database
|
|
14088
|
+
// and no freshness check. `getLanguageForFile` is the same grammar
|
|
14089
|
+
// loader indexing uses, which is what makes the classification
|
|
14090
|
+
// evidence rather than a heuristic over file extensions.
|
|
14091
|
+
return searchRepoText(path.resolve(repoPath), term, getLanguageForFile, () => new Parser(), options);
|
|
14092
|
+
},
|
|
13798
14093
|
async search(query, repoPath, limit, options = {}) {
|
|
13799
14094
|
const resolvedRepoPath = path.resolve(repoPath);
|
|
13800
14095
|
const offset = options.offset ?? 0;
|