knodin 0.6.0 → 0.7.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +26 -3
- package/dist/bin/cli.js +168 -12
- package/dist/bin/launcher.js +11 -0
- package/dist/src/agent-integration.js +3 -1
- package/dist/src/cli-args.js +8 -1
- package/dist/src/cli-model.js +35 -1
- package/dist/src/codeflow-replay.js +80 -0
- package/dist/src/competitive-cold-mcp.js +40 -0
- package/dist/src/competitive-manifest.js +106 -25
- package/dist/src/competitive-runner.js +37 -3
- package/dist/src/competitive-sandbox.js +1 -1
- package/dist/src/diagnostics.js +449 -0
- package/dist/src/engine/git-history.js +289 -0
- package/dist/src/engine/index.js +417 -70
- package/dist/src/engine/scip-import.js +408 -0
- package/dist/src/execution-profile.js +203 -0
- package/dist/src/failure-diagnosis.js +69 -10
- package/dist/src/hook-manager-integration.js +156 -0
- package/dist/src/init.js +319 -33
- package/dist/src/lifecycle-health.js +42 -4
- package/dist/src/output-telemetry.js +4 -0
- package/dist/src/progressive-evidence.js +473 -0
- package/dist/src/pure-compression-cli.js +101 -0
- package/dist/src/release-preflight.js +510 -0
- package/dist/src/repository-management.js +142 -0
- package/dist/src/response-budget.js +11 -1
- package/dist/src/server.js +22 -2
- package/dist/src/structural-fast-path.js +303 -0
- package/dist/src/structural-snapshot.js +33 -0
- package/dist/src/tools/knodin-tools.js +105 -14
- package/dist/src/update-ceremony.js +158 -0
- package/docs/CLI.md +22 -0
- package/docs/COMMAND-OUTPUT-COMPRESSION.md +31 -15
- package/docs/CONTAINED-EXECUTION.md +77 -0
- package/docs/DIAGNOSTICS.md +45 -0
- package/docs/DOCTOR-AND-UPDATES.md +5 -2
- package/docs/GIT-HISTORY-REVIEW.md +39 -0
- package/docs/MCP.md +15 -0
- package/docs/PROGRESSIVE-EVIDENCE.md +37 -0
- package/docs/REPOSITORIES-AND-WORKTREES.md +30 -0
- package/docs/SCIP-IMPORT.md +57 -0
- package/docs/SIGNED-UPDATES.md +5 -0
- package/docs/TELEMETRY.md +4 -0
- package/docs/releases/0.7.0.md +24 -0
- package/docs/releases/0.7.1.md +21 -0
- package/docs/releases/0.7.2.md +21 -0
- package/docs/releases/0.7.3.md +23 -0
- package/package.json +33 -2
package/dist/src/engine/index.js
CHANGED
|
@@ -20,12 +20,15 @@ import chokidar from "chokidar";
|
|
|
20
20
|
import Parser from "web-tree-sitter";
|
|
21
21
|
import { readIndexActivity } from "../index-activity.js";
|
|
22
22
|
import { runReadOnlyLspQuery } from "../lsp-readonly.js";
|
|
23
|
+
import { contentFingerprint, writeStructuralSnapshot, } from "../structural-snapshot.js";
|
|
23
24
|
import { KNODIN_VERSION } from "../version.js";
|
|
24
25
|
import * as ann from "./ann-hnsw.js";
|
|
25
26
|
import { computeSimilarity, generateEmbedding, generateEmbeddings, } from "./embeddings.js";
|
|
26
27
|
import { walkRepoFiles } from "./file-walker.js";
|
|
28
|
+
import { clearGitHistorySignalCache, collectGitHistorySignals, } from "./git-history.js";
|
|
27
29
|
import { beginPerfPhase, measurePerfPhase, measurePerfPhaseSync } from "./perf.js";
|
|
28
30
|
import { isIndexablePath, makeWatchIgnorePredicate } from "./prune.js";
|
|
31
|
+
import { readScipIndex, SCIP_DEFAULT_LIMITS, } from "./scip-import.js";
|
|
29
32
|
import { isIndexableSourcePath } from "./source-policy.js";
|
|
30
33
|
import { Database } from "./sqlite.js";
|
|
31
34
|
import { deleteAllSymbols, deleteSymbolsForFile, deleteSymbolsMatchingPath, ORPHANED_EMBEDDING_PREDICATE, purgeOrphanEmbeddings, } from "./symbol-delete.js";
|
|
@@ -3900,6 +3903,73 @@ async function indexLsifDump(absolutePath, _relativePath, repoPath, db) {
|
|
|
3900
3903
|
console.error("Error indexing LSIF:", error);
|
|
3901
3904
|
}
|
|
3902
3905
|
}
|
|
3906
|
+
/** Replace only the optional SCIP tier, preserving native and LSIF facts verbatim. */
|
|
3907
|
+
function persistScipFacts(db, repoPath, facts) {
|
|
3908
|
+
const affectedFiles = [...new Set(facts.symbols.map((symbol) => symbol.filePath))].sort();
|
|
3909
|
+
const insertSymbol = db.prepare(`
|
|
3910
|
+
INSERT INTO symbols (name, kind, filePath, startLine, endLine, startCol, endCol, summary)
|
|
3911
|
+
VALUES (?, ?, ?, ?, ?, ?, ?, ?)
|
|
3912
|
+
`);
|
|
3913
|
+
const insertReference = db.prepare(`
|
|
3914
|
+
INSERT INTO "references" (callerSymbol, callerFile, calleeSymbol, calleeFile, line, column, kind, confidence)
|
|
3915
|
+
VALUES (?, ?, ?, ?, ?, ?, ?, 1.0)
|
|
3916
|
+
`);
|
|
3917
|
+
const insertDependency = db.prepare(`
|
|
3918
|
+
INSERT INTO dependencies (fromFile, toFile, kind, confidence, sourceEvidence)
|
|
3919
|
+
VALUES (?, ?, 'scip_import', 1.0, ?)
|
|
3920
|
+
`);
|
|
3921
|
+
db.run("BEGIN TRANSACTION;");
|
|
3922
|
+
try {
|
|
3923
|
+
db.run("DELETE FROM symbols WHERE kind LIKE 'SCIP_%'");
|
|
3924
|
+
db.run("DELETE FROM \"references\" WHERE kind LIKE 'scip_%'");
|
|
3925
|
+
db.run("DELETE FROM dependencies WHERE kind = 'scip_import'");
|
|
3926
|
+
for (const symbol of facts.symbols) {
|
|
3927
|
+
insertSymbol.run(symbol.name, symbol.kind, symbol.filePath, symbol.startLine, symbol.endLine, symbol.startCol, symbol.endCol, JSON.stringify({ provenance: "scip", symbol: symbol.symbol }));
|
|
3928
|
+
}
|
|
3929
|
+
const dependencies = new Set();
|
|
3930
|
+
for (const reference of facts.references) {
|
|
3931
|
+
insertReference.run(reference.callerSymbol, reference.callerFile, reference.calleeSymbol, reference.calleeFile, reference.line, reference.column, reference.kind);
|
|
3932
|
+
if (reference.kind === "scip_reference" && reference.callerFile !== reference.calleeFile) {
|
|
3933
|
+
const key = `${reference.callerFile}\0${reference.calleeFile}`;
|
|
3934
|
+
if (!dependencies.has(key)) {
|
|
3935
|
+
dependencies.add(key);
|
|
3936
|
+
insertDependency.run(reference.callerFile, reference.calleeFile, JSON.stringify({ provenance: "scip", input: facts.inputPath }));
|
|
3937
|
+
}
|
|
3938
|
+
}
|
|
3939
|
+
}
|
|
3940
|
+
db.run("COMMIT;");
|
|
3941
|
+
}
|
|
3942
|
+
catch (error) {
|
|
3943
|
+
db.run("ROLLBACK;");
|
|
3944
|
+
throw error;
|
|
3945
|
+
}
|
|
3946
|
+
finally {
|
|
3947
|
+
insertSymbol.finalize();
|
|
3948
|
+
insertReference.finalize();
|
|
3949
|
+
insertDependency.finalize();
|
|
3950
|
+
}
|
|
3951
|
+
persistSymbolIdentities(db, repoPath, affectedFiles);
|
|
3952
|
+
return affectedFiles;
|
|
3953
|
+
}
|
|
3954
|
+
/** Any source refresh makes the previous compiler snapshot stale as a unit. */
|
|
3955
|
+
function invalidateScipFacts(db) {
|
|
3956
|
+
const present = db
|
|
3957
|
+
.query("SELECT 1 AS present FROM symbols WHERE kind LIKE 'SCIP_%' LIMIT 1")
|
|
3958
|
+
.get();
|
|
3959
|
+
if (!present)
|
|
3960
|
+
return;
|
|
3961
|
+
db.run("BEGIN TRANSACTION;");
|
|
3962
|
+
try {
|
|
3963
|
+
db.run("DELETE FROM symbols WHERE kind LIKE 'SCIP_%'");
|
|
3964
|
+
db.run("DELETE FROM \"references\" WHERE kind LIKE 'scip_%'");
|
|
3965
|
+
db.run("DELETE FROM dependencies WHERE kind = 'scip_import'");
|
|
3966
|
+
db.run("COMMIT;");
|
|
3967
|
+
}
|
|
3968
|
+
catch (error) {
|
|
3969
|
+
db.run("ROLLBACK;");
|
|
3970
|
+
throw error;
|
|
3971
|
+
}
|
|
3972
|
+
}
|
|
3903
3973
|
async function indexTerraformFile(content, relativePath, _repoPath, db) {
|
|
3904
3974
|
const symbols = [];
|
|
3905
3975
|
const references = [];
|
|
@@ -5014,6 +5084,7 @@ export async function indexDbtManifestFile(content, _absolutePath, relativePath,
|
|
|
5014
5084
|
/** Index a single file's symbols and references into the SQLite database. */
|
|
5015
5085
|
async function indexFile(absolutePath, relativePath, repoPath, db) {
|
|
5016
5086
|
try {
|
|
5087
|
+
invalidateScipFacts(db);
|
|
5017
5088
|
const baseName = path.basename(absolutePath).toLowerCase();
|
|
5018
5089
|
if (baseName === "manifest.json") {
|
|
5019
5090
|
const content = fs.readFileSync(absolutePath, "utf-8");
|
|
@@ -6112,7 +6183,16 @@ function noteFreshness(repoPath, staleness) {
|
|
|
6112
6183
|
async function ensureIndexFresh(repoPath, db) {
|
|
6113
6184
|
const key = path.resolve(repoPath);
|
|
6114
6185
|
const cached = freshnessProbes.get(key);
|
|
6115
|
-
|
|
6186
|
+
// A watcher cannot prove that an edit made immediately before this call has
|
|
6187
|
+
// already reached its event queue. Git worktrees therefore take the bounded
|
|
6188
|
+
// porcelain proof on every answer. The long watcher lease remains safe for
|
|
6189
|
+
// non-Git directories, where the watcher is the only new-file signal.
|
|
6190
|
+
const gitBacked = fs.existsSync(path.join(key, ".git"));
|
|
6191
|
+
const lease = gitBacked
|
|
6192
|
+
? 0
|
|
6193
|
+
: watchQueues.get(key)?.ready
|
|
6194
|
+
? WATCHED_FRESHNESS_LEASE_MS
|
|
6195
|
+
: FRESHNESS_PROBE_TTL_MS;
|
|
6116
6196
|
const head = lease === WATCHED_FRESHNESS_LEASE_MS ? readGitHeadFast(key) : null;
|
|
6117
6197
|
const headUnchanged = head === null || (head !== undefined && head === getMeta(db, "lastIndexedHead"));
|
|
6118
6198
|
if (cached && Date.now() - cached.at < lease && headUnchanged) {
|
|
@@ -6949,6 +7029,41 @@ function queryFileSymbols(db, filePath, repoPath) {
|
|
|
6949
7029
|
};
|
|
6950
7030
|
});
|
|
6951
7031
|
}
|
|
7032
|
+
function publishStructuralSnapshot(repoPath, db) {
|
|
7033
|
+
const indexed = db
|
|
7034
|
+
.query("SELECT filePath FROM index_state ORDER BY filePath")
|
|
7035
|
+
.all();
|
|
7036
|
+
const files = [];
|
|
7037
|
+
for (const { filePath } of indexed) {
|
|
7038
|
+
const absolute = path.resolve(repoPath, filePath);
|
|
7039
|
+
try {
|
|
7040
|
+
const content = fs.readFileSync(absolute);
|
|
7041
|
+
const stat = fs.statSync(absolute);
|
|
7042
|
+
files.push({
|
|
7043
|
+
path: filePath,
|
|
7044
|
+
sizeBytes: content.byteLength,
|
|
7045
|
+
mtimeMs: stat.mtimeMs,
|
|
7046
|
+
contentFingerprint: contentFingerprint(content),
|
|
7047
|
+
symbols: queryFileSymbols(db, filePath, repoPath).map((row) => ({
|
|
7048
|
+
identity: row.identity ?? null,
|
|
7049
|
+
symbol: row.symbol ?? "",
|
|
7050
|
+
kind: row.kind ?? "unknown",
|
|
7051
|
+
line: row.line ?? 1,
|
|
7052
|
+
endLine: row.endLine ?? row.line ?? 1,
|
|
7053
|
+
...(row.signature ? { signature: row.signature } : {}),
|
|
7054
|
+
...(row.visibility ? { visibility: row.visibility } : {}),
|
|
7055
|
+
...(row.exported !== undefined ? { exported: row.exported } : {}),
|
|
7056
|
+
...(row.parent ? { parent: row.parent } : {}),
|
|
7057
|
+
evidenceQuality: row.evidenceQuality ?? "parser-grounded",
|
|
7058
|
+
})),
|
|
7059
|
+
});
|
|
7060
|
+
}
|
|
7061
|
+
catch {
|
|
7062
|
+
// Health/freshness reports missing files; snapshots never invent an entry.
|
|
7063
|
+
}
|
|
7064
|
+
}
|
|
7065
|
+
writeStructuralSnapshot(repoPath, files);
|
|
7066
|
+
}
|
|
6952
7067
|
/**
|
|
6953
7068
|
* Name-level call adjacency from the `references` table — the single SQL +
|
|
6954
7069
|
* Map-building shared by `shortest_path` (directed) and `traverse` (undirected),
|
|
@@ -7643,43 +7758,16 @@ export function calculateASTComplexity(node) {
|
|
|
7643
7758
|
traverse(node);
|
|
7644
7759
|
return score;
|
|
7645
7760
|
}
|
|
7646
|
-
const gitChurnCache = new Map();
|
|
7647
|
-
const GIT_CHURN_CACHE_LIMIT = 8;
|
|
7648
7761
|
/**
|
|
7649
7762
|
* Count commits touching requested files with the legacy path-specific Git
|
|
7650
7763
|
* semantics. Results are cached by resolved repository + HEAD, so warm review
|
|
7651
7764
|
* launches no per-file history processes while preserving existing risk scores.
|
|
7652
7765
|
*/
|
|
7653
7766
|
export function getGitChurns(filePaths, repoPath) {
|
|
7654
|
-
const resolvedRepoPath = path.resolve(repoPath);
|
|
7655
7767
|
const requested = [...new Set(filePaths.map((file) => safeReviewFile(file)))];
|
|
7656
|
-
|
|
7657
|
-
|
|
7658
|
-
|
|
7659
|
-
}
|
|
7660
|
-
catch {
|
|
7661
|
-
// An unborn repository has no history, so every requested file has zero churn.
|
|
7662
|
-
}
|
|
7663
|
-
const cacheKey = `${resolvedRepoPath}\0${head}`;
|
|
7664
|
-
let counts = gitChurnCache.get(cacheKey);
|
|
7665
|
-
if (!counts) {
|
|
7666
|
-
counts = new Map();
|
|
7667
|
-
setBoundedCache(gitChurnCache, cacheKey, counts, GIT_CHURN_CACHE_LIMIT);
|
|
7668
|
-
}
|
|
7669
|
-
if (head !== "unborn") {
|
|
7670
|
-
for (const file of requested) {
|
|
7671
|
-
if (counts.has(file))
|
|
7672
|
-
continue;
|
|
7673
|
-
try {
|
|
7674
|
-
const output = runGit(resolvedRepoPath, ["log", "--oneline", "--", file]).trim();
|
|
7675
|
-
counts.set(file, output ? output.split("\n").length : 0);
|
|
7676
|
-
}
|
|
7677
|
-
catch {
|
|
7678
|
-
counts.set(file, 0);
|
|
7679
|
-
}
|
|
7680
|
-
}
|
|
7681
|
-
}
|
|
7682
|
-
return new Map(requested.map((file) => [file, counts?.get(file) ?? 0]));
|
|
7768
|
+
const history = collectGitHistorySignals(repoPath, requested);
|
|
7769
|
+
const counts = new Map(history.churn.map((fact) => [fact.file, fact.commitCount]));
|
|
7770
|
+
return new Map(requested.map((file) => [file, counts.get(file) ?? 0]));
|
|
7683
7771
|
}
|
|
7684
7772
|
export function getGitChurn(filePath, repoPath) {
|
|
7685
7773
|
return getGitChurns([filePath], repoPath).get(safeReviewFile(filePath)) ?? 0;
|
|
@@ -9097,10 +9185,12 @@ export function createEngine() {
|
|
|
9097
9185
|
changedFiles: reviewChangedFiles(base, repoPath, options, modifiedFiles),
|
|
9098
9186
|
};
|
|
9099
9187
|
});
|
|
9100
|
-
const
|
|
9188
|
+
const historySignals = measurePerfPhaseSync("review_churn", () => collectGitHistorySignals(repoPath, changedFiles, options.historyLimits));
|
|
9189
|
+
const churnByFile = new Map(historySignals.churn.map((fact) => [fact.file, fact.commitCount]));
|
|
9101
9190
|
const changedSymbols = new Set();
|
|
9102
9191
|
const testGaps = new Set();
|
|
9103
9192
|
const symbolRisks = [];
|
|
9193
|
+
const structuralCentrality = new Map();
|
|
9104
9194
|
const finishParsing = beginPerfPhase("review_parsing");
|
|
9105
9195
|
try {
|
|
9106
9196
|
for (const filePath of changedFiles) {
|
|
@@ -9143,6 +9233,19 @@ export function createEngine() {
|
|
|
9143
9233
|
Array.from(lineSet).some((line) => line >= sym.startLine && line <= sym.endLine);
|
|
9144
9234
|
if (isChanged) {
|
|
9145
9235
|
changedSymbols.add(sym.name);
|
|
9236
|
+
const inboundReferences = db
|
|
9237
|
+
.query('SELECT COUNT(*) AS count FROM "references" WHERE calleeSymbol = ? AND (calleeFile = ? OR calleeFile IS NULL)')
|
|
9238
|
+
.get(sym.name, filePath)?.count ?? 0;
|
|
9239
|
+
const outboundReferences = db
|
|
9240
|
+
.query('SELECT COUNT(*) AS count FROM "references" WHERE callerSymbol = ? AND callerFile = ?')
|
|
9241
|
+
.get(sym.name, filePath)?.count ?? 0;
|
|
9242
|
+
structuralCentrality.set(`${filePath}\0${sym.name}`, {
|
|
9243
|
+
symbol: sym.name,
|
|
9244
|
+
file: filePath,
|
|
9245
|
+
inboundReferences,
|
|
9246
|
+
outboundReferences,
|
|
9247
|
+
totalReferences: inboundReferences + outboundReferences,
|
|
9248
|
+
});
|
|
9146
9249
|
let complexity = 1;
|
|
9147
9250
|
if (tree) {
|
|
9148
9251
|
try {
|
|
@@ -9214,11 +9317,18 @@ export function createEngine() {
|
|
|
9214
9317
|
const unmappedChangedFiles = changedFiles.filter((file) => !mappedSet.has(file));
|
|
9215
9318
|
const flowsList = Array.from(affectedFlowsSet);
|
|
9216
9319
|
const gapsList = Array.from(testGaps);
|
|
9320
|
+
const centralityFacts = [...structuralCentrality.values()].sort((a, b) => b.totalReferences - a.totalReferences ||
|
|
9321
|
+
a.file.localeCompare(b.file) ||
|
|
9322
|
+
a.symbol.localeCompare(b.symbol));
|
|
9217
9323
|
// Minimal detail caps each list so a client LLM isn't handed the full
|
|
9218
9324
|
// (potentially very large) result; the *Count fields always report the
|
|
9219
9325
|
// true totals so nothing is silently hidden.
|
|
9220
9326
|
const MINIMAL_CAP = 25;
|
|
9221
9327
|
const minimal = detailLevel === "minimal";
|
|
9328
|
+
const signalFiles = minimal ? changedFiles.slice(0, MINIMAL_CAP) : changedFiles;
|
|
9329
|
+
const signalFlows = minimal ? flowsList.slice(0, MINIMAL_CAP) : flowsList;
|
|
9330
|
+
const signalGaps = minimal ? gapsList.slice(0, MINIMAL_CAP) : gapsList;
|
|
9331
|
+
const signalCentrality = minimal ? centralityFacts.slice(0, MINIMAL_CAP) : centralityFacts;
|
|
9222
9332
|
const truncated = minimal &&
|
|
9223
9333
|
(changedFiles.length > MINIMAL_CAP ||
|
|
9224
9334
|
mappedChangedFiles.length > MINIMAL_CAP ||
|
|
@@ -9238,6 +9348,22 @@ export function createEngine() {
|
|
|
9238
9348
|
: unmappedChangedFiles,
|
|
9239
9349
|
unmappedChangedFileCount: unmappedChangedFiles.length,
|
|
9240
9350
|
riskScore,
|
|
9351
|
+
signals: {
|
|
9352
|
+
graphImpact: {
|
|
9353
|
+
affectedFlows: signalFlows,
|
|
9354
|
+
changedFiles: signalFiles,
|
|
9355
|
+
confidence: "exact-indexed-relationships",
|
|
9356
|
+
},
|
|
9357
|
+
testGaps: {
|
|
9358
|
+
facts: signalGaps,
|
|
9359
|
+
confidence: "indexed-references-and-test-paths",
|
|
9360
|
+
},
|
|
9361
|
+
structuralCentrality: {
|
|
9362
|
+
facts: signalCentrality,
|
|
9363
|
+
confidence: "exact-indexed-references",
|
|
9364
|
+
},
|
|
9365
|
+
history: historySignals,
|
|
9366
|
+
},
|
|
9241
9367
|
changedSymbols: minimal ? changedList.slice(0, MINIMAL_CAP) : changedList,
|
|
9242
9368
|
changedSymbolCount: changedList.length,
|
|
9243
9369
|
affectedFlows: minimal ? flowsList.slice(0, MINIMAL_CAP) : flowsList,
|
|
@@ -10311,6 +10437,7 @@ export function createEngine() {
|
|
|
10311
10437
|
async index(repoPath, files, clean = false, options) {
|
|
10312
10438
|
const indexed = [];
|
|
10313
10439
|
const unchanged = [];
|
|
10440
|
+
let scipReport;
|
|
10314
10441
|
const progress = createIndexProgressReporter(options?.onProgress);
|
|
10315
10442
|
progress("starting", 0, "Opening local graph database");
|
|
10316
10443
|
const db = await getOrInitDb(repoPath, {
|
|
@@ -10396,6 +10523,33 @@ export function createEngine() {
|
|
|
10396
10523
|
indexed.push(...collectRepoFiles(repoPath));
|
|
10397
10524
|
completionMessage = `${indexed.length.toLocaleString()} repository file(s) are current`;
|
|
10398
10525
|
}
|
|
10526
|
+
if (options?.scip) {
|
|
10527
|
+
const facts = readScipIndex(repoPath, options.scip);
|
|
10528
|
+
const affectedFiles = persistScipFacts(db, repoPath, facts);
|
|
10529
|
+
progress("finalizing", 0, "Refreshing SCIP identities and semantic embeddings");
|
|
10530
|
+
await indexEmbeddings(db, repoPath, progress);
|
|
10531
|
+
indexGeneration++;
|
|
10532
|
+
const bounds = { ...SCIP_DEFAULT_LIMITS, ...options.scip };
|
|
10533
|
+
scipReport = {
|
|
10534
|
+
provenance: "scip",
|
|
10535
|
+
inputPath: facts.inputPath,
|
|
10536
|
+
bytes: facts.bytes,
|
|
10537
|
+
files: facts.files,
|
|
10538
|
+
facts: facts.facts,
|
|
10539
|
+
languages: facts.languages,
|
|
10540
|
+
indexers: facts.indexers,
|
|
10541
|
+
elapsedMs: facts.elapsedMs,
|
|
10542
|
+
bounds: {
|
|
10543
|
+
maxBytes: bounds.maxBytes,
|
|
10544
|
+
maxFiles: bounds.maxFiles,
|
|
10545
|
+
maxFacts: bounds.maxFacts,
|
|
10546
|
+
timeoutMs: bounds.timeoutMs,
|
|
10547
|
+
},
|
|
10548
|
+
};
|
|
10549
|
+
for (const file of affectedFiles)
|
|
10550
|
+
if (!indexed.includes(file))
|
|
10551
|
+
indexed.push(file);
|
|
10552
|
+
}
|
|
10399
10553
|
progress("verifying", 0, "Verifying local graph health", {
|
|
10400
10554
|
phaseTotal: 1,
|
|
10401
10555
|
});
|
|
@@ -10431,6 +10585,10 @@ export function createEngine() {
|
|
|
10431
10585
|
recordFreshnessBaseline(repoPath, db);
|
|
10432
10586
|
}
|
|
10433
10587
|
noteFreshness(repoPath, "fresh");
|
|
10588
|
+
// Test engines use an in-memory graph and must not create untracked
|
|
10589
|
+
// repository artifacts that pollute Git-diff/review fixtures.
|
|
10590
|
+
if (!process.env.VITEST && process.env.NODE_ENV !== "test")
|
|
10591
|
+
publishStructuralSnapshot(repoPath, db);
|
|
10434
10592
|
}
|
|
10435
10593
|
else {
|
|
10436
10594
|
freshnessProbes.delete(path.resolve(repoPath));
|
|
@@ -10445,6 +10603,7 @@ export function createEngine() {
|
|
|
10445
10603
|
return {
|
|
10446
10604
|
indexed,
|
|
10447
10605
|
unchanged,
|
|
10606
|
+
scip: scipReport,
|
|
10448
10607
|
verification: {
|
|
10449
10608
|
status: health.status,
|
|
10450
10609
|
issueCount,
|
|
@@ -10462,6 +10621,7 @@ export function createEngine() {
|
|
|
10462
10621
|
throw new Error("knodin search: invalid testScope");
|
|
10463
10622
|
const repos = options.federate === false ? [resolvedRepoPath] : await getFederatedRepos(repoPath);
|
|
10464
10623
|
const allRepos = await Promise.all(repos.map(async (r) => ({ path: r, db: await getOrInitDb(r) })));
|
|
10624
|
+
const searchSnapshots = new Map(allRepos.map((repo) => [repo.path, getSearchMetadataSnapshot(repo.path, repo.db)]));
|
|
10465
10625
|
// typeof + Number.isFinite rejects undefined/null/NaN alike, so any non-numeric
|
|
10466
10626
|
// or unset `limit` falls back to the default instead of clamping to 0/1 or
|
|
10467
10627
|
// propagating NaN into Array.prototype.slice (which silently empties results).
|
|
@@ -10503,12 +10663,90 @@ export function createEngine() {
|
|
|
10503
10663
|
return true;
|
|
10504
10664
|
};
|
|
10505
10665
|
const formatPath = makePathFormatter(resolvedRepoPath);
|
|
10666
|
+
const retrievalMode = options.retrievalMode ?? "complete";
|
|
10667
|
+
const normalize = (value) => value.toLocaleLowerCase().replace(/[^a-z0-9]/g, "");
|
|
10668
|
+
const hydrateStructural = (matches) => {
|
|
10669
|
+
const fileToCommunityMap = matches.length > 0
|
|
10670
|
+
? getSearchCommunityLookup(allRepos, resolvedRepoPath)
|
|
10671
|
+
: new Map();
|
|
10672
|
+
return matches.map(({ repo, row }) => {
|
|
10673
|
+
let source = "";
|
|
10674
|
+
if (options.includeSource !== false)
|
|
10675
|
+
source = getSourceRange(repo.path, row.filePath, row.startLine, row.endLine, 500);
|
|
10676
|
+
const callersStmt = repo.db.query(`
|
|
10677
|
+
SELECT DISTINCT callerSymbol FROM "references"
|
|
10678
|
+
WHERE calleeSymbol = ? AND callerSymbol IS NOT NULL AND callerSymbol != ?
|
|
10679
|
+
AND (calleeFile = ? OR calleeFile IS NULL)
|
|
10680
|
+
`);
|
|
10681
|
+
const directCallers = callersStmt
|
|
10682
|
+
.all(row.name, row.name, row.filePath)
|
|
10683
|
+
.map((caller) => caller.callerSymbol)
|
|
10684
|
+
.sort((a, b) => a.localeCompare(b));
|
|
10685
|
+
callersStmt.finalize();
|
|
10686
|
+
const formattedFile = formatPath(repo.path, row.filePath);
|
|
10687
|
+
return {
|
|
10688
|
+
identity: symbolIdentity(repo.path, row),
|
|
10689
|
+
symbol: row.name,
|
|
10690
|
+
kind: row.kind,
|
|
10691
|
+
filePath: formattedFile,
|
|
10692
|
+
similarity: 1,
|
|
10693
|
+
rrfScore: 1,
|
|
10694
|
+
source,
|
|
10695
|
+
belongsToCommunity: fileToCommunityMap.get(formattedFile) || "core-module",
|
|
10696
|
+
directCallers,
|
|
10697
|
+
};
|
|
10698
|
+
});
|
|
10699
|
+
};
|
|
10700
|
+
const expandBounded = (seeds) => {
|
|
10701
|
+
const expansionWorkLimit = Math.min(200, Math.max(candidateLimit * 4, 20));
|
|
10702
|
+
const selected = seeds.slice(0, expansionWorkLimit);
|
|
10703
|
+
const seen = new Set(selected.map(({ repo, row }) => `${repo.path}:${row.id}`));
|
|
10704
|
+
let frontier = [...seeds];
|
|
10705
|
+
for (let depth = 0; depth < 2 && frontier.length > 0 && selected.length < expansionWorkLimit; depth++) {
|
|
10706
|
+
const next = [];
|
|
10707
|
+
for (const current of frontier) {
|
|
10708
|
+
const refs = current.repo.db
|
|
10709
|
+
.query(`SELECT callerSymbol, callerFile, calleeSymbol, calleeFile FROM "references"
|
|
10710
|
+
WHERE (callerSymbol = ? AND callerFile = ?)
|
|
10711
|
+
OR (calleeSymbol = ? AND calleeFile = ?)
|
|
10712
|
+
ORDER BY callerFile, callerSymbol, calleeFile, calleeSymbol
|
|
10713
|
+
LIMIT 200`)
|
|
10714
|
+
.all(current.row.name, current.row.filePath, current.row.name, current.row.filePath);
|
|
10715
|
+
const snapshot = searchSnapshots.get(current.repo.path);
|
|
10716
|
+
for (const ref of refs) {
|
|
10717
|
+
const fromCaller = ref.callerSymbol === current.row.name && ref.callerFile === current.row.filePath;
|
|
10718
|
+
const neighborName = fromCaller ? ref.calleeSymbol : ref.callerSymbol;
|
|
10719
|
+
const neighborFile = fromCaller ? ref.calleeFile : ref.callerFile;
|
|
10720
|
+
if (!neighborName || !neighborFile)
|
|
10721
|
+
continue;
|
|
10722
|
+
for (const row of snapshot.rows) {
|
|
10723
|
+
if (row.name !== neighborName || !accepts(row))
|
|
10724
|
+
continue;
|
|
10725
|
+
if (neighborFile && row.filePath !== neighborFile)
|
|
10726
|
+
continue;
|
|
10727
|
+
const key = `${current.repo.path}:${row.id}`;
|
|
10728
|
+
if (seen.has(key))
|
|
10729
|
+
continue;
|
|
10730
|
+
if (seen.size >= expansionWorkLimit)
|
|
10731
|
+
break;
|
|
10732
|
+
seen.add(key);
|
|
10733
|
+
next.push({ repo: current.repo, row });
|
|
10734
|
+
}
|
|
10735
|
+
}
|
|
10736
|
+
}
|
|
10737
|
+
next.sort((a, b) => a.repo.path.localeCompare(b.repo.path) ||
|
|
10738
|
+
a.row.filePath.localeCompare(b.row.filePath) ||
|
|
10739
|
+
a.row.startLine - b.row.startLine);
|
|
10740
|
+
selected.push(...next.slice(0, expansionWorkLimit - selected.length));
|
|
10741
|
+
frontier = next;
|
|
10742
|
+
}
|
|
10743
|
+
return selected;
|
|
10744
|
+
};
|
|
10506
10745
|
if (options.structural) {
|
|
10507
|
-
const normalize = (value) => value.toLocaleLowerCase().replace(/[^a-z0-9]/g, "");
|
|
10508
10746
|
const needle = normalize(query);
|
|
10509
10747
|
const matches = allRepos
|
|
10510
|
-
.flatMap((repo) =>
|
|
10511
|
-
.
|
|
10748
|
+
.flatMap((repo) => searchSnapshots.get(repo.path).rows
|
|
10749
|
+
.filter((row) => accepts(row) && normalize(row.name).includes(needle))
|
|
10512
10750
|
.map((row) => ({ repo, row })))
|
|
10513
10751
|
.sort((left, right) => left.row.name.localeCompare(right.row.name) ||
|
|
10514
10752
|
left.row.filePath.localeCompare(right.row.filePath) ||
|
|
@@ -10532,8 +10770,140 @@ export function createEngine() {
|
|
|
10532
10770
|
offset,
|
|
10533
10771
|
limit: maxResults,
|
|
10534
10772
|
hasMore: offset + maxResults < matches.length,
|
|
10773
|
+
retrieval: { route: ["exact-name"], embeddingsUsed: false },
|
|
10535
10774
|
};
|
|
10536
10775
|
}
|
|
10776
|
+
if (retrievalMode !== "vector-only") {
|
|
10777
|
+
const trimmedQuery = query.trim().replaceAll("\\", "/");
|
|
10778
|
+
const normalizedQuery = normalize(trimmedQuery);
|
|
10779
|
+
const identityQuery = trimmedQuery.startsWith("sym_") ? trimmedQuery : undefined;
|
|
10780
|
+
const pathQuery = trimmedQuery.includes("/") || /\.[A-Za-z0-9]+$/.test(trimmedQuery);
|
|
10781
|
+
const seedMatches = allRepos
|
|
10782
|
+
.flatMap((repo) => searchSnapshots.get(repo.path).rows
|
|
10783
|
+
.filter((row) => {
|
|
10784
|
+
if (!accepts(row))
|
|
10785
|
+
return false;
|
|
10786
|
+
if (identityQuery)
|
|
10787
|
+
return symbolIdentityMatches(symbolIdentity(repo.path, row), identityQuery);
|
|
10788
|
+
if (pathQuery) {
|
|
10789
|
+
const candidate = row.filePath.replaceAll("\\", "/");
|
|
10790
|
+
return candidate === trimmedQuery || candidate.endsWith(`/${trimmedQuery}`);
|
|
10791
|
+
}
|
|
10792
|
+
return normalizedQuery.length > 0 && normalize(row.name) === normalizedQuery;
|
|
10793
|
+
})
|
|
10794
|
+
.map((row) => ({ repo, row })))
|
|
10795
|
+
.sort((a, b) => a.repo.path.localeCompare(b.repo.path) ||
|
|
10796
|
+
a.row.filePath.localeCompare(b.row.filePath) ||
|
|
10797
|
+
a.row.startLine - b.row.startLine);
|
|
10798
|
+
if (seedMatches.length > 0) {
|
|
10799
|
+
const selected = retrievalMode === "complete" || retrievalMode === "bounded-graph"
|
|
10800
|
+
? expandBounded(seedMatches)
|
|
10801
|
+
: seedMatches;
|
|
10802
|
+
const pageMatches = selected.slice(offset, offset + maxResults);
|
|
10803
|
+
return {
|
|
10804
|
+
results: hydrateStructural(pageMatches),
|
|
10805
|
+
total: selected.length,
|
|
10806
|
+
offset,
|
|
10807
|
+
limit: maxResults,
|
|
10808
|
+
hasMore: offset + maxResults < selected.length,
|
|
10809
|
+
retrieval: {
|
|
10810
|
+
route: retrievalMode === "complete" || retrievalMode === "bounded-graph"
|
|
10811
|
+
? [
|
|
10812
|
+
identityQuery ? "stable-identity" : pathQuery ? "exact-path" : "exact-name",
|
|
10813
|
+
"bounded-graph",
|
|
10814
|
+
]
|
|
10815
|
+
: [identityQuery ? "stable-identity" : pathQuery ? "exact-path" : "exact-name"],
|
|
10816
|
+
embeddingsUsed: false,
|
|
10817
|
+
},
|
|
10818
|
+
};
|
|
10819
|
+
}
|
|
10820
|
+
}
|
|
10821
|
+
const lexicalRanks = new Map();
|
|
10822
|
+
if (retrievalMode !== "vector-only") {
|
|
10823
|
+
for (const repo of allRepos) {
|
|
10824
|
+
const snapshot = searchSnapshots.get(repo.path);
|
|
10825
|
+
const allowedIds = new Set(snapshot.rows.filter(accepts).map((row) => row.id));
|
|
10826
|
+
const tokens = query
|
|
10827
|
+
.replace(/[^\w\s]/g, " ")
|
|
10828
|
+
.trim()
|
|
10829
|
+
.split(/\s+/)
|
|
10830
|
+
.filter((word) => word.length >= 4)
|
|
10831
|
+
.map((word) => `"${word}"`);
|
|
10832
|
+
const runFts = (separator) => {
|
|
10833
|
+
if (tokens.length === 0)
|
|
10834
|
+
return [];
|
|
10835
|
+
try {
|
|
10836
|
+
const statement = repo.db.query(`
|
|
10837
|
+
SELECT symbolId FROM symbols_fts
|
|
10838
|
+
WHERE symbols_fts MATCH ?
|
|
10839
|
+
ORDER BY bm25(symbols_fts), symbolId
|
|
10840
|
+
LIMIT 100
|
|
10841
|
+
`);
|
|
10842
|
+
const ids = statement
|
|
10843
|
+
.all(tokens.join(separator))
|
|
10844
|
+
.map(({ symbolId }) => symbolId)
|
|
10845
|
+
.filter((id) => allowedIds.has(id));
|
|
10846
|
+
statement.finalize();
|
|
10847
|
+
return ids;
|
|
10848
|
+
}
|
|
10849
|
+
catch (error) {
|
|
10850
|
+
console.error(`FTS query failed for "${query}":`, error);
|
|
10851
|
+
return [];
|
|
10852
|
+
}
|
|
10853
|
+
};
|
|
10854
|
+
const and = measurePerfPhaseSync("fts", () => runFts(" "));
|
|
10855
|
+
const or = tokens.length > 1 ? measurePerfPhaseSync("fts", () => runFts(" OR ")) : and;
|
|
10856
|
+
const lexicalSeeds = or
|
|
10857
|
+
.map((id) => snapshot.rowById.get(id))
|
|
10858
|
+
.filter((row) => row !== undefined)
|
|
10859
|
+
.map((row) => ({ repo, row }));
|
|
10860
|
+
const structural = expandBounded(lexicalSeeds)
|
|
10861
|
+
.filter(({ repo: matchRepo }) => matchRepo.path === repo.path)
|
|
10862
|
+
.map(({ row }) => row.id);
|
|
10863
|
+
const sufficientSeeds = and
|
|
10864
|
+
.map((id) => snapshot.rowById.get(id))
|
|
10865
|
+
.filter((row) => row !== undefined)
|
|
10866
|
+
.map((row) => ({ repo, row }));
|
|
10867
|
+
const sufficient = expandBounded(sufficientSeeds)
|
|
10868
|
+
.filter(({ repo: matchRepo }) => matchRepo.path === repo.path)
|
|
10869
|
+
.map(({ row }) => row.id);
|
|
10870
|
+
lexicalRanks.set(repo.path, { and, or, structural, sufficient });
|
|
10871
|
+
}
|
|
10872
|
+
const structurallySufficient = retrievalMode === "complete" &&
|
|
10873
|
+
!ann.annEnabled() &&
|
|
10874
|
+
[...lexicalRanks.values()].some((ranks) => ranks.and.length > 0);
|
|
10875
|
+
if (retrievalMode === "lexical-only" ||
|
|
10876
|
+
retrievalMode === "bounded-graph" ||
|
|
10877
|
+
structurallySufficient) {
|
|
10878
|
+
const matches = allRepos.flatMap((repo) => {
|
|
10879
|
+
const snapshot = searchSnapshots.get(repo.path);
|
|
10880
|
+
const ranks = lexicalRanks.get(repo.path);
|
|
10881
|
+
const ids = structurallySufficient
|
|
10882
|
+
? ranks?.sufficient
|
|
10883
|
+
: retrievalMode === "bounded-graph"
|
|
10884
|
+
? ranks?.structural
|
|
10885
|
+
: ranks?.or;
|
|
10886
|
+
return (ids ?? [])
|
|
10887
|
+
.map((id) => snapshot.rowById.get(id))
|
|
10888
|
+
.filter((row) => row !== undefined)
|
|
10889
|
+
.map((row) => ({ repo, row }));
|
|
10890
|
+
});
|
|
10891
|
+
const pageMatches = matches.slice(offset, offset + maxResults);
|
|
10892
|
+
return {
|
|
10893
|
+
results: hydrateStructural(pageMatches),
|
|
10894
|
+
total: matches.length,
|
|
10895
|
+
offset,
|
|
10896
|
+
limit: maxResults,
|
|
10897
|
+
hasMore: offset + maxResults < matches.length,
|
|
10898
|
+
retrieval: {
|
|
10899
|
+
route: retrievalMode === "bounded-graph" || structurallySufficient
|
|
10900
|
+
? ["lexical", "bounded-graph"]
|
|
10901
|
+
: ["lexical"],
|
|
10902
|
+
embeddingsUsed: false,
|
|
10903
|
+
},
|
|
10904
|
+
};
|
|
10905
|
+
}
|
|
10906
|
+
}
|
|
10537
10907
|
// 1. Generate query embedding once
|
|
10538
10908
|
const queryVec = await generateMeasuredEmbedding(query, true);
|
|
10539
10909
|
const rankedCandidates = [];
|
|
@@ -10542,7 +10912,7 @@ export function createEngine() {
|
|
|
10542
10912
|
const db = repo.db;
|
|
10543
10913
|
// 2. Reuse decoded generation-scoped metadata. Cheap metadata-only
|
|
10544
10914
|
// filters run before any exact vector score is computed.
|
|
10545
|
-
const snapshot =
|
|
10915
|
+
const snapshot = searchSnapshots.get(repo.path);
|
|
10546
10916
|
const allRows = snapshot.rows.filter(accepts);
|
|
10547
10917
|
const allowedIds = new Set(allRows.map((row) => row.id));
|
|
10548
10918
|
totalMatches += allowedIds.size;
|
|
@@ -10587,39 +10957,10 @@ export function createEngine() {
|
|
|
10587
10957
|
const sortedSemantics = semanticMatches.sort((a, b) => b.score - a.score);
|
|
10588
10958
|
const semanticScoresAreTied = sortedSemantics.length > 1 &&
|
|
10589
10959
|
sortedSemantics.every((item) => item.score === sortedSemantics[0]?.score);
|
|
10590
|
-
// 3.
|
|
10591
|
-
const
|
|
10592
|
-
|
|
10593
|
-
|
|
10594
|
-
const cleanQuery = query.replace(/[^\w\s]/g, " ").trim();
|
|
10595
|
-
if (cleanQuery) {
|
|
10596
|
-
// Quote each token so FTS5 treats bare AND/OR/NOT/NEAR as literal search
|
|
10597
|
-
// words instead of query-syntax operators (which would otherwise throw
|
|
10598
|
-
// a MATCH syntax error for a query like "NOT authenticated").
|
|
10599
|
-
const ftsQuery = cleanQuery
|
|
10600
|
-
.split(/\s+/)
|
|
10601
|
-
.filter((word) => word.length >= 4)
|
|
10602
|
-
.map((word) => `"${word}"`)
|
|
10603
|
-
// With a useful semantic ordering, retain the precise all-token
|
|
10604
|
-
// FTS signal. When a deterministic/fallback embedder ties every
|
|
10605
|
-
// candidate, permit partial lexical matches so connective words
|
|
10606
|
-
// do not reduce ranking to arbitrary insertion order.
|
|
10607
|
-
.join(semanticScoresAreTied ? " OR " : " ");
|
|
10608
|
-
const ftsStmt = db.query(`
|
|
10609
|
-
SELECT symbolId FROM symbols_fts WHERE symbols_fts MATCH ? ORDER BY bm25(symbols_fts) LIMIT 100
|
|
10610
|
-
`);
|
|
10611
|
-
matches = ftsStmt
|
|
10612
|
-
.all(ftsQuery)
|
|
10613
|
-
.map((m) => m.symbolId)
|
|
10614
|
-
.filter((id) => allowedIds.has(id));
|
|
10615
|
-
ftsStmt.finalize();
|
|
10616
|
-
}
|
|
10617
|
-
}
|
|
10618
|
-
catch (err) {
|
|
10619
|
-
console.error(`FTS query failed for "${query}":`, err);
|
|
10620
|
-
}
|
|
10621
|
-
return matches;
|
|
10622
|
-
});
|
|
10960
|
+
// 3. Reuse the lexical and bounded-graph ranks computed before query inference.
|
|
10961
|
+
const lexical = lexicalRanks.get(repo.path);
|
|
10962
|
+
const baseLexical = semanticScoresAreTied ? lexical?.or : lexical?.and;
|
|
10963
|
+
const ftsMatches = [...new Set(baseLexical ?? [])].filter((id) => allowedIds.has(id));
|
|
10623
10964
|
// Rank mappings for Reciprocal Rank Fusion (RRF)
|
|
10624
10965
|
const semanticRankMap = new Map();
|
|
10625
10966
|
const semanticById = new Map();
|
|
@@ -10708,6 +11049,12 @@ export function createEngine() {
|
|
|
10708
11049
|
offset,
|
|
10709
11050
|
limit: maxResults,
|
|
10710
11051
|
hasMore: offset + maxResults < total,
|
|
11052
|
+
retrieval: {
|
|
11053
|
+
route: retrievalMode === "vector-only"
|
|
11054
|
+
? ["embeddings"]
|
|
11055
|
+
: ["lexical", "bounded-graph", "embeddings"],
|
|
11056
|
+
embeddingsUsed: true,
|
|
11057
|
+
},
|
|
10711
11058
|
};
|
|
10712
11059
|
},
|
|
10713
11060
|
async query(pattern, target, repoPath, to, limit, depth, detailLevel, selector = {}, impactOptions = {}, options = {}) {
|
|
@@ -13018,7 +13365,7 @@ export function createEngine() {
|
|
|
13018
13365
|
statusCache.clear();
|
|
13019
13366
|
flowsCache.clear();
|
|
13020
13367
|
fileToFlowLookupCache.clear();
|
|
13021
|
-
|
|
13368
|
+
clearGitHistorySignalCache();
|
|
13022
13369
|
annIndexCache.clear();
|
|
13023
13370
|
searchMetadataCache.clear();
|
|
13024
13371
|
searchCommunityLookupCache.clear();
|