knodin 0.5.1 → 0.7.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +190 -535
- package/dist/bin/cli.js +173 -17
- package/dist/bin/launcher.js +11 -0
- package/dist/src/agent-integration.js +4 -16
- package/dist/src/artifact-refresh.js +1 -1
- package/dist/src/cli-args.js +8 -1
- package/dist/src/cli-model.js +35 -1
- package/dist/src/codeflow-replay.js +80 -0
- package/dist/src/competitive-cold-mcp.js +40 -0
- package/dist/src/competitive-manifest.js +106 -25
- package/dist/src/competitive-runner.js +37 -3
- package/dist/src/competitive-sandbox.js +4 -4
- package/dist/src/context-export.js +1 -1
- package/dist/src/diagnostics.js +449 -0
- package/dist/src/docs-sections.js +5 -5
- package/dist/src/engine/ann-hnsw.js +5 -5
- package/dist/src/engine/embeddings.js +4 -4
- package/dist/src/engine/git-history.js +289 -0
- package/dist/src/engine/index.js +450 -103
- package/dist/src/engine/prune.js +2 -2
- package/dist/src/engine/scip-import.js +408 -0
- package/dist/src/engine/symbol-delete.js +1 -1
- package/dist/src/execution-profile.js +203 -0
- package/dist/src/failure-diagnosis.js +69 -10
- package/dist/src/hook-manager-integration.js +156 -0
- package/dist/src/index-activity.js +1 -1
- package/dist/src/init-progress-worker.js +1 -1
- package/dist/src/init.js +361 -92
- package/dist/src/lifecycle-health.js +51 -14
- package/dist/src/lsp-readonly.js +1 -1
- package/dist/src/output-compression.js +1 -1
- package/dist/src/output-telemetry.js +8 -4
- package/dist/src/pr-triage.js +2 -2
- package/dist/src/progressive-evidence.js +473 -0
- package/dist/src/pure-compression-cli.js +101 -0
- package/dist/src/release-preflight.js +510 -0
- package/dist/src/repair-progress-worker.js +1 -1
- package/dist/src/repository-init-process.js +1 -1
- package/dist/src/repository-management.js +145 -3
- package/dist/src/response-budget.js +11 -1
- package/dist/src/server.js +23 -3
- package/dist/src/structural-fast-path.js +303 -0
- package/dist/src/structural-snapshot.js +33 -0
- package/dist/src/system-config.js +7 -7
- package/dist/src/tools/knodin-tools.js +106 -15
- package/dist/src/update-ceremony.js +158 -0
- package/dist/src/update-policy.js +4 -4
- package/dist/src/wait-for-fresh.js +2 -2
- package/dist/src/worktree-lifecycle.js +2 -2
- package/docs/CLI.md +22 -0
- package/docs/COMMAND-OUTPUT-COMPRESSION.md +32 -16
- package/docs/CONTAINED-EXECUTION.md +77 -0
- package/docs/DIAGNOSTICS.md +45 -0
- package/docs/DOCTOR-AND-UPDATES.md +7 -4
- package/docs/GIT-HISTORY-REVIEW.md +39 -0
- package/docs/INSTALLATION.md +1 -1
- package/docs/MCP.md +15 -0
- package/docs/PROGRESSIVE-EVIDENCE.md +37 -0
- package/docs/REPOSITORIES-AND-WORKTREES.md +31 -1
- package/docs/SCIP-IMPORT.md +57 -0
- package/docs/SIGNED-UPDATES.md +5 -0
- package/docs/SYSTEMS-AND-RELATIONSHIPS.md +2 -2
- package/docs/TELEMETRY.md +6 -2
- package/docs/releases/0.6.0.md +18 -0
- package/docs/releases/0.7.0.md +24 -0
- package/docs/releases/0.7.1.md +21 -0
- package/docs/releases/0.7.2.md +21 -0
- package/docs/releases/0.7.3.md +23 -0
- package/package.json +34 -3
- package/dist/src/tools/reckon-tools.js +0 -5
package/dist/src/engine/index.js
CHANGED
|
@@ -20,12 +20,15 @@ import chokidar from "chokidar";
|
|
|
20
20
|
import Parser from "web-tree-sitter";
|
|
21
21
|
import { readIndexActivity } from "../index-activity.js";
|
|
22
22
|
import { runReadOnlyLspQuery } from "../lsp-readonly.js";
|
|
23
|
+
import { contentFingerprint, writeStructuralSnapshot, } from "../structural-snapshot.js";
|
|
23
24
|
import { KNODIN_VERSION } from "../version.js";
|
|
24
25
|
import * as ann from "./ann-hnsw.js";
|
|
25
26
|
import { computeSimilarity, generateEmbedding, generateEmbeddings, } from "./embeddings.js";
|
|
26
27
|
import { walkRepoFiles } from "./file-walker.js";
|
|
28
|
+
import { clearGitHistorySignalCache, collectGitHistorySignals, } from "./git-history.js";
|
|
27
29
|
import { beginPerfPhase, measurePerfPhase, measurePerfPhaseSync } from "./perf.js";
|
|
28
30
|
import { isIndexablePath, makeWatchIgnorePredicate } from "./prune.js";
|
|
31
|
+
import { readScipIndex, SCIP_DEFAULT_LIMITS, } from "./scip-import.js";
|
|
29
32
|
import { isIndexableSourcePath } from "./source-policy.js";
|
|
30
33
|
import { Database } from "./sqlite.js";
|
|
31
34
|
import { deleteAllSymbols, deleteSymbolsForFile, deleteSymbolsMatchingPath, ORPHANED_EMBEDDING_PREDICATE, purgeOrphanEmbeddings, } from "./symbol-delete.js";
|
|
@@ -796,7 +799,7 @@ const FLOW_MAX_ENTRIES = 200;
|
|
|
796
799
|
const flowsCache = new Map();
|
|
797
800
|
const fileToFlowLookupCache = new Map();
|
|
798
801
|
const FILE_TO_FLOW_LOOKUP_CACHE_LIMIT = 8;
|
|
799
|
-
// R18: opt-in ANN (
|
|
802
|
+
// R18: opt-in ANN (KNODIN_ANN=1). An in-memory HNSW index over a repo's
|
|
800
803
|
// embeddings, built lazily on first ANN query and cached per repo + invalidated
|
|
801
804
|
// by indexGeneration exactly like mapCache/flowsCache — so a reindex rebuilds it
|
|
802
805
|
// and close() drops it. Holds NO stored state (no schema bump, nothing to
|
|
@@ -3900,6 +3903,73 @@ async function indexLsifDump(absolutePath, _relativePath, repoPath, db) {
|
|
|
3900
3903
|
console.error("Error indexing LSIF:", error);
|
|
3901
3904
|
}
|
|
3902
3905
|
}
|
|
3906
|
+
/** Replace only the optional SCIP tier, preserving native and LSIF facts verbatim. */
|
|
3907
|
+
function persistScipFacts(db, repoPath, facts) {
|
|
3908
|
+
const affectedFiles = [...new Set(facts.symbols.map((symbol) => symbol.filePath))].sort();
|
|
3909
|
+
const insertSymbol = db.prepare(`
|
|
3910
|
+
INSERT INTO symbols (name, kind, filePath, startLine, endLine, startCol, endCol, summary)
|
|
3911
|
+
VALUES (?, ?, ?, ?, ?, ?, ?, ?)
|
|
3912
|
+
`);
|
|
3913
|
+
const insertReference = db.prepare(`
|
|
3914
|
+
INSERT INTO "references" (callerSymbol, callerFile, calleeSymbol, calleeFile, line, column, kind, confidence)
|
|
3915
|
+
VALUES (?, ?, ?, ?, ?, ?, ?, 1.0)
|
|
3916
|
+
`);
|
|
3917
|
+
const insertDependency = db.prepare(`
|
|
3918
|
+
INSERT INTO dependencies (fromFile, toFile, kind, confidence, sourceEvidence)
|
|
3919
|
+
VALUES (?, ?, 'scip_import', 1.0, ?)
|
|
3920
|
+
`);
|
|
3921
|
+
db.run("BEGIN TRANSACTION;");
|
|
3922
|
+
try {
|
|
3923
|
+
db.run("DELETE FROM symbols WHERE kind LIKE 'SCIP_%'");
|
|
3924
|
+
db.run("DELETE FROM \"references\" WHERE kind LIKE 'scip_%'");
|
|
3925
|
+
db.run("DELETE FROM dependencies WHERE kind = 'scip_import'");
|
|
3926
|
+
for (const symbol of facts.symbols) {
|
|
3927
|
+
insertSymbol.run(symbol.name, symbol.kind, symbol.filePath, symbol.startLine, symbol.endLine, symbol.startCol, symbol.endCol, JSON.stringify({ provenance: "scip", symbol: symbol.symbol }));
|
|
3928
|
+
}
|
|
3929
|
+
const dependencies = new Set();
|
|
3930
|
+
for (const reference of facts.references) {
|
|
3931
|
+
insertReference.run(reference.callerSymbol, reference.callerFile, reference.calleeSymbol, reference.calleeFile, reference.line, reference.column, reference.kind);
|
|
3932
|
+
if (reference.kind === "scip_reference" && reference.callerFile !== reference.calleeFile) {
|
|
3933
|
+
const key = `${reference.callerFile}\0${reference.calleeFile}`;
|
|
3934
|
+
if (!dependencies.has(key)) {
|
|
3935
|
+
dependencies.add(key);
|
|
3936
|
+
insertDependency.run(reference.callerFile, reference.calleeFile, JSON.stringify({ provenance: "scip", input: facts.inputPath }));
|
|
3937
|
+
}
|
|
3938
|
+
}
|
|
3939
|
+
}
|
|
3940
|
+
db.run("COMMIT;");
|
|
3941
|
+
}
|
|
3942
|
+
catch (error) {
|
|
3943
|
+
db.run("ROLLBACK;");
|
|
3944
|
+
throw error;
|
|
3945
|
+
}
|
|
3946
|
+
finally {
|
|
3947
|
+
insertSymbol.finalize();
|
|
3948
|
+
insertReference.finalize();
|
|
3949
|
+
insertDependency.finalize();
|
|
3950
|
+
}
|
|
3951
|
+
persistSymbolIdentities(db, repoPath, affectedFiles);
|
|
3952
|
+
return affectedFiles;
|
|
3953
|
+
}
|
|
3954
|
+
/** Any source refresh makes the previous compiler snapshot stale as a unit. */
|
|
3955
|
+
function invalidateScipFacts(db) {
|
|
3956
|
+
const present = db
|
|
3957
|
+
.query("SELECT 1 AS present FROM symbols WHERE kind LIKE 'SCIP_%' LIMIT 1")
|
|
3958
|
+
.get();
|
|
3959
|
+
if (!present)
|
|
3960
|
+
return;
|
|
3961
|
+
db.run("BEGIN TRANSACTION;");
|
|
3962
|
+
try {
|
|
3963
|
+
db.run("DELETE FROM symbols WHERE kind LIKE 'SCIP_%'");
|
|
3964
|
+
db.run("DELETE FROM \"references\" WHERE kind LIKE 'scip_%'");
|
|
3965
|
+
db.run("DELETE FROM dependencies WHERE kind = 'scip_import'");
|
|
3966
|
+
db.run("COMMIT;");
|
|
3967
|
+
}
|
|
3968
|
+
catch (error) {
|
|
3969
|
+
db.run("ROLLBACK;");
|
|
3970
|
+
throw error;
|
|
3971
|
+
}
|
|
3972
|
+
}
|
|
3903
3973
|
async function indexTerraformFile(content, relativePath, _repoPath, db) {
|
|
3904
3974
|
const symbols = [];
|
|
3905
3975
|
const references = [];
|
|
@@ -4814,7 +4884,7 @@ export function parseSimpleNameFromUniqueId(uniqueId) {
|
|
|
4814
4884
|
export async function indexDbtManifestFile(content, _absolutePath, relativePath, repoPath, db) {
|
|
4815
4885
|
try {
|
|
4816
4886
|
const manifest = JSON.parse(content);
|
|
4817
|
-
const dbtDir = path.join(repoPath, ".
|
|
4887
|
+
const dbtDir = path.join(repoPath, ".knodin/dbt");
|
|
4818
4888
|
await fs.promises.mkdir(dbtDir, { recursive: true });
|
|
4819
4889
|
const assets = [];
|
|
4820
4890
|
// 1. Process nodes (models, seeds, tests)
|
|
@@ -4951,9 +5021,9 @@ export async function indexDbtManifestFile(content, _absolutePath, relativePath,
|
|
|
4951
5021
|
db.run("BEGIN TRANSACTION;");
|
|
4952
5022
|
try {
|
|
4953
5023
|
// Clear out any previous records associated with this dbt manifest or virtual paths
|
|
4954
|
-
deleteSymbolsMatchingPath(db, ".
|
|
4955
|
-
db.run("DELETE FROM \"references\" WHERE callerFile LIKE '.
|
|
4956
|
-
db.run("DELETE FROM dependencies WHERE fromFile LIKE '.
|
|
5024
|
+
deleteSymbolsMatchingPath(db, ".knodin/dbt/%");
|
|
5025
|
+
db.run("DELETE FROM \"references\" WHERE callerFile LIKE '.knodin/dbt/%'");
|
|
5026
|
+
db.run("DELETE FROM dependencies WHERE fromFile LIKE '.knodin/dbt/%'");
|
|
4957
5027
|
deleteSymbolsForFile(db, relativePath);
|
|
4958
5028
|
db.run('DELETE FROM "references" WHERE callerFile = ?', [relativePath]);
|
|
4959
5029
|
db.run("DELETE FROM dependencies WHERE fromFile = ?", [relativePath]);
|
|
@@ -4971,7 +5041,7 @@ export async function indexDbtManifestFile(content, _absolutePath, relativePath,
|
|
|
4971
5041
|
`);
|
|
4972
5042
|
for (const asset of assets) {
|
|
4973
5043
|
const safeId = asset.unique_id.replace(/[/\\]/g, "_");
|
|
4974
|
-
const virtualFilePath = `.
|
|
5044
|
+
const virtualFilePath = `.knodin/dbt/${safeId}.sql`;
|
|
4975
5045
|
const absVirtualPath = path.join(repoPath, virtualFilePath);
|
|
4976
5046
|
let code = asset.sqlCode;
|
|
4977
5047
|
if (!code) {
|
|
@@ -4990,7 +5060,7 @@ export async function indexDbtManifestFile(content, _absolutePath, relativePath,
|
|
|
4990
5060
|
for (const depId of asset.depends_on_nodes) {
|
|
4991
5061
|
const depAsset = assetMap.get(depId);
|
|
4992
5062
|
const safeDepId = depId.replace(/[/\\]/g, "_");
|
|
4993
|
-
const depVirtualFilePath = `.
|
|
5063
|
+
const depVirtualFilePath = `.knodin/dbt/${safeDepId}.sql`;
|
|
4994
5064
|
insertDep.run(virtualFilePath, depVirtualFilePath, "dbt-lineage");
|
|
4995
5065
|
const depSimpleName = depAsset ? depAsset.name : parseSimpleNameFromUniqueId(depId);
|
|
4996
5066
|
insertRef.run(asset.name, virtualFilePath, depSimpleName, depVirtualFilePath, 1, 1);
|
|
@@ -5014,6 +5084,7 @@ export async function indexDbtManifestFile(content, _absolutePath, relativePath,
|
|
|
5014
5084
|
/** Index a single file's symbols and references into the SQLite database. */
|
|
5015
5085
|
async function indexFile(absolutePath, relativePath, repoPath, db) {
|
|
5016
5086
|
try {
|
|
5087
|
+
invalidateScipFacts(db);
|
|
5017
5088
|
const baseName = path.basename(absolutePath).toLowerCase();
|
|
5018
5089
|
if (baseName === "manifest.json") {
|
|
5019
5090
|
const content = fs.readFileSync(absolutePath, "utf-8");
|
|
@@ -5247,7 +5318,7 @@ const MAX_EMBEDDING_BATCH_SIZE = 32;
|
|
|
5247
5318
|
// every symbol without turning source preparation into an unbounded memory sink.
|
|
5248
5319
|
const EMBEDDING_SOURCE_CACHE_MAX_BYTES = 64 * 1024 * 1024;
|
|
5249
5320
|
function embeddingBatchSize() {
|
|
5250
|
-
const raw = process.env.
|
|
5321
|
+
const raw = process.env.KNODIN_EMBEDDING_BATCH_SIZE;
|
|
5251
5322
|
if (raw === undefined)
|
|
5252
5323
|
return DEFAULT_EMBEDDING_BATCH_SIZE;
|
|
5253
5324
|
const parsed = Number(raw);
|
|
@@ -5829,11 +5900,11 @@ const WATCHED_FRESHNESS_LEASE_MS = 60_000;
|
|
|
5829
5900
|
const FRESHNESS_FALLBACK_FILE_BOUND = 50000;
|
|
5830
5901
|
/**
|
|
5831
5902
|
* The non-git fallback's file bound, overridable with
|
|
5832
|
-
* `
|
|
5903
|
+
* `KNODIN_FRESHNESS_FILE_BOUND` for repos where even a stat sweep of the
|
|
5833
5904
|
* indexed set is too much to pay per query.
|
|
5834
5905
|
*/
|
|
5835
5906
|
function freshnessFallbackBound() {
|
|
5836
|
-
const raw = process.env.
|
|
5907
|
+
const raw = process.env.KNODIN_FRESHNESS_FILE_BOUND;
|
|
5837
5908
|
if (raw !== undefined) {
|
|
5838
5909
|
const parsed = Number.parseInt(raw, 10);
|
|
5839
5910
|
if (Number.isFinite(parsed) && parsed >= 0)
|
|
@@ -6112,7 +6183,16 @@ function noteFreshness(repoPath, staleness) {
|
|
|
6112
6183
|
async function ensureIndexFresh(repoPath, db) {
|
|
6113
6184
|
const key = path.resolve(repoPath);
|
|
6114
6185
|
const cached = freshnessProbes.get(key);
|
|
6115
|
-
|
|
6186
|
+
// A watcher cannot prove that an edit made immediately before this call has
|
|
6187
|
+
// already reached its event queue. Git worktrees therefore take the bounded
|
|
6188
|
+
// porcelain proof on every answer. The long watcher lease remains safe for
|
|
6189
|
+
// non-Git directories, where the watcher is the only new-file signal.
|
|
6190
|
+
const gitBacked = fs.existsSync(path.join(key, ".git"));
|
|
6191
|
+
const lease = gitBacked
|
|
6192
|
+
? 0
|
|
6193
|
+
: watchQueues.get(key)?.ready
|
|
6194
|
+
? WATCHED_FRESHNESS_LEASE_MS
|
|
6195
|
+
: FRESHNESS_PROBE_TTL_MS;
|
|
6116
6196
|
const head = lease === WATCHED_FRESHNESS_LEASE_MS ? readGitHeadFast(key) : null;
|
|
6117
6197
|
const headUnchanged = head === null || (head !== undefined && head === getMeta(db, "lastIndexedHead"));
|
|
6118
6198
|
if (cached && Date.now() - cached.at < lease && headUnchanged) {
|
|
@@ -6336,9 +6416,9 @@ async function getOrInitDb(repoPath, options = {}) {
|
|
|
6336
6416
|
dbPath = ":memory:";
|
|
6337
6417
|
}
|
|
6338
6418
|
else {
|
|
6339
|
-
const knodinDir = path.join(normalizedPath, ".
|
|
6419
|
+
const knodinDir = path.join(normalizedPath, ".knodin");
|
|
6340
6420
|
await fs.promises.mkdir(knodinDir, { recursive: true });
|
|
6341
|
-
// Ensure .
|
|
6421
|
+
// Ensure .knodin is in .gitignore
|
|
6342
6422
|
try {
|
|
6343
6423
|
const gitignorePath = path.join(normalizedPath, ".gitignore");
|
|
6344
6424
|
let gitignoreContent = "";
|
|
@@ -6346,11 +6426,11 @@ async function getOrInitDb(repoPath, options = {}) {
|
|
|
6346
6426
|
gitignoreContent = await fs.promises.readFile(gitignorePath, "utf-8");
|
|
6347
6427
|
}
|
|
6348
6428
|
const lines = gitignoreContent.split("\n").map((l) => l.trim());
|
|
6349
|
-
if (!lines.includes(".
|
|
6350
|
-
!lines.includes(".
|
|
6351
|
-
!lines.includes("/.
|
|
6429
|
+
if (!lines.includes(".knodin") &&
|
|
6430
|
+
!lines.includes(".knodin/") &&
|
|
6431
|
+
!lines.includes("/.knodin")) {
|
|
6352
6432
|
const prefix = gitignoreContent.length > 0 && !gitignoreContent.endsWith("\n") ? "\n" : "";
|
|
6353
|
-
await fs.promises.appendFile(gitignorePath, `${prefix}\n# knodin\n.
|
|
6433
|
+
await fs.promises.appendFile(gitignorePath, `${prefix}\n# knodin\n.knodin\n`);
|
|
6354
6434
|
}
|
|
6355
6435
|
}
|
|
6356
6436
|
catch (e) {
|
|
@@ -6401,7 +6481,7 @@ async function getOrInitDb(repoPath, options = {}) {
|
|
|
6401
6481
|
// an unknown response shape, preserving C9's precision-first rule. v21
|
|
6402
6482
|
// adds source-only LWR/Experience Bundle topology and must revisit its
|
|
6403
6483
|
// newly indexable JSON files.
|
|
6404
|
-
const
|
|
6484
|
+
const KNODIN_SCHEMA_VERSION = 21;
|
|
6405
6485
|
// Highest version whose upgrade needs the stored data REBUILT. Versions
|
|
6406
6486
|
// above it migrate in place, so an upgrade costs a DELETE rather than a
|
|
6407
6487
|
// full re-index + re-embed (~40 min of CPU on an 18k-symbol corpus).
|
|
@@ -6411,7 +6491,7 @@ async function getOrInitDb(repoPath, options = {}) {
|
|
|
6411
6491
|
const needsMcpBackfill = storedVersion < 17;
|
|
6412
6492
|
// v13 -> v14 purges embeddings orphaned while the FK cascade was inert.
|
|
6413
6493
|
// Runs after the CREATE TABLEs below, since the table must exist.
|
|
6414
|
-
const needsOrphanPurge = storedVersion <
|
|
6494
|
+
const needsOrphanPurge = storedVersion < KNODIN_SCHEMA_VERSION;
|
|
6415
6495
|
if (storedVersion < LAST_REBUILD_SCHEMA_VERSION) {
|
|
6416
6496
|
db.run("DROP TRIGGER IF EXISTS after_symbol_insert;");
|
|
6417
6497
|
db.run("DROP TRIGGER IF EXISTS after_symbol_delete;");
|
|
@@ -6426,8 +6506,8 @@ async function getOrInitDb(repoPath, options = {}) {
|
|
|
6426
6506
|
db.run("DROP TABLE IF EXISTS mcp_tools;");
|
|
6427
6507
|
db.run("DROP TABLE IF EXISTS api_contracts;");
|
|
6428
6508
|
}
|
|
6429
|
-
if (storedVersion <
|
|
6430
|
-
db.run(`PRAGMA user_version = ${
|
|
6509
|
+
if (storedVersion < KNODIN_SCHEMA_VERSION) {
|
|
6510
|
+
db.run(`PRAGMA user_version = ${KNODIN_SCHEMA_VERSION};`);
|
|
6431
6511
|
}
|
|
6432
6512
|
db.run(`
|
|
6433
6513
|
CREATE TABLE IF NOT EXISTS symbols (
|
|
@@ -6712,15 +6792,15 @@ async function getFederatedRepos(repoPath) {
|
|
|
6712
6792
|
const repos = [resolvedRepoPath];
|
|
6713
6793
|
// Federation is EXPLICIT by default: a repo is treated as standalone unless it
|
|
6714
6794
|
// is told otherwise, so results never silently depend on what else happens to
|
|
6715
|
-
// live next to it. Two opt-in signals, read from `.
|
|
6795
|
+
// live next to it. Two opt-in signals, read from `.knodin/federation.json`
|
|
6716
6796
|
// ({ "repos"?: string[], "autoDiscover"?: boolean }):
|
|
6717
6797
|
// • repos — sibling paths (relative to this repo) to federate to.
|
|
6718
6798
|
// • autoDiscover — scan the parent directory for already-indexed siblings.
|
|
6719
|
-
// autoDiscover can also be forced with
|
|
6799
|
+
// autoDiscover can also be forced with KNODIN_AUTO_FEDERATE=1. Without either
|
|
6720
6800
|
// signal, knodin federates to exactly the repos listed in `repos` (possibly
|
|
6721
6801
|
// none) and nothing else.
|
|
6722
|
-
let autoDiscover = process.env.
|
|
6723
|
-
const configPath = path.join(resolvedRepoPath, ".
|
|
6802
|
+
let autoDiscover = process.env.KNODIN_AUTO_FEDERATE === "1" || process.env.KNODIN_AUTO_FEDERATE === "true";
|
|
6803
|
+
const configPath = path.join(resolvedRepoPath, ".knodin/federation.json");
|
|
6724
6804
|
if (fs.existsSync(configPath)) {
|
|
6725
6805
|
try {
|
|
6726
6806
|
const parsed = JSON.parse(fs.readFileSync(configPath, "utf-8"));
|
|
@@ -6744,7 +6824,7 @@ async function getFederatedRepos(repoPath) {
|
|
|
6744
6824
|
return repos;
|
|
6745
6825
|
}
|
|
6746
6826
|
// Auto-discovery (opt-in): siblings in the parent directory that knodin has
|
|
6747
|
-
// already indexed (i.e. have a .
|
|
6827
|
+
// already indexed (i.e. have a .knodin/db.sqlite). Under test the DB is
|
|
6748
6828
|
// in-memory, so no repo has that on-disk marker; it would only ever match
|
|
6749
6829
|
// repos a *production* run indexed elsewhere on the machine, so test
|
|
6750
6830
|
// federation is driven purely by in-process `dbInstances` — skip the disk
|
|
@@ -6761,7 +6841,7 @@ async function getFederatedRepos(repoPath) {
|
|
|
6761
6841
|
if (siblingPath === resolvedRepoPath)
|
|
6762
6842
|
continue;
|
|
6763
6843
|
const isAlreadyIndexed = dbInstances.has(path.resolve(siblingPath)) ||
|
|
6764
|
-
(!isTest && fs.existsSync(path.join(siblingPath, ".
|
|
6844
|
+
(!isTest && fs.existsSync(path.join(siblingPath, ".knodin", "db.sqlite")));
|
|
6765
6845
|
if (isAlreadyIndexed) {
|
|
6766
6846
|
repos.push(siblingPath);
|
|
6767
6847
|
}
|
|
@@ -6949,6 +7029,41 @@ function queryFileSymbols(db, filePath, repoPath) {
|
|
|
6949
7029
|
};
|
|
6950
7030
|
});
|
|
6951
7031
|
}
|
|
7032
|
+
function publishStructuralSnapshot(repoPath, db) {
|
|
7033
|
+
const indexed = db
|
|
7034
|
+
.query("SELECT filePath FROM index_state ORDER BY filePath")
|
|
7035
|
+
.all();
|
|
7036
|
+
const files = [];
|
|
7037
|
+
for (const { filePath } of indexed) {
|
|
7038
|
+
const absolute = path.resolve(repoPath, filePath);
|
|
7039
|
+
try {
|
|
7040
|
+
const content = fs.readFileSync(absolute);
|
|
7041
|
+
const stat = fs.statSync(absolute);
|
|
7042
|
+
files.push({
|
|
7043
|
+
path: filePath,
|
|
7044
|
+
sizeBytes: content.byteLength,
|
|
7045
|
+
mtimeMs: stat.mtimeMs,
|
|
7046
|
+
contentFingerprint: contentFingerprint(content),
|
|
7047
|
+
symbols: queryFileSymbols(db, filePath, repoPath).map((row) => ({
|
|
7048
|
+
identity: row.identity ?? null,
|
|
7049
|
+
symbol: row.symbol ?? "",
|
|
7050
|
+
kind: row.kind ?? "unknown",
|
|
7051
|
+
line: row.line ?? 1,
|
|
7052
|
+
endLine: row.endLine ?? row.line ?? 1,
|
|
7053
|
+
...(row.signature ? { signature: row.signature } : {}),
|
|
7054
|
+
...(row.visibility ? { visibility: row.visibility } : {}),
|
|
7055
|
+
...(row.exported !== undefined ? { exported: row.exported } : {}),
|
|
7056
|
+
...(row.parent ? { parent: row.parent } : {}),
|
|
7057
|
+
evidenceQuality: row.evidenceQuality ?? "parser-grounded",
|
|
7058
|
+
})),
|
|
7059
|
+
});
|
|
7060
|
+
}
|
|
7061
|
+
catch {
|
|
7062
|
+
// Health/freshness reports missing files; snapshots never invent an entry.
|
|
7063
|
+
}
|
|
7064
|
+
}
|
|
7065
|
+
writeStructuralSnapshot(repoPath, files);
|
|
7066
|
+
}
|
|
6952
7067
|
/**
|
|
6953
7068
|
* Name-level call adjacency from the `references` table — the single SQL +
|
|
6954
7069
|
* Map-building shared by `shortest_path` (directed) and `traverse` (undirected),
|
|
@@ -7643,43 +7758,16 @@ export function calculateASTComplexity(node) {
|
|
|
7643
7758
|
traverse(node);
|
|
7644
7759
|
return score;
|
|
7645
7760
|
}
|
|
7646
|
-
const gitChurnCache = new Map();
|
|
7647
|
-
const GIT_CHURN_CACHE_LIMIT = 8;
|
|
7648
7761
|
/**
|
|
7649
7762
|
* Count commits touching requested files with the legacy path-specific Git
|
|
7650
7763
|
* semantics. Results are cached by resolved repository + HEAD, so warm review
|
|
7651
7764
|
* launches no per-file history processes while preserving existing risk scores.
|
|
7652
7765
|
*/
|
|
7653
7766
|
export function getGitChurns(filePaths, repoPath) {
|
|
7654
|
-
const resolvedRepoPath = path.resolve(repoPath);
|
|
7655
7767
|
const requested = [...new Set(filePaths.map((file) => safeReviewFile(file)))];
|
|
7656
|
-
|
|
7657
|
-
|
|
7658
|
-
|
|
7659
|
-
}
|
|
7660
|
-
catch {
|
|
7661
|
-
// An unborn repository has no history, so every requested file has zero churn.
|
|
7662
|
-
}
|
|
7663
|
-
const cacheKey = `${resolvedRepoPath}\0${head}`;
|
|
7664
|
-
let counts = gitChurnCache.get(cacheKey);
|
|
7665
|
-
if (!counts) {
|
|
7666
|
-
counts = new Map();
|
|
7667
|
-
setBoundedCache(gitChurnCache, cacheKey, counts, GIT_CHURN_CACHE_LIMIT);
|
|
7668
|
-
}
|
|
7669
|
-
if (head !== "unborn") {
|
|
7670
|
-
for (const file of requested) {
|
|
7671
|
-
if (counts.has(file))
|
|
7672
|
-
continue;
|
|
7673
|
-
try {
|
|
7674
|
-
const output = runGit(resolvedRepoPath, ["log", "--oneline", "--", file]).trim();
|
|
7675
|
-
counts.set(file, output ? output.split("\n").length : 0);
|
|
7676
|
-
}
|
|
7677
|
-
catch {
|
|
7678
|
-
counts.set(file, 0);
|
|
7679
|
-
}
|
|
7680
|
-
}
|
|
7681
|
-
}
|
|
7682
|
-
return new Map(requested.map((file) => [file, counts?.get(file) ?? 0]));
|
|
7768
|
+
const history = collectGitHistorySignals(repoPath, requested);
|
|
7769
|
+
const counts = new Map(history.churn.map((fact) => [fact.file, fact.commitCount]));
|
|
7770
|
+
return new Map(requested.map((file) => [file, counts.get(file) ?? 0]));
|
|
7683
7771
|
}
|
|
7684
7772
|
export function getGitChurn(filePath, repoPath) {
|
|
7685
7773
|
return getGitChurns([filePath], repoPath).get(safeReviewFile(filePath)) ?? 0;
|
|
@@ -8643,7 +8731,7 @@ function renderCommunityPage(community, map) {
|
|
|
8643
8731
|
lines.push("");
|
|
8644
8732
|
return lines.join("\n");
|
|
8645
8733
|
}
|
|
8646
|
-
/** Renders `.
|
|
8734
|
+
/** Renders `.knodin/wiki/index.md`: repo path, community table, one link per page. */
|
|
8647
8735
|
function renderWikiIndex(resolvedRepoPath, map, slugs) {
|
|
8648
8736
|
const lines = [];
|
|
8649
8737
|
lines.push("# knodin Wiki", "");
|
|
@@ -8681,14 +8769,14 @@ function writeWikiPageIfChanged(dirPath, relFile, content, force, written, skipp
|
|
|
8681
8769
|
written.push(relFile);
|
|
8682
8770
|
}
|
|
8683
8771
|
/**
|
|
8684
|
-
* Generates the graph wiki (R5): `.
|
|
8772
|
+
* Generates the graph wiki (R5): `.knodin/wiki/index.md` + one page per
|
|
8685
8773
|
* `map()` community. Reuses `engine.map(, "standard")` for community/hub/bridge/edge
|
|
8686
8774
|
* data — no community-detection logic is duplicated here.
|
|
8687
8775
|
*/
|
|
8688
8776
|
async function writeWiki(engine, repoPath, force) {
|
|
8689
8777
|
const resolvedRepoPath = path.resolve(repoPath);
|
|
8690
8778
|
const map = await engine.map(repoPath, "standard");
|
|
8691
|
-
const wikiDir = path.join(resolvedRepoPath, ".
|
|
8779
|
+
const wikiDir = path.join(resolvedRepoPath, ".knodin", "wiki");
|
|
8692
8780
|
fs.mkdirSync(wikiDir, { recursive: true });
|
|
8693
8781
|
// Slugify community names into unique page filenames (lowercase, hyphenated).
|
|
8694
8782
|
const slugs = new Map();
|
|
@@ -9097,10 +9185,12 @@ export function createEngine() {
|
|
|
9097
9185
|
changedFiles: reviewChangedFiles(base, repoPath, options, modifiedFiles),
|
|
9098
9186
|
};
|
|
9099
9187
|
});
|
|
9100
|
-
const
|
|
9188
|
+
const historySignals = measurePerfPhaseSync("review_churn", () => collectGitHistorySignals(repoPath, changedFiles, options.historyLimits));
|
|
9189
|
+
const churnByFile = new Map(historySignals.churn.map((fact) => [fact.file, fact.commitCount]));
|
|
9101
9190
|
const changedSymbols = new Set();
|
|
9102
9191
|
const testGaps = new Set();
|
|
9103
9192
|
const symbolRisks = [];
|
|
9193
|
+
const structuralCentrality = new Map();
|
|
9104
9194
|
const finishParsing = beginPerfPhase("review_parsing");
|
|
9105
9195
|
try {
|
|
9106
9196
|
for (const filePath of changedFiles) {
|
|
@@ -9143,6 +9233,19 @@ export function createEngine() {
|
|
|
9143
9233
|
Array.from(lineSet).some((line) => line >= sym.startLine && line <= sym.endLine);
|
|
9144
9234
|
if (isChanged) {
|
|
9145
9235
|
changedSymbols.add(sym.name);
|
|
9236
|
+
const inboundReferences = db
|
|
9237
|
+
.query('SELECT COUNT(*) AS count FROM "references" WHERE calleeSymbol = ? AND (calleeFile = ? OR calleeFile IS NULL)')
|
|
9238
|
+
.get(sym.name, filePath)?.count ?? 0;
|
|
9239
|
+
const outboundReferences = db
|
|
9240
|
+
.query('SELECT COUNT(*) AS count FROM "references" WHERE callerSymbol = ? AND callerFile = ?')
|
|
9241
|
+
.get(sym.name, filePath)?.count ?? 0;
|
|
9242
|
+
structuralCentrality.set(`${filePath}\0${sym.name}`, {
|
|
9243
|
+
symbol: sym.name,
|
|
9244
|
+
file: filePath,
|
|
9245
|
+
inboundReferences,
|
|
9246
|
+
outboundReferences,
|
|
9247
|
+
totalReferences: inboundReferences + outboundReferences,
|
|
9248
|
+
});
|
|
9146
9249
|
let complexity = 1;
|
|
9147
9250
|
if (tree) {
|
|
9148
9251
|
try {
|
|
@@ -9214,11 +9317,18 @@ export function createEngine() {
|
|
|
9214
9317
|
const unmappedChangedFiles = changedFiles.filter((file) => !mappedSet.has(file));
|
|
9215
9318
|
const flowsList = Array.from(affectedFlowsSet);
|
|
9216
9319
|
const gapsList = Array.from(testGaps);
|
|
9320
|
+
const centralityFacts = [...structuralCentrality.values()].sort((a, b) => b.totalReferences - a.totalReferences ||
|
|
9321
|
+
a.file.localeCompare(b.file) ||
|
|
9322
|
+
a.symbol.localeCompare(b.symbol));
|
|
9217
9323
|
// Minimal detail caps each list so a client LLM isn't handed the full
|
|
9218
9324
|
// (potentially very large) result; the *Count fields always report the
|
|
9219
9325
|
// true totals so nothing is silently hidden.
|
|
9220
9326
|
const MINIMAL_CAP = 25;
|
|
9221
9327
|
const minimal = detailLevel === "minimal";
|
|
9328
|
+
const signalFiles = minimal ? changedFiles.slice(0, MINIMAL_CAP) : changedFiles;
|
|
9329
|
+
const signalFlows = minimal ? flowsList.slice(0, MINIMAL_CAP) : flowsList;
|
|
9330
|
+
const signalGaps = minimal ? gapsList.slice(0, MINIMAL_CAP) : gapsList;
|
|
9331
|
+
const signalCentrality = minimal ? centralityFacts.slice(0, MINIMAL_CAP) : centralityFacts;
|
|
9222
9332
|
const truncated = minimal &&
|
|
9223
9333
|
(changedFiles.length > MINIMAL_CAP ||
|
|
9224
9334
|
mappedChangedFiles.length > MINIMAL_CAP ||
|
|
@@ -9238,6 +9348,22 @@ export function createEngine() {
|
|
|
9238
9348
|
: unmappedChangedFiles,
|
|
9239
9349
|
unmappedChangedFileCount: unmappedChangedFiles.length,
|
|
9240
9350
|
riskScore,
|
|
9351
|
+
signals: {
|
|
9352
|
+
graphImpact: {
|
|
9353
|
+
affectedFlows: signalFlows,
|
|
9354
|
+
changedFiles: signalFiles,
|
|
9355
|
+
confidence: "exact-indexed-relationships",
|
|
9356
|
+
},
|
|
9357
|
+
testGaps: {
|
|
9358
|
+
facts: signalGaps,
|
|
9359
|
+
confidence: "indexed-references-and-test-paths",
|
|
9360
|
+
},
|
|
9361
|
+
structuralCentrality: {
|
|
9362
|
+
facts: signalCentrality,
|
|
9363
|
+
confidence: "exact-indexed-references",
|
|
9364
|
+
},
|
|
9365
|
+
history: historySignals,
|
|
9366
|
+
},
|
|
9241
9367
|
changedSymbols: minimal ? changedList.slice(0, MINIMAL_CAP) : changedList,
|
|
9242
9368
|
changedSymbolCount: changedList.length,
|
|
9243
9369
|
affectedFlows: minimal ? flowsList.slice(0, MINIMAL_CAP) : flowsList,
|
|
@@ -9651,7 +9777,7 @@ export function createEngine() {
|
|
|
9651
9777
|
// repair and timestamp drift before reporting it. Reuse an active DB in
|
|
9652
9778
|
// long-lived processes, or open the existing local DB read-only.
|
|
9653
9779
|
const activeDb = dbInstances.get(resolved);
|
|
9654
|
-
const diskDbPath = path.join(resolved, ".
|
|
9780
|
+
const diskDbPath = path.join(resolved, ".knodin", "db.sqlite");
|
|
9655
9781
|
const db = activeDb ??
|
|
9656
9782
|
(fs.existsSync(diskDbPath) ? new Database(diskDbPath, { readonly: true }) : null);
|
|
9657
9783
|
const verifiedAt = new Date().toISOString();
|
|
@@ -10057,7 +10183,7 @@ export function createEngine() {
|
|
|
10057
10183
|
// Repair partial current-schema databases before normal initialization,
|
|
10058
10184
|
// whose reconciliation queries require these columns. ALTER preserves all
|
|
10059
10185
|
// healthy index_state rows; the reconciler then refreshes their snapshots.
|
|
10060
|
-
const diskDbPath = path.join(resolved, ".
|
|
10186
|
+
const diskDbPath = path.join(resolved, ".knodin", "db.sqlite");
|
|
10061
10187
|
if (fs.existsSync(diskDbPath)) {
|
|
10062
10188
|
const activeDb = dbInstances.get(resolved);
|
|
10063
10189
|
const schemaDb = activeDb ?? new Database(diskDbPath);
|
|
@@ -10311,6 +10437,7 @@ export function createEngine() {
|
|
|
10311
10437
|
async index(repoPath, files, clean = false, options) {
|
|
10312
10438
|
const indexed = [];
|
|
10313
10439
|
const unchanged = [];
|
|
10440
|
+
let scipReport;
|
|
10314
10441
|
const progress = createIndexProgressReporter(options?.onProgress);
|
|
10315
10442
|
progress("starting", 0, "Opening local graph database");
|
|
10316
10443
|
const db = await getOrInitDb(repoPath, {
|
|
@@ -10396,6 +10523,33 @@ export function createEngine() {
|
|
|
10396
10523
|
indexed.push(...collectRepoFiles(repoPath));
|
|
10397
10524
|
completionMessage = `${indexed.length.toLocaleString()} repository file(s) are current`;
|
|
10398
10525
|
}
|
|
10526
|
+
if (options?.scip) {
|
|
10527
|
+
const facts = readScipIndex(repoPath, options.scip);
|
|
10528
|
+
const affectedFiles = persistScipFacts(db, repoPath, facts);
|
|
10529
|
+
progress("finalizing", 0, "Refreshing SCIP identities and semantic embeddings");
|
|
10530
|
+
await indexEmbeddings(db, repoPath, progress);
|
|
10531
|
+
indexGeneration++;
|
|
10532
|
+
const bounds = { ...SCIP_DEFAULT_LIMITS, ...options.scip };
|
|
10533
|
+
scipReport = {
|
|
10534
|
+
provenance: "scip",
|
|
10535
|
+
inputPath: facts.inputPath,
|
|
10536
|
+
bytes: facts.bytes,
|
|
10537
|
+
files: facts.files,
|
|
10538
|
+
facts: facts.facts,
|
|
10539
|
+
languages: facts.languages,
|
|
10540
|
+
indexers: facts.indexers,
|
|
10541
|
+
elapsedMs: facts.elapsedMs,
|
|
10542
|
+
bounds: {
|
|
10543
|
+
maxBytes: bounds.maxBytes,
|
|
10544
|
+
maxFiles: bounds.maxFiles,
|
|
10545
|
+
maxFacts: bounds.maxFacts,
|
|
10546
|
+
timeoutMs: bounds.timeoutMs,
|
|
10547
|
+
},
|
|
10548
|
+
};
|
|
10549
|
+
for (const file of affectedFiles)
|
|
10550
|
+
if (!indexed.includes(file))
|
|
10551
|
+
indexed.push(file);
|
|
10552
|
+
}
|
|
10399
10553
|
progress("verifying", 0, "Verifying local graph health", {
|
|
10400
10554
|
phaseTotal: 1,
|
|
10401
10555
|
});
|
|
@@ -10431,6 +10585,10 @@ export function createEngine() {
|
|
|
10431
10585
|
recordFreshnessBaseline(repoPath, db);
|
|
10432
10586
|
}
|
|
10433
10587
|
noteFreshness(repoPath, "fresh");
|
|
10588
|
+
// Test engines use an in-memory graph and must not create untracked
|
|
10589
|
+
// repository artifacts that pollute Git-diff/review fixtures.
|
|
10590
|
+
if (!process.env.VITEST && process.env.NODE_ENV !== "test")
|
|
10591
|
+
publishStructuralSnapshot(repoPath, db);
|
|
10434
10592
|
}
|
|
10435
10593
|
else {
|
|
10436
10594
|
freshnessProbes.delete(path.resolve(repoPath));
|
|
@@ -10445,6 +10603,7 @@ export function createEngine() {
|
|
|
10445
10603
|
return {
|
|
10446
10604
|
indexed,
|
|
10447
10605
|
unchanged,
|
|
10606
|
+
scip: scipReport,
|
|
10448
10607
|
verification: {
|
|
10449
10608
|
status: health.status,
|
|
10450
10609
|
issueCount,
|
|
@@ -10462,6 +10621,7 @@ export function createEngine() {
|
|
|
10462
10621
|
throw new Error("knodin search: invalid testScope");
|
|
10463
10622
|
const repos = options.federate === false ? [resolvedRepoPath] : await getFederatedRepos(repoPath);
|
|
10464
10623
|
const allRepos = await Promise.all(repos.map(async (r) => ({ path: r, db: await getOrInitDb(r) })));
|
|
10624
|
+
const searchSnapshots = new Map(allRepos.map((repo) => [repo.path, getSearchMetadataSnapshot(repo.path, repo.db)]));
|
|
10465
10625
|
// typeof + Number.isFinite rejects undefined/null/NaN alike, so any non-numeric
|
|
10466
10626
|
// or unset `limit` falls back to the default instead of clamping to 0/1 or
|
|
10467
10627
|
// propagating NaN into Array.prototype.slice (which silently empties results).
|
|
@@ -10503,12 +10663,90 @@ export function createEngine() {
|
|
|
10503
10663
|
return true;
|
|
10504
10664
|
};
|
|
10505
10665
|
const formatPath = makePathFormatter(resolvedRepoPath);
|
|
10666
|
+
const retrievalMode = options.retrievalMode ?? "complete";
|
|
10667
|
+
const normalize = (value) => value.toLocaleLowerCase().replace(/[^a-z0-9]/g, "");
|
|
10668
|
+
const hydrateStructural = (matches) => {
|
|
10669
|
+
const fileToCommunityMap = matches.length > 0
|
|
10670
|
+
? getSearchCommunityLookup(allRepos, resolvedRepoPath)
|
|
10671
|
+
: new Map();
|
|
10672
|
+
return matches.map(({ repo, row }) => {
|
|
10673
|
+
let source = "";
|
|
10674
|
+
if (options.includeSource !== false)
|
|
10675
|
+
source = getSourceRange(repo.path, row.filePath, row.startLine, row.endLine, 500);
|
|
10676
|
+
const callersStmt = repo.db.query(`
|
|
10677
|
+
SELECT DISTINCT callerSymbol FROM "references"
|
|
10678
|
+
WHERE calleeSymbol = ? AND callerSymbol IS NOT NULL AND callerSymbol != ?
|
|
10679
|
+
AND (calleeFile = ? OR calleeFile IS NULL)
|
|
10680
|
+
`);
|
|
10681
|
+
const directCallers = callersStmt
|
|
10682
|
+
.all(row.name, row.name, row.filePath)
|
|
10683
|
+
.map((caller) => caller.callerSymbol)
|
|
10684
|
+
.sort((a, b) => a.localeCompare(b));
|
|
10685
|
+
callersStmt.finalize();
|
|
10686
|
+
const formattedFile = formatPath(repo.path, row.filePath);
|
|
10687
|
+
return {
|
|
10688
|
+
identity: symbolIdentity(repo.path, row),
|
|
10689
|
+
symbol: row.name,
|
|
10690
|
+
kind: row.kind,
|
|
10691
|
+
filePath: formattedFile,
|
|
10692
|
+
similarity: 1,
|
|
10693
|
+
rrfScore: 1,
|
|
10694
|
+
source,
|
|
10695
|
+
belongsToCommunity: fileToCommunityMap.get(formattedFile) || "core-module",
|
|
10696
|
+
directCallers,
|
|
10697
|
+
};
|
|
10698
|
+
});
|
|
10699
|
+
};
|
|
10700
|
+
const expandBounded = (seeds) => {
|
|
10701
|
+
const expansionWorkLimit = Math.min(200, Math.max(candidateLimit * 4, 20));
|
|
10702
|
+
const selected = seeds.slice(0, expansionWorkLimit);
|
|
10703
|
+
const seen = new Set(selected.map(({ repo, row }) => `${repo.path}:${row.id}`));
|
|
10704
|
+
let frontier = [...seeds];
|
|
10705
|
+
for (let depth = 0; depth < 2 && frontier.length > 0 && selected.length < expansionWorkLimit; depth++) {
|
|
10706
|
+
const next = [];
|
|
10707
|
+
for (const current of frontier) {
|
|
10708
|
+
const refs = current.repo.db
|
|
10709
|
+
.query(`SELECT callerSymbol, callerFile, calleeSymbol, calleeFile FROM "references"
|
|
10710
|
+
WHERE (callerSymbol = ? AND callerFile = ?)
|
|
10711
|
+
OR (calleeSymbol = ? AND calleeFile = ?)
|
|
10712
|
+
ORDER BY callerFile, callerSymbol, calleeFile, calleeSymbol
|
|
10713
|
+
LIMIT 200`)
|
|
10714
|
+
.all(current.row.name, current.row.filePath, current.row.name, current.row.filePath);
|
|
10715
|
+
const snapshot = searchSnapshots.get(current.repo.path);
|
|
10716
|
+
for (const ref of refs) {
|
|
10717
|
+
const fromCaller = ref.callerSymbol === current.row.name && ref.callerFile === current.row.filePath;
|
|
10718
|
+
const neighborName = fromCaller ? ref.calleeSymbol : ref.callerSymbol;
|
|
10719
|
+
const neighborFile = fromCaller ? ref.calleeFile : ref.callerFile;
|
|
10720
|
+
if (!neighborName || !neighborFile)
|
|
10721
|
+
continue;
|
|
10722
|
+
for (const row of snapshot.rows) {
|
|
10723
|
+
if (row.name !== neighborName || !accepts(row))
|
|
10724
|
+
continue;
|
|
10725
|
+
if (neighborFile && row.filePath !== neighborFile)
|
|
10726
|
+
continue;
|
|
10727
|
+
const key = `${current.repo.path}:${row.id}`;
|
|
10728
|
+
if (seen.has(key))
|
|
10729
|
+
continue;
|
|
10730
|
+
if (seen.size >= expansionWorkLimit)
|
|
10731
|
+
break;
|
|
10732
|
+
seen.add(key);
|
|
10733
|
+
next.push({ repo: current.repo, row });
|
|
10734
|
+
}
|
|
10735
|
+
}
|
|
10736
|
+
}
|
|
10737
|
+
next.sort((a, b) => a.repo.path.localeCompare(b.repo.path) ||
|
|
10738
|
+
a.row.filePath.localeCompare(b.row.filePath) ||
|
|
10739
|
+
a.row.startLine - b.row.startLine);
|
|
10740
|
+
selected.push(...next.slice(0, expansionWorkLimit - selected.length));
|
|
10741
|
+
frontier = next;
|
|
10742
|
+
}
|
|
10743
|
+
return selected;
|
|
10744
|
+
};
|
|
10506
10745
|
if (options.structural) {
|
|
10507
|
-
const normalize = (value) => value.toLocaleLowerCase().replace(/[^a-z0-9]/g, "");
|
|
10508
10746
|
const needle = normalize(query);
|
|
10509
10747
|
const matches = allRepos
|
|
10510
|
-
.flatMap((repo) =>
|
|
10511
|
-
.
|
|
10748
|
+
.flatMap((repo) => searchSnapshots.get(repo.path).rows
|
|
10749
|
+
.filter((row) => accepts(row) && normalize(row.name).includes(needle))
|
|
10512
10750
|
.map((row) => ({ repo, row })))
|
|
10513
10751
|
.sort((left, right) => left.row.name.localeCompare(right.row.name) ||
|
|
10514
10752
|
left.row.filePath.localeCompare(right.row.filePath) ||
|
|
@@ -10532,8 +10770,140 @@ export function createEngine() {
|
|
|
10532
10770
|
offset,
|
|
10533
10771
|
limit: maxResults,
|
|
10534
10772
|
hasMore: offset + maxResults < matches.length,
|
|
10773
|
+
retrieval: { route: ["exact-name"], embeddingsUsed: false },
|
|
10535
10774
|
};
|
|
10536
10775
|
}
|
|
10776
|
+
if (retrievalMode !== "vector-only") {
|
|
10777
|
+
const trimmedQuery = query.trim().replaceAll("\\", "/");
|
|
10778
|
+
const normalizedQuery = normalize(trimmedQuery);
|
|
10779
|
+
const identityQuery = trimmedQuery.startsWith("sym_") ? trimmedQuery : undefined;
|
|
10780
|
+
const pathQuery = trimmedQuery.includes("/") || /\.[A-Za-z0-9]+$/.test(trimmedQuery);
|
|
10781
|
+
const seedMatches = allRepos
|
|
10782
|
+
.flatMap((repo) => searchSnapshots.get(repo.path).rows
|
|
10783
|
+
.filter((row) => {
|
|
10784
|
+
if (!accepts(row))
|
|
10785
|
+
return false;
|
|
10786
|
+
if (identityQuery)
|
|
10787
|
+
return symbolIdentityMatches(symbolIdentity(repo.path, row), identityQuery);
|
|
10788
|
+
if (pathQuery) {
|
|
10789
|
+
const candidate = row.filePath.replaceAll("\\", "/");
|
|
10790
|
+
return candidate === trimmedQuery || candidate.endsWith(`/${trimmedQuery}`);
|
|
10791
|
+
}
|
|
10792
|
+
return normalizedQuery.length > 0 && normalize(row.name) === normalizedQuery;
|
|
10793
|
+
})
|
|
10794
|
+
.map((row) => ({ repo, row })))
|
|
10795
|
+
.sort((a, b) => a.repo.path.localeCompare(b.repo.path) ||
|
|
10796
|
+
a.row.filePath.localeCompare(b.row.filePath) ||
|
|
10797
|
+
a.row.startLine - b.row.startLine);
|
|
10798
|
+
if (seedMatches.length > 0) {
|
|
10799
|
+
const selected = retrievalMode === "complete" || retrievalMode === "bounded-graph"
|
|
10800
|
+
? expandBounded(seedMatches)
|
|
10801
|
+
: seedMatches;
|
|
10802
|
+
const pageMatches = selected.slice(offset, offset + maxResults);
|
|
10803
|
+
return {
|
|
10804
|
+
results: hydrateStructural(pageMatches),
|
|
10805
|
+
total: selected.length,
|
|
10806
|
+
offset,
|
|
10807
|
+
limit: maxResults,
|
|
10808
|
+
hasMore: offset + maxResults < selected.length,
|
|
10809
|
+
retrieval: {
|
|
10810
|
+
route: retrievalMode === "complete" || retrievalMode === "bounded-graph"
|
|
10811
|
+
? [
|
|
10812
|
+
identityQuery ? "stable-identity" : pathQuery ? "exact-path" : "exact-name",
|
|
10813
|
+
"bounded-graph",
|
|
10814
|
+
]
|
|
10815
|
+
: [identityQuery ? "stable-identity" : pathQuery ? "exact-path" : "exact-name"],
|
|
10816
|
+
embeddingsUsed: false,
|
|
10817
|
+
},
|
|
10818
|
+
};
|
|
10819
|
+
}
|
|
10820
|
+
}
|
|
10821
|
+
const lexicalRanks = new Map();
|
|
10822
|
+
if (retrievalMode !== "vector-only") {
|
|
10823
|
+
for (const repo of allRepos) {
|
|
10824
|
+
const snapshot = searchSnapshots.get(repo.path);
|
|
10825
|
+
const allowedIds = new Set(snapshot.rows.filter(accepts).map((row) => row.id));
|
|
10826
|
+
const tokens = query
|
|
10827
|
+
.replace(/[^\w\s]/g, " ")
|
|
10828
|
+
.trim()
|
|
10829
|
+
.split(/\s+/)
|
|
10830
|
+
.filter((word) => word.length >= 4)
|
|
10831
|
+
.map((word) => `"${word}"`);
|
|
10832
|
+
const runFts = (separator) => {
|
|
10833
|
+
if (tokens.length === 0)
|
|
10834
|
+
return [];
|
|
10835
|
+
try {
|
|
10836
|
+
const statement = repo.db.query(`
|
|
10837
|
+
SELECT symbolId FROM symbols_fts
|
|
10838
|
+
WHERE symbols_fts MATCH ?
|
|
10839
|
+
ORDER BY bm25(symbols_fts), symbolId
|
|
10840
|
+
LIMIT 100
|
|
10841
|
+
`);
|
|
10842
|
+
const ids = statement
|
|
10843
|
+
.all(tokens.join(separator))
|
|
10844
|
+
.map(({ symbolId }) => symbolId)
|
|
10845
|
+
.filter((id) => allowedIds.has(id));
|
|
10846
|
+
statement.finalize();
|
|
10847
|
+
return ids;
|
|
10848
|
+
}
|
|
10849
|
+
catch (error) {
|
|
10850
|
+
console.error(`FTS query failed for "${query}":`, error);
|
|
10851
|
+
return [];
|
|
10852
|
+
}
|
|
10853
|
+
};
|
|
10854
|
+
const and = measurePerfPhaseSync("fts", () => runFts(" "));
|
|
10855
|
+
const or = tokens.length > 1 ? measurePerfPhaseSync("fts", () => runFts(" OR ")) : and;
|
|
10856
|
+
const lexicalSeeds = or
|
|
10857
|
+
.map((id) => snapshot.rowById.get(id))
|
|
10858
|
+
.filter((row) => row !== undefined)
|
|
10859
|
+
.map((row) => ({ repo, row }));
|
|
10860
|
+
const structural = expandBounded(lexicalSeeds)
|
|
10861
|
+
.filter(({ repo: matchRepo }) => matchRepo.path === repo.path)
|
|
10862
|
+
.map(({ row }) => row.id);
|
|
10863
|
+
const sufficientSeeds = and
|
|
10864
|
+
.map((id) => snapshot.rowById.get(id))
|
|
10865
|
+
.filter((row) => row !== undefined)
|
|
10866
|
+
.map((row) => ({ repo, row }));
|
|
10867
|
+
const sufficient = expandBounded(sufficientSeeds)
|
|
10868
|
+
.filter(({ repo: matchRepo }) => matchRepo.path === repo.path)
|
|
10869
|
+
.map(({ row }) => row.id);
|
|
10870
|
+
lexicalRanks.set(repo.path, { and, or, structural, sufficient });
|
|
10871
|
+
}
|
|
10872
|
+
const structurallySufficient = retrievalMode === "complete" &&
|
|
10873
|
+
!ann.annEnabled() &&
|
|
10874
|
+
[...lexicalRanks.values()].some((ranks) => ranks.and.length > 0);
|
|
10875
|
+
if (retrievalMode === "lexical-only" ||
|
|
10876
|
+
retrievalMode === "bounded-graph" ||
|
|
10877
|
+
structurallySufficient) {
|
|
10878
|
+
const matches = allRepos.flatMap((repo) => {
|
|
10879
|
+
const snapshot = searchSnapshots.get(repo.path);
|
|
10880
|
+
const ranks = lexicalRanks.get(repo.path);
|
|
10881
|
+
const ids = structurallySufficient
|
|
10882
|
+
? ranks?.sufficient
|
|
10883
|
+
: retrievalMode === "bounded-graph"
|
|
10884
|
+
? ranks?.structural
|
|
10885
|
+
: ranks?.or;
|
|
10886
|
+
return (ids ?? [])
|
|
10887
|
+
.map((id) => snapshot.rowById.get(id))
|
|
10888
|
+
.filter((row) => row !== undefined)
|
|
10889
|
+
.map((row) => ({ repo, row }));
|
|
10890
|
+
});
|
|
10891
|
+
const pageMatches = matches.slice(offset, offset + maxResults);
|
|
10892
|
+
return {
|
|
10893
|
+
results: hydrateStructural(pageMatches),
|
|
10894
|
+
total: matches.length,
|
|
10895
|
+
offset,
|
|
10896
|
+
limit: maxResults,
|
|
10897
|
+
hasMore: offset + maxResults < matches.length,
|
|
10898
|
+
retrieval: {
|
|
10899
|
+
route: retrievalMode === "bounded-graph" || structurallySufficient
|
|
10900
|
+
? ["lexical", "bounded-graph"]
|
|
10901
|
+
: ["lexical"],
|
|
10902
|
+
embeddingsUsed: false,
|
|
10903
|
+
},
|
|
10904
|
+
};
|
|
10905
|
+
}
|
|
10906
|
+
}
|
|
10537
10907
|
// 1. Generate query embedding once
|
|
10538
10908
|
const queryVec = await generateMeasuredEmbedding(query, true);
|
|
10539
10909
|
const rankedCandidates = [];
|
|
@@ -10542,7 +10912,7 @@ export function createEngine() {
|
|
|
10542
10912
|
const db = repo.db;
|
|
10543
10913
|
// 2. Reuse decoded generation-scoped metadata. Cheap metadata-only
|
|
10544
10914
|
// filters run before any exact vector score is computed.
|
|
10545
|
-
const snapshot =
|
|
10915
|
+
const snapshot = searchSnapshots.get(repo.path);
|
|
10546
10916
|
const allRows = snapshot.rows.filter(accepts);
|
|
10547
10917
|
const allowedIds = new Set(allRows.map((row) => row.id));
|
|
10548
10918
|
totalMatches += allowedIds.size;
|
|
@@ -10551,7 +10921,7 @@ export function createEngine() {
|
|
|
10551
10921
|
// full-corpus graph. Keep the opt-in ANN path for unfiltered searches;
|
|
10552
10922
|
// filtered requests use the exact default over only eligible rows.
|
|
10553
10923
|
if (ann.annEnabled() && !hasFilters) {
|
|
10554
|
-
// R18 OPT-IN ANN PATH (
|
|
10924
|
+
// R18 OPT-IN ANN PATH (KNODIN_ANN=1). Build/reuse an in-memory HNSW for
|
|
10555
10925
|
// this repo and query it instead of computing cosine over every row. The
|
|
10556
10926
|
// exact scan below stays the DEFAULT at every corpus size; this path only
|
|
10557
10927
|
// engages behind the env flag, until a measured >=95% top-10 recall gate
|
|
@@ -10587,39 +10957,10 @@ export function createEngine() {
|
|
|
10587
10957
|
const sortedSemantics = semanticMatches.sort((a, b) => b.score - a.score);
|
|
10588
10958
|
const semanticScoresAreTied = sortedSemantics.length > 1 &&
|
|
10589
10959
|
sortedSemantics.every((item) => item.score === sortedSemantics[0]?.score);
|
|
10590
|
-
// 3.
|
|
10591
|
-
const
|
|
10592
|
-
|
|
10593
|
-
|
|
10594
|
-
const cleanQuery = query.replace(/[^\w\s]/g, " ").trim();
|
|
10595
|
-
if (cleanQuery) {
|
|
10596
|
-
// Quote each token so FTS5 treats bare AND/OR/NOT/NEAR as literal search
|
|
10597
|
-
// words instead of query-syntax operators (which would otherwise throw
|
|
10598
|
-
// a MATCH syntax error for a query like "NOT authenticated").
|
|
10599
|
-
const ftsQuery = cleanQuery
|
|
10600
|
-
.split(/\s+/)
|
|
10601
|
-
.filter((word) => word.length >= 4)
|
|
10602
|
-
.map((word) => `"${word}"`)
|
|
10603
|
-
// With a useful semantic ordering, retain the precise all-token
|
|
10604
|
-
// FTS signal. When a deterministic/fallback embedder ties every
|
|
10605
|
-
// candidate, permit partial lexical matches so connective words
|
|
10606
|
-
// do not reduce ranking to arbitrary insertion order.
|
|
10607
|
-
.join(semanticScoresAreTied ? " OR " : " ");
|
|
10608
|
-
const ftsStmt = db.query(`
|
|
10609
|
-
SELECT symbolId FROM symbols_fts WHERE symbols_fts MATCH ? ORDER BY bm25(symbols_fts) LIMIT 100
|
|
10610
|
-
`);
|
|
10611
|
-
matches = ftsStmt
|
|
10612
|
-
.all(ftsQuery)
|
|
10613
|
-
.map((m) => m.symbolId)
|
|
10614
|
-
.filter((id) => allowedIds.has(id));
|
|
10615
|
-
ftsStmt.finalize();
|
|
10616
|
-
}
|
|
10617
|
-
}
|
|
10618
|
-
catch (err) {
|
|
10619
|
-
console.error(`FTS query failed for "${query}":`, err);
|
|
10620
|
-
}
|
|
10621
|
-
return matches;
|
|
10622
|
-
});
|
|
10960
|
+
// 3. Reuse the lexical and bounded-graph ranks computed before query inference.
|
|
10961
|
+
const lexical = lexicalRanks.get(repo.path);
|
|
10962
|
+
const baseLexical = semanticScoresAreTied ? lexical?.or : lexical?.and;
|
|
10963
|
+
const ftsMatches = [...new Set(baseLexical ?? [])].filter((id) => allowedIds.has(id));
|
|
10623
10964
|
// Rank mappings for Reciprocal Rank Fusion (RRF)
|
|
10624
10965
|
const semanticRankMap = new Map();
|
|
10625
10966
|
const semanticById = new Map();
|
|
@@ -10708,6 +11049,12 @@ export function createEngine() {
|
|
|
10708
11049
|
offset,
|
|
10709
11050
|
limit: maxResults,
|
|
10710
11051
|
hasMore: offset + maxResults < total,
|
|
11052
|
+
retrieval: {
|
|
11053
|
+
route: retrievalMode === "vector-only"
|
|
11054
|
+
? ["embeddings"]
|
|
11055
|
+
: ["lexical", "bounded-graph", "embeddings"],
|
|
11056
|
+
embeddingsUsed: true,
|
|
11057
|
+
},
|
|
10711
11058
|
};
|
|
10712
11059
|
},
|
|
10713
11060
|
async query(pattern, target, repoPath, to, limit, depth, detailLevel, selector = {}, impactOptions = {}, options = {}) {
|
|
@@ -12933,7 +13280,7 @@ export function createEngine() {
|
|
|
12933
13280
|
for (const rel of wouldChange) {
|
|
12934
13281
|
const abs = path.resolve(resolvedRepoPath, rel);
|
|
12935
13282
|
const content = newContents.get(rel);
|
|
12936
|
-
const tmp = path.join(path.dirname(abs), `.${path.basename(abs)}.
|
|
13283
|
+
const tmp = path.join(path.dirname(abs), `.${path.basename(abs)}.knodin-rename-${process.pid}-${Date.now()}.tmp`);
|
|
12937
13284
|
fs.writeFileSync(tmp, content, "utf-8");
|
|
12938
13285
|
fs.renameSync(tmp, abs);
|
|
12939
13286
|
}
|
|
@@ -13018,7 +13365,7 @@ export function createEngine() {
|
|
|
13018
13365
|
statusCache.clear();
|
|
13019
13366
|
flowsCache.clear();
|
|
13020
13367
|
fileToFlowLookupCache.clear();
|
|
13021
|
-
|
|
13368
|
+
clearGitHistorySignalCache();
|
|
13022
13369
|
annIndexCache.clear();
|
|
13023
13370
|
searchMetadataCache.clear();
|
|
13024
13371
|
searchCommunityLookupCache.clear();
|