knodin 0.6.0 → 0.7.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. package/README.md +26 -3
  2. package/dist/bin/cli.js +168 -12
  3. package/dist/bin/launcher.js +11 -0
  4. package/dist/src/agent-integration.js +3 -1
  5. package/dist/src/cli-args.js +8 -1
  6. package/dist/src/cli-model.js +35 -1
  7. package/dist/src/codeflow-replay.js +80 -0
  8. package/dist/src/competitive-cold-mcp.js +40 -0
  9. package/dist/src/competitive-manifest.js +106 -25
  10. package/dist/src/competitive-runner.js +37 -3
  11. package/dist/src/competitive-sandbox.js +1 -1
  12. package/dist/src/diagnostics.js +449 -0
  13. package/dist/src/engine/git-history.js +289 -0
  14. package/dist/src/engine/index.js +417 -70
  15. package/dist/src/engine/scip-import.js +408 -0
  16. package/dist/src/execution-profile.js +203 -0
  17. package/dist/src/failure-diagnosis.js +69 -10
  18. package/dist/src/hook-manager-integration.js +156 -0
  19. package/dist/src/init.js +319 -33
  20. package/dist/src/lifecycle-health.js +42 -4
  21. package/dist/src/output-telemetry.js +4 -0
  22. package/dist/src/progressive-evidence.js +473 -0
  23. package/dist/src/pure-compression-cli.js +101 -0
  24. package/dist/src/release-preflight.js +510 -0
  25. package/dist/src/repository-management.js +142 -0
  26. package/dist/src/response-budget.js +11 -1
  27. package/dist/src/server.js +22 -2
  28. package/dist/src/structural-fast-path.js +338 -0
  29. package/dist/src/structural-snapshot.js +33 -0
  30. package/dist/src/tools/knodin-tools.js +105 -14
  31. package/dist/src/update-ceremony.js +158 -0
  32. package/docs/CLI.md +22 -0
  33. package/docs/COMMAND-OUTPUT-COMPRESSION.md +31 -15
  34. package/docs/CONTAINED-EXECUTION.md +77 -0
  35. package/docs/DIAGNOSTICS.md +45 -0
  36. package/docs/DOCTOR-AND-UPDATES.md +5 -2
  37. package/docs/GIT-HISTORY-REVIEW.md +39 -0
  38. package/docs/MCP.md +15 -0
  39. package/docs/PROGRESSIVE-EVIDENCE.md +37 -0
  40. package/docs/REPOSITORIES-AND-WORKTREES.md +30 -0
  41. package/docs/SCIP-IMPORT.md +57 -0
  42. package/docs/SIGNED-UPDATES.md +5 -0
  43. package/docs/TELEMETRY.md +4 -0
  44. package/docs/releases/0.7.0.md +24 -0
  45. package/docs/releases/0.7.1.md +21 -0
  46. package/docs/releases/0.7.2.md +21 -0
  47. package/docs/releases/0.7.3.md +23 -0
  48. package/docs/releases/0.7.4.md +17 -0
  49. package/package.json +34 -2
@@ -20,12 +20,15 @@ import chokidar from "chokidar";
20
20
  import Parser from "web-tree-sitter";
21
21
  import { readIndexActivity } from "../index-activity.js";
22
22
  import { runReadOnlyLspQuery } from "../lsp-readonly.js";
23
+ import { contentFingerprint, writeStructuralSnapshot, } from "../structural-snapshot.js";
23
24
  import { KNODIN_VERSION } from "../version.js";
24
25
  import * as ann from "./ann-hnsw.js";
25
26
  import { computeSimilarity, generateEmbedding, generateEmbeddings, } from "./embeddings.js";
26
27
  import { walkRepoFiles } from "./file-walker.js";
28
+ import { clearGitHistorySignalCache, collectGitHistorySignals, } from "./git-history.js";
27
29
  import { beginPerfPhase, measurePerfPhase, measurePerfPhaseSync } from "./perf.js";
28
30
  import { isIndexablePath, makeWatchIgnorePredicate } from "./prune.js";
31
+ import { readScipIndex, SCIP_DEFAULT_LIMITS, } from "./scip-import.js";
29
32
  import { isIndexableSourcePath } from "./source-policy.js";
30
33
  import { Database } from "./sqlite.js";
31
34
  import { deleteAllSymbols, deleteSymbolsForFile, deleteSymbolsMatchingPath, ORPHANED_EMBEDDING_PREDICATE, purgeOrphanEmbeddings, } from "./symbol-delete.js";
@@ -3900,6 +3903,73 @@ async function indexLsifDump(absolutePath, _relativePath, repoPath, db) {
3900
3903
  console.error("Error indexing LSIF:", error);
3901
3904
  }
3902
3905
  }
3906
+ /** Replace only the optional SCIP tier, preserving native and LSIF facts verbatim. */
3907
+ function persistScipFacts(db, repoPath, facts) {
3908
+ const affectedFiles = [...new Set(facts.symbols.map((symbol) => symbol.filePath))].sort();
3909
+ const insertSymbol = db.prepare(`
3910
+ INSERT INTO symbols (name, kind, filePath, startLine, endLine, startCol, endCol, summary)
3911
+ VALUES (?, ?, ?, ?, ?, ?, ?, ?)
3912
+ `);
3913
+ const insertReference = db.prepare(`
3914
+ INSERT INTO "references" (callerSymbol, callerFile, calleeSymbol, calleeFile, line, column, kind, confidence)
3915
+ VALUES (?, ?, ?, ?, ?, ?, ?, 1.0)
3916
+ `);
3917
+ const insertDependency = db.prepare(`
3918
+ INSERT INTO dependencies (fromFile, toFile, kind, confidence, sourceEvidence)
3919
+ VALUES (?, ?, 'scip_import', 1.0, ?)
3920
+ `);
3921
+ db.run("BEGIN TRANSACTION;");
3922
+ try {
3923
+ db.run("DELETE FROM symbols WHERE kind LIKE 'SCIP_%'");
3924
+ db.run("DELETE FROM \"references\" WHERE kind LIKE 'scip_%'");
3925
+ db.run("DELETE FROM dependencies WHERE kind = 'scip_import'");
3926
+ for (const symbol of facts.symbols) {
3927
+ insertSymbol.run(symbol.name, symbol.kind, symbol.filePath, symbol.startLine, symbol.endLine, symbol.startCol, symbol.endCol, JSON.stringify({ provenance: "scip", symbol: symbol.symbol }));
3928
+ }
3929
+ const dependencies = new Set();
3930
+ for (const reference of facts.references) {
3931
+ insertReference.run(reference.callerSymbol, reference.callerFile, reference.calleeSymbol, reference.calleeFile, reference.line, reference.column, reference.kind);
3932
+ if (reference.kind === "scip_reference" && reference.callerFile !== reference.calleeFile) {
3933
+ const key = `${reference.callerFile}\0${reference.calleeFile}`;
3934
+ if (!dependencies.has(key)) {
3935
+ dependencies.add(key);
3936
+ insertDependency.run(reference.callerFile, reference.calleeFile, JSON.stringify({ provenance: "scip", input: facts.inputPath }));
3937
+ }
3938
+ }
3939
+ }
3940
+ db.run("COMMIT;");
3941
+ }
3942
+ catch (error) {
3943
+ db.run("ROLLBACK;");
3944
+ throw error;
3945
+ }
3946
+ finally {
3947
+ insertSymbol.finalize();
3948
+ insertReference.finalize();
3949
+ insertDependency.finalize();
3950
+ }
3951
+ persistSymbolIdentities(db, repoPath, affectedFiles);
3952
+ return affectedFiles;
3953
+ }
3954
+ /** Any source refresh makes the previous compiler snapshot stale as a unit. */
3955
+ function invalidateScipFacts(db) {
3956
+ const present = db
3957
+ .query("SELECT 1 AS present FROM symbols WHERE kind LIKE 'SCIP_%' LIMIT 1")
3958
+ .get();
3959
+ if (!present)
3960
+ return;
3961
+ db.run("BEGIN TRANSACTION;");
3962
+ try {
3963
+ db.run("DELETE FROM symbols WHERE kind LIKE 'SCIP_%'");
3964
+ db.run("DELETE FROM \"references\" WHERE kind LIKE 'scip_%'");
3965
+ db.run("DELETE FROM dependencies WHERE kind = 'scip_import'");
3966
+ db.run("COMMIT;");
3967
+ }
3968
+ catch (error) {
3969
+ db.run("ROLLBACK;");
3970
+ throw error;
3971
+ }
3972
+ }
3903
3973
  async function indexTerraformFile(content, relativePath, _repoPath, db) {
3904
3974
  const symbols = [];
3905
3975
  const references = [];
@@ -5014,6 +5084,7 @@ export async function indexDbtManifestFile(content, _absolutePath, relativePath,
5014
5084
  /** Index a single file's symbols and references into the SQLite database. */
5015
5085
  async function indexFile(absolutePath, relativePath, repoPath, db) {
5016
5086
  try {
5087
+ invalidateScipFacts(db);
5017
5088
  const baseName = path.basename(absolutePath).toLowerCase();
5018
5089
  if (baseName === "manifest.json") {
5019
5090
  const content = fs.readFileSync(absolutePath, "utf-8");
@@ -6112,7 +6183,16 @@ function noteFreshness(repoPath, staleness) {
6112
6183
  async function ensureIndexFresh(repoPath, db) {
6113
6184
  const key = path.resolve(repoPath);
6114
6185
  const cached = freshnessProbes.get(key);
6115
- const lease = watchQueues.get(key)?.ready ? WATCHED_FRESHNESS_LEASE_MS : FRESHNESS_PROBE_TTL_MS;
6186
+ // A watcher cannot prove that an edit made immediately before this call has
6187
+ // already reached its event queue. Git worktrees therefore take the bounded
6188
+ // porcelain proof on every answer. The long watcher lease remains safe for
6189
+ // non-Git directories, where the watcher is the only new-file signal.
6190
+ const gitBacked = fs.existsSync(path.join(key, ".git"));
6191
+ const lease = gitBacked
6192
+ ? 0
6193
+ : watchQueues.get(key)?.ready
6194
+ ? WATCHED_FRESHNESS_LEASE_MS
6195
+ : FRESHNESS_PROBE_TTL_MS;
6116
6196
  const head = lease === WATCHED_FRESHNESS_LEASE_MS ? readGitHeadFast(key) : null;
6117
6197
  const headUnchanged = head === null || (head !== undefined && head === getMeta(db, "lastIndexedHead"));
6118
6198
  if (cached && Date.now() - cached.at < lease && headUnchanged) {
@@ -6949,6 +7029,41 @@ function queryFileSymbols(db, filePath, repoPath) {
6949
7029
  };
6950
7030
  });
6951
7031
  }
7032
+ function publishStructuralSnapshot(repoPath, db) {
7033
+ const indexed = db
7034
+ .query("SELECT filePath FROM index_state ORDER BY filePath")
7035
+ .all();
7036
+ const files = [];
7037
+ for (const { filePath } of indexed) {
7038
+ const absolute = path.resolve(repoPath, filePath);
7039
+ try {
7040
+ const content = fs.readFileSync(absolute);
7041
+ const stat = fs.statSync(absolute);
7042
+ files.push({
7043
+ path: filePath,
7044
+ sizeBytes: content.byteLength,
7045
+ mtimeMs: stat.mtimeMs,
7046
+ contentFingerprint: contentFingerprint(content),
7047
+ symbols: queryFileSymbols(db, filePath, repoPath).map((row) => ({
7048
+ identity: row.identity ?? null,
7049
+ symbol: row.symbol ?? "",
7050
+ kind: row.kind ?? "unknown",
7051
+ line: row.line ?? 1,
7052
+ endLine: row.endLine ?? row.line ?? 1,
7053
+ ...(row.signature ? { signature: row.signature } : {}),
7054
+ ...(row.visibility ? { visibility: row.visibility } : {}),
7055
+ ...(row.exported !== undefined ? { exported: row.exported } : {}),
7056
+ ...(row.parent ? { parent: row.parent } : {}),
7057
+ evidenceQuality: row.evidenceQuality ?? "parser-grounded",
7058
+ })),
7059
+ });
7060
+ }
7061
+ catch {
7062
+ // Health/freshness reports missing files; snapshots never invent an entry.
7063
+ }
7064
+ }
7065
+ writeStructuralSnapshot(repoPath, files);
7066
+ }
6952
7067
  /**
6953
7068
  * Name-level call adjacency from the `references` table — the single SQL +
6954
7069
  * Map-building shared by `shortest_path` (directed) and `traverse` (undirected),
@@ -7643,43 +7758,16 @@ export function calculateASTComplexity(node) {
7643
7758
  traverse(node);
7644
7759
  return score;
7645
7760
  }
7646
- const gitChurnCache = new Map();
7647
- const GIT_CHURN_CACHE_LIMIT = 8;
7648
7761
  /**
7649
7762
  * Count commits touching requested files with the legacy path-specific Git
7650
7763
  * semantics. Results are cached by resolved repository + HEAD, so warm review
7651
7764
  * launches no per-file history processes while preserving existing risk scores.
7652
7765
  */
7653
7766
  export function getGitChurns(filePaths, repoPath) {
7654
- const resolvedRepoPath = path.resolve(repoPath);
7655
7767
  const requested = [...new Set(filePaths.map((file) => safeReviewFile(file)))];
7656
- let head = "unborn";
7657
- try {
7658
- head = runGit(resolvedRepoPath, ["rev-parse", "--verify", "HEAD"]).trim();
7659
- }
7660
- catch {
7661
- // An unborn repository has no history, so every requested file has zero churn.
7662
- }
7663
- const cacheKey = `${resolvedRepoPath}\0${head}`;
7664
- let counts = gitChurnCache.get(cacheKey);
7665
- if (!counts) {
7666
- counts = new Map();
7667
- setBoundedCache(gitChurnCache, cacheKey, counts, GIT_CHURN_CACHE_LIMIT);
7668
- }
7669
- if (head !== "unborn") {
7670
- for (const file of requested) {
7671
- if (counts.has(file))
7672
- continue;
7673
- try {
7674
- const output = runGit(resolvedRepoPath, ["log", "--oneline", "--", file]).trim();
7675
- counts.set(file, output ? output.split("\n").length : 0);
7676
- }
7677
- catch {
7678
- counts.set(file, 0);
7679
- }
7680
- }
7681
- }
7682
- return new Map(requested.map((file) => [file, counts?.get(file) ?? 0]));
7768
+ const history = collectGitHistorySignals(repoPath, requested);
7769
+ const counts = new Map(history.churn.map((fact) => [fact.file, fact.commitCount]));
7770
+ return new Map(requested.map((file) => [file, counts.get(file) ?? 0]));
7683
7771
  }
7684
7772
  export function getGitChurn(filePath, repoPath) {
7685
7773
  return getGitChurns([filePath], repoPath).get(safeReviewFile(filePath)) ?? 0;
@@ -9097,10 +9185,12 @@ export function createEngine() {
9097
9185
  changedFiles: reviewChangedFiles(base, repoPath, options, modifiedFiles),
9098
9186
  };
9099
9187
  });
9100
- const churnByFile = measurePerfPhaseSync("review_churn", () => getGitChurns(changedFiles, repoPath));
9188
+ const historySignals = measurePerfPhaseSync("review_churn", () => collectGitHistorySignals(repoPath, changedFiles, options.historyLimits));
9189
+ const churnByFile = new Map(historySignals.churn.map((fact) => [fact.file, fact.commitCount]));
9101
9190
  const changedSymbols = new Set();
9102
9191
  const testGaps = new Set();
9103
9192
  const symbolRisks = [];
9193
+ const structuralCentrality = new Map();
9104
9194
  const finishParsing = beginPerfPhase("review_parsing");
9105
9195
  try {
9106
9196
  for (const filePath of changedFiles) {
@@ -9143,6 +9233,19 @@ export function createEngine() {
9143
9233
  Array.from(lineSet).some((line) => line >= sym.startLine && line <= sym.endLine);
9144
9234
  if (isChanged) {
9145
9235
  changedSymbols.add(sym.name);
9236
+ const inboundReferences = db
9237
+ .query('SELECT COUNT(*) AS count FROM "references" WHERE calleeSymbol = ? AND (calleeFile = ? OR calleeFile IS NULL)')
9238
+ .get(sym.name, filePath)?.count ?? 0;
9239
+ const outboundReferences = db
9240
+ .query('SELECT COUNT(*) AS count FROM "references" WHERE callerSymbol = ? AND callerFile = ?')
9241
+ .get(sym.name, filePath)?.count ?? 0;
9242
+ structuralCentrality.set(`${filePath}\0${sym.name}`, {
9243
+ symbol: sym.name,
9244
+ file: filePath,
9245
+ inboundReferences,
9246
+ outboundReferences,
9247
+ totalReferences: inboundReferences + outboundReferences,
9248
+ });
9146
9249
  let complexity = 1;
9147
9250
  if (tree) {
9148
9251
  try {
@@ -9214,11 +9317,18 @@ export function createEngine() {
9214
9317
  const unmappedChangedFiles = changedFiles.filter((file) => !mappedSet.has(file));
9215
9318
  const flowsList = Array.from(affectedFlowsSet);
9216
9319
  const gapsList = Array.from(testGaps);
9320
+ const centralityFacts = [...structuralCentrality.values()].sort((a, b) => b.totalReferences - a.totalReferences ||
9321
+ a.file.localeCompare(b.file) ||
9322
+ a.symbol.localeCompare(b.symbol));
9217
9323
  // Minimal detail caps each list so a client LLM isn't handed the full
9218
9324
  // (potentially very large) result; the *Count fields always report the
9219
9325
  // true totals so nothing is silently hidden.
9220
9326
  const MINIMAL_CAP = 25;
9221
9327
  const minimal = detailLevel === "minimal";
9328
+ const signalFiles = minimal ? changedFiles.slice(0, MINIMAL_CAP) : changedFiles;
9329
+ const signalFlows = minimal ? flowsList.slice(0, MINIMAL_CAP) : flowsList;
9330
+ const signalGaps = minimal ? gapsList.slice(0, MINIMAL_CAP) : gapsList;
9331
+ const signalCentrality = minimal ? centralityFacts.slice(0, MINIMAL_CAP) : centralityFacts;
9222
9332
  const truncated = minimal &&
9223
9333
  (changedFiles.length > MINIMAL_CAP ||
9224
9334
  mappedChangedFiles.length > MINIMAL_CAP ||
@@ -9238,6 +9348,22 @@ export function createEngine() {
9238
9348
  : unmappedChangedFiles,
9239
9349
  unmappedChangedFileCount: unmappedChangedFiles.length,
9240
9350
  riskScore,
9351
+ signals: {
9352
+ graphImpact: {
9353
+ affectedFlows: signalFlows,
9354
+ changedFiles: signalFiles,
9355
+ confidence: "exact-indexed-relationships",
9356
+ },
9357
+ testGaps: {
9358
+ facts: signalGaps,
9359
+ confidence: "indexed-references-and-test-paths",
9360
+ },
9361
+ structuralCentrality: {
9362
+ facts: signalCentrality,
9363
+ confidence: "exact-indexed-references",
9364
+ },
9365
+ history: historySignals,
9366
+ },
9241
9367
  changedSymbols: minimal ? changedList.slice(0, MINIMAL_CAP) : changedList,
9242
9368
  changedSymbolCount: changedList.length,
9243
9369
  affectedFlows: minimal ? flowsList.slice(0, MINIMAL_CAP) : flowsList,
@@ -10311,6 +10437,7 @@ export function createEngine() {
10311
10437
  async index(repoPath, files, clean = false, options) {
10312
10438
  const indexed = [];
10313
10439
  const unchanged = [];
10440
+ let scipReport;
10314
10441
  const progress = createIndexProgressReporter(options?.onProgress);
10315
10442
  progress("starting", 0, "Opening local graph database");
10316
10443
  const db = await getOrInitDb(repoPath, {
@@ -10396,6 +10523,33 @@ export function createEngine() {
10396
10523
  indexed.push(...collectRepoFiles(repoPath));
10397
10524
  completionMessage = `${indexed.length.toLocaleString()} repository file(s) are current`;
10398
10525
  }
10526
+ if (options?.scip) {
10527
+ const facts = readScipIndex(repoPath, options.scip);
10528
+ const affectedFiles = persistScipFacts(db, repoPath, facts);
10529
+ progress("finalizing", 0, "Refreshing SCIP identities and semantic embeddings");
10530
+ await indexEmbeddings(db, repoPath, progress);
10531
+ indexGeneration++;
10532
+ const bounds = { ...SCIP_DEFAULT_LIMITS, ...options.scip };
10533
+ scipReport = {
10534
+ provenance: "scip",
10535
+ inputPath: facts.inputPath,
10536
+ bytes: facts.bytes,
10537
+ files: facts.files,
10538
+ facts: facts.facts,
10539
+ languages: facts.languages,
10540
+ indexers: facts.indexers,
10541
+ elapsedMs: facts.elapsedMs,
10542
+ bounds: {
10543
+ maxBytes: bounds.maxBytes,
10544
+ maxFiles: bounds.maxFiles,
10545
+ maxFacts: bounds.maxFacts,
10546
+ timeoutMs: bounds.timeoutMs,
10547
+ },
10548
+ };
10549
+ for (const file of affectedFiles)
10550
+ if (!indexed.includes(file))
10551
+ indexed.push(file);
10552
+ }
10399
10553
  progress("verifying", 0, "Verifying local graph health", {
10400
10554
  phaseTotal: 1,
10401
10555
  });
@@ -10431,6 +10585,10 @@ export function createEngine() {
10431
10585
  recordFreshnessBaseline(repoPath, db);
10432
10586
  }
10433
10587
  noteFreshness(repoPath, "fresh");
10588
+ // Test engines use an in-memory graph and must not create untracked
10589
+ // repository artifacts that pollute Git-diff/review fixtures.
10590
+ if (!process.env.VITEST && process.env.NODE_ENV !== "test")
10591
+ publishStructuralSnapshot(repoPath, db);
10434
10592
  }
10435
10593
  else {
10436
10594
  freshnessProbes.delete(path.resolve(repoPath));
@@ -10445,6 +10603,7 @@ export function createEngine() {
10445
10603
  return {
10446
10604
  indexed,
10447
10605
  unchanged,
10606
+ scip: scipReport,
10448
10607
  verification: {
10449
10608
  status: health.status,
10450
10609
  issueCount,
@@ -10462,6 +10621,7 @@ export function createEngine() {
10462
10621
  throw new Error("knodin search: invalid testScope");
10463
10622
  const repos = options.federate === false ? [resolvedRepoPath] : await getFederatedRepos(repoPath);
10464
10623
  const allRepos = await Promise.all(repos.map(async (r) => ({ path: r, db: await getOrInitDb(r) })));
10624
+ const searchSnapshots = new Map(allRepos.map((repo) => [repo.path, getSearchMetadataSnapshot(repo.path, repo.db)]));
10465
10625
  // typeof + Number.isFinite rejects undefined/null/NaN alike, so any non-numeric
10466
10626
  // or unset `limit` falls back to the default instead of clamping to 0/1 or
10467
10627
  // propagating NaN into Array.prototype.slice (which silently empties results).
@@ -10503,12 +10663,90 @@ export function createEngine() {
10503
10663
  return true;
10504
10664
  };
10505
10665
  const formatPath = makePathFormatter(resolvedRepoPath);
10666
+ const retrievalMode = options.retrievalMode ?? "complete";
10667
+ const normalize = (value) => value.toLocaleLowerCase().replace(/[^a-z0-9]/g, "");
10668
+ const hydrateStructural = (matches) => {
10669
+ const fileToCommunityMap = matches.length > 0
10670
+ ? getSearchCommunityLookup(allRepos, resolvedRepoPath)
10671
+ : new Map();
10672
+ return matches.map(({ repo, row }) => {
10673
+ let source = "";
10674
+ if (options.includeSource !== false)
10675
+ source = getSourceRange(repo.path, row.filePath, row.startLine, row.endLine, 500);
10676
+ const callersStmt = repo.db.query(`
10677
+ SELECT DISTINCT callerSymbol FROM "references"
10678
+ WHERE calleeSymbol = ? AND callerSymbol IS NOT NULL AND callerSymbol != ?
10679
+ AND (calleeFile = ? OR calleeFile IS NULL)
10680
+ `);
10681
+ const directCallers = callersStmt
10682
+ .all(row.name, row.name, row.filePath)
10683
+ .map((caller) => caller.callerSymbol)
10684
+ .sort((a, b) => a.localeCompare(b));
10685
+ callersStmt.finalize();
10686
+ const formattedFile = formatPath(repo.path, row.filePath);
10687
+ return {
10688
+ identity: symbolIdentity(repo.path, row),
10689
+ symbol: row.name,
10690
+ kind: row.kind,
10691
+ filePath: formattedFile,
10692
+ similarity: 1,
10693
+ rrfScore: 1,
10694
+ source,
10695
+ belongsToCommunity: fileToCommunityMap.get(formattedFile) || "core-module",
10696
+ directCallers,
10697
+ };
10698
+ });
10699
+ };
10700
+ const expandBounded = (seeds) => {
10701
+ const expansionWorkLimit = Math.min(200, Math.max(candidateLimit * 4, 20));
10702
+ const selected = seeds.slice(0, expansionWorkLimit);
10703
+ const seen = new Set(selected.map(({ repo, row }) => `${repo.path}:${row.id}`));
10704
+ let frontier = [...seeds];
10705
+ for (let depth = 0; depth < 2 && frontier.length > 0 && selected.length < expansionWorkLimit; depth++) {
10706
+ const next = [];
10707
+ for (const current of frontier) {
10708
+ const refs = current.repo.db
10709
+ .query(`SELECT callerSymbol, callerFile, calleeSymbol, calleeFile FROM "references"
10710
+ WHERE (callerSymbol = ? AND callerFile = ?)
10711
+ OR (calleeSymbol = ? AND calleeFile = ?)
10712
+ ORDER BY callerFile, callerSymbol, calleeFile, calleeSymbol
10713
+ LIMIT 200`)
10714
+ .all(current.row.name, current.row.filePath, current.row.name, current.row.filePath);
10715
+ const snapshot = searchSnapshots.get(current.repo.path);
10716
+ for (const ref of refs) {
10717
+ const fromCaller = ref.callerSymbol === current.row.name && ref.callerFile === current.row.filePath;
10718
+ const neighborName = fromCaller ? ref.calleeSymbol : ref.callerSymbol;
10719
+ const neighborFile = fromCaller ? ref.calleeFile : ref.callerFile;
10720
+ if (!neighborName || !neighborFile)
10721
+ continue;
10722
+ for (const row of snapshot.rows) {
10723
+ if (row.name !== neighborName || !accepts(row))
10724
+ continue;
10725
+ if (neighborFile && row.filePath !== neighborFile)
10726
+ continue;
10727
+ const key = `${current.repo.path}:${row.id}`;
10728
+ if (seen.has(key))
10729
+ continue;
10730
+ if (seen.size >= expansionWorkLimit)
10731
+ break;
10732
+ seen.add(key);
10733
+ next.push({ repo: current.repo, row });
10734
+ }
10735
+ }
10736
+ }
10737
+ next.sort((a, b) => a.repo.path.localeCompare(b.repo.path) ||
10738
+ a.row.filePath.localeCompare(b.row.filePath) ||
10739
+ a.row.startLine - b.row.startLine);
10740
+ selected.push(...next.slice(0, expansionWorkLimit - selected.length));
10741
+ frontier = next;
10742
+ }
10743
+ return selected;
10744
+ };
10506
10745
  if (options.structural) {
10507
- const normalize = (value) => value.toLocaleLowerCase().replace(/[^a-z0-9]/g, "");
10508
10746
  const needle = normalize(query);
10509
10747
  const matches = allRepos
10510
- .flatMap((repo) => getSearchMetadataSnapshot(repo.path, repo.db)
10511
- .rows.filter((row) => accepts(row) && normalize(row.name).includes(needle))
10748
+ .flatMap((repo) => searchSnapshots.get(repo.path).rows
10749
+ .filter((row) => accepts(row) && normalize(row.name).includes(needle))
10512
10750
  .map((row) => ({ repo, row })))
10513
10751
  .sort((left, right) => left.row.name.localeCompare(right.row.name) ||
10514
10752
  left.row.filePath.localeCompare(right.row.filePath) ||
@@ -10532,8 +10770,140 @@ export function createEngine() {
10532
10770
  offset,
10533
10771
  limit: maxResults,
10534
10772
  hasMore: offset + maxResults < matches.length,
10773
+ retrieval: { route: ["exact-name"], embeddingsUsed: false },
10535
10774
  };
10536
10775
  }
10776
+ if (retrievalMode !== "vector-only") {
10777
+ const trimmedQuery = query.trim().replaceAll("\\", "/");
10778
+ const normalizedQuery = normalize(trimmedQuery);
10779
+ const identityQuery = trimmedQuery.startsWith("sym_") ? trimmedQuery : undefined;
10780
+ const pathQuery = trimmedQuery.includes("/") || /\.[A-Za-z0-9]+$/.test(trimmedQuery);
10781
+ const seedMatches = allRepos
10782
+ .flatMap((repo) => searchSnapshots.get(repo.path).rows
10783
+ .filter((row) => {
10784
+ if (!accepts(row))
10785
+ return false;
10786
+ if (identityQuery)
10787
+ return symbolIdentityMatches(symbolIdentity(repo.path, row), identityQuery);
10788
+ if (pathQuery) {
10789
+ const candidate = row.filePath.replaceAll("\\", "/");
10790
+ return candidate === trimmedQuery || candidate.endsWith(`/${trimmedQuery}`);
10791
+ }
10792
+ return normalizedQuery.length > 0 && normalize(row.name) === normalizedQuery;
10793
+ })
10794
+ .map((row) => ({ repo, row })))
10795
+ .sort((a, b) => a.repo.path.localeCompare(b.repo.path) ||
10796
+ a.row.filePath.localeCompare(b.row.filePath) ||
10797
+ a.row.startLine - b.row.startLine);
10798
+ if (seedMatches.length > 0) {
10799
+ const selected = retrievalMode === "complete" || retrievalMode === "bounded-graph"
10800
+ ? expandBounded(seedMatches)
10801
+ : seedMatches;
10802
+ const pageMatches = selected.slice(offset, offset + maxResults);
10803
+ return {
10804
+ results: hydrateStructural(pageMatches),
10805
+ total: selected.length,
10806
+ offset,
10807
+ limit: maxResults,
10808
+ hasMore: offset + maxResults < selected.length,
10809
+ retrieval: {
10810
+ route: retrievalMode === "complete" || retrievalMode === "bounded-graph"
10811
+ ? [
10812
+ identityQuery ? "stable-identity" : pathQuery ? "exact-path" : "exact-name",
10813
+ "bounded-graph",
10814
+ ]
10815
+ : [identityQuery ? "stable-identity" : pathQuery ? "exact-path" : "exact-name"],
10816
+ embeddingsUsed: false,
10817
+ },
10818
+ };
10819
+ }
10820
+ }
10821
+ const lexicalRanks = new Map();
10822
+ if (retrievalMode !== "vector-only") {
10823
+ for (const repo of allRepos) {
10824
+ const snapshot = searchSnapshots.get(repo.path);
10825
+ const allowedIds = new Set(snapshot.rows.filter(accepts).map((row) => row.id));
10826
+ const tokens = query
10827
+ .replace(/[^\w\s]/g, " ")
10828
+ .trim()
10829
+ .split(/\s+/)
10830
+ .filter((word) => word.length >= 4)
10831
+ .map((word) => `"${word}"`);
10832
+ const runFts = (separator) => {
10833
+ if (tokens.length === 0)
10834
+ return [];
10835
+ try {
10836
+ const statement = repo.db.query(`
10837
+ SELECT symbolId FROM symbols_fts
10838
+ WHERE symbols_fts MATCH ?
10839
+ ORDER BY bm25(symbols_fts), symbolId
10840
+ LIMIT 100
10841
+ `);
10842
+ const ids = statement
10843
+ .all(tokens.join(separator))
10844
+ .map(({ symbolId }) => symbolId)
10845
+ .filter((id) => allowedIds.has(id));
10846
+ statement.finalize();
10847
+ return ids;
10848
+ }
10849
+ catch (error) {
10850
+ console.error(`FTS query failed for "${query}":`, error);
10851
+ return [];
10852
+ }
10853
+ };
10854
+ const and = measurePerfPhaseSync("fts", () => runFts(" "));
10855
+ const or = tokens.length > 1 ? measurePerfPhaseSync("fts", () => runFts(" OR ")) : and;
10856
+ const lexicalSeeds = or
10857
+ .map((id) => snapshot.rowById.get(id))
10858
+ .filter((row) => row !== undefined)
10859
+ .map((row) => ({ repo, row }));
10860
+ const structural = expandBounded(lexicalSeeds)
10861
+ .filter(({ repo: matchRepo }) => matchRepo.path === repo.path)
10862
+ .map(({ row }) => row.id);
10863
+ const sufficientSeeds = and
10864
+ .map((id) => snapshot.rowById.get(id))
10865
+ .filter((row) => row !== undefined)
10866
+ .map((row) => ({ repo, row }));
10867
+ const sufficient = expandBounded(sufficientSeeds)
10868
+ .filter(({ repo: matchRepo }) => matchRepo.path === repo.path)
10869
+ .map(({ row }) => row.id);
10870
+ lexicalRanks.set(repo.path, { and, or, structural, sufficient });
10871
+ }
10872
+ const structurallySufficient = retrievalMode === "complete" &&
10873
+ !ann.annEnabled() &&
10874
+ [...lexicalRanks.values()].some((ranks) => ranks.and.length > 0);
10875
+ if (retrievalMode === "lexical-only" ||
10876
+ retrievalMode === "bounded-graph" ||
10877
+ structurallySufficient) {
10878
+ const matches = allRepos.flatMap((repo) => {
10879
+ const snapshot = searchSnapshots.get(repo.path);
10880
+ const ranks = lexicalRanks.get(repo.path);
10881
+ const ids = structurallySufficient
10882
+ ? ranks?.sufficient
10883
+ : retrievalMode === "bounded-graph"
10884
+ ? ranks?.structural
10885
+ : ranks?.or;
10886
+ return (ids ?? [])
10887
+ .map((id) => snapshot.rowById.get(id))
10888
+ .filter((row) => row !== undefined)
10889
+ .map((row) => ({ repo, row }));
10890
+ });
10891
+ const pageMatches = matches.slice(offset, offset + maxResults);
10892
+ return {
10893
+ results: hydrateStructural(pageMatches),
10894
+ total: matches.length,
10895
+ offset,
10896
+ limit: maxResults,
10897
+ hasMore: offset + maxResults < matches.length,
10898
+ retrieval: {
10899
+ route: retrievalMode === "bounded-graph" || structurallySufficient
10900
+ ? ["lexical", "bounded-graph"]
10901
+ : ["lexical"],
10902
+ embeddingsUsed: false,
10903
+ },
10904
+ };
10905
+ }
10906
+ }
10537
10907
  // 1. Generate query embedding once
10538
10908
  const queryVec = await generateMeasuredEmbedding(query, true);
10539
10909
  const rankedCandidates = [];
@@ -10542,7 +10912,7 @@ export function createEngine() {
10542
10912
  const db = repo.db;
10543
10913
  // 2. Reuse decoded generation-scoped metadata. Cheap metadata-only
10544
10914
  // filters run before any exact vector score is computed.
10545
- const snapshot = getSearchMetadataSnapshot(repo.path, db);
10915
+ const snapshot = searchSnapshots.get(repo.path);
10546
10916
  const allRows = snapshot.rows.filter(accepts);
10547
10917
  const allowedIds = new Set(allRows.map((row) => row.id));
10548
10918
  totalMatches += allowedIds.size;
@@ -10587,39 +10957,10 @@ export function createEngine() {
10587
10957
  const sortedSemantics = semanticMatches.sort((a, b) => b.score - a.score);
10588
10958
  const semanticScoresAreTied = sortedSemantics.length > 1 &&
10589
10959
  sortedSemantics.every((item) => item.score === sortedSemantics[0]?.score);
10590
- // 3. FTS Fallback / Hybrid scoring
10591
- const ftsMatches = measurePerfPhaseSync("fts", () => {
10592
- let matches = [];
10593
- try {
10594
- const cleanQuery = query.replace(/[^\w\s]/g, " ").trim();
10595
- if (cleanQuery) {
10596
- // Quote each token so FTS5 treats bare AND/OR/NOT/NEAR as literal search
10597
- // words instead of query-syntax operators (which would otherwise throw
10598
- // a MATCH syntax error for a query like "NOT authenticated").
10599
- const ftsQuery = cleanQuery
10600
- .split(/\s+/)
10601
- .filter((word) => word.length >= 4)
10602
- .map((word) => `"${word}"`)
10603
- // With a useful semantic ordering, retain the precise all-token
10604
- // FTS signal. When a deterministic/fallback embedder ties every
10605
- // candidate, permit partial lexical matches so connective words
10606
- // do not reduce ranking to arbitrary insertion order.
10607
- .join(semanticScoresAreTied ? " OR " : " ");
10608
- const ftsStmt = db.query(`
10609
- SELECT symbolId FROM symbols_fts WHERE symbols_fts MATCH ? ORDER BY bm25(symbols_fts) LIMIT 100
10610
- `);
10611
- matches = ftsStmt
10612
- .all(ftsQuery)
10613
- .map((m) => m.symbolId)
10614
- .filter((id) => allowedIds.has(id));
10615
- ftsStmt.finalize();
10616
- }
10617
- }
10618
- catch (err) {
10619
- console.error(`FTS query failed for "${query}":`, err);
10620
- }
10621
- return matches;
10622
- });
10960
+ // 3. Reuse the lexical and bounded-graph ranks computed before query inference.
10961
+ const lexical = lexicalRanks.get(repo.path);
10962
+ const baseLexical = semanticScoresAreTied ? lexical?.or : lexical?.and;
10963
+ const ftsMatches = [...new Set(baseLexical ?? [])].filter((id) => allowedIds.has(id));
10623
10964
  // Rank mappings for Reciprocal Rank Fusion (RRF)
10624
10965
  const semanticRankMap = new Map();
10625
10966
  const semanticById = new Map();
@@ -10708,6 +11049,12 @@ export function createEngine() {
10708
11049
  offset,
10709
11050
  limit: maxResults,
10710
11051
  hasMore: offset + maxResults < total,
11052
+ retrieval: {
11053
+ route: retrievalMode === "vector-only"
11054
+ ? ["embeddings"]
11055
+ : ["lexical", "bounded-graph", "embeddings"],
11056
+ embeddingsUsed: true,
11057
+ },
10711
11058
  };
10712
11059
  },
10713
11060
  async query(pattern, target, repoPath, to, limit, depth, detailLevel, selector = {}, impactOptions = {}, options = {}) {
@@ -13018,7 +13365,7 @@ export function createEngine() {
13018
13365
  statusCache.clear();
13019
13366
  flowsCache.clear();
13020
13367
  fileToFlowLookupCache.clear();
13021
- gitChurnCache.clear();
13368
+ clearGitHistorySignalCache();
13022
13369
  annIndexCache.clear();
13023
13370
  searchMetadataCache.clear();
13024
13371
  searchCommunityLookupCache.clear();