knodin 0.10.8 → 0.12.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -826,6 +826,97 @@ function freshnessMechanismFor(repoPath, policy) {
826
826
  // Bumped whenever any repo is (re)indexed, so cached community-detection results
827
827
  // below can be invalidated cheaply instead of recomputed on every call.
828
828
  export const KNODIN_SCHEMA_VERSION = 24;
829
+ /**
830
+ * Reciprocal Rank Fusion's smoothing constant. Was written as a bare `60` at
831
+ * each use site; named so the ranking formula can be read and changed in one
832
+ * place. Adjacent ranks differ by roughly 0.00026 at this k, which is the unit
833
+ * every other ranking weight has to be reasoned about in.
834
+ */
835
+ const RRF_K = 60;
836
+ /**
837
+ * How hard graph centrality pulls on ranking, as a fraction of one full RRF
838
+ * component.
839
+ *
840
+ * Semantic similarity alone ranks a minutes-old duplicate above the
841
+ * implementation it duplicates: the duplicate is short and made almost entirely
842
+ * of the query's own words, while the canonical version is diluted by a real
843
+ * body. The searcher then reads their own new leaf as evidence that nothing
844
+ * existed (KNODIN-15). Community and degree data was already in the graph and
845
+ * already in the output; it simply was not weighted into ranking.
846
+ *
847
+ * 0.05 caps the boost at `0.05 / (RRF_K + 1)` ≈ 0.00082, about three rank
848
+ * positions. That is deliberately modest, because the opposite failure is
849
+ * equally real and lands on the same symbols: a *seam* is by definition a
850
+ * well-documented symbol with no callers, and those are already the hardest
851
+ * things to retrieve (KNODIN-28, KNODIN-30). A large centrality weight would
852
+ * bury them to fix this, trading one silent miss for another. So a connected
853
+ * implementation climbs past neighbours it is scoring near, and does not
854
+ * override a decisively stronger semantic match.
855
+ *
856
+ * Zero-in-degree symbols are excluded from the centrality ranking rather than
857
+ * ranked last, so an uncalled symbol is never *penalised* — it only forgoes a
858
+ * boost it has no evidence for.
859
+ */
860
+ const CENTRALITY_RRF_WEIGHT = 0.05;
861
+ /**
862
+ * The live ranking weights, exported so a benchmark can record which values
863
+ * produced its numbers.
864
+ *
865
+ * Comparing two ranking runs whose constants differed is not a measurement, and
866
+ * the difference is invisible unless the run writes them down. Reading them here
867
+ * rather than copying them into the benchmark keeps that record from going stale
868
+ * the first time someone tunes a weight.
869
+ */
870
+ /**
871
+ * How hard a partial (OR) lexical match pulls, as a fraction of one full RRF
872
+ * component, and how many of them are considered.
873
+ *
874
+ * The weight buys a statable property: at 0.75 a top-ranked partial match
875
+ * contributes `0.75 / (RRF_K + 1)` ≈ 0.0123, which is enough to lift a symbol
876
+ * sitting around semantic rank 90 above an unsupported semantic rank 1. So
877
+ * lexical evidence can rescue a symbol the embedder under-ranked, and cannot
878
+ * resurrect one the embedder judged irrelevant. At 1.0 the loose channel becomes
879
+ * a full peer of semantic similarity and wins from arbitrarily deep, which is
880
+ * over-correction.
881
+ *
882
+ * The limit is a separate lever and conflating the two is how this gets
883
+ * mistuned. RRF has a floor — the hundredth partial match still scores
884
+ * `w / (RRF_K + 100)` — so an unbounded channel gives a meaningful shove to a
885
+ * hundred symbols at once on any query full of common words. Truncating raises
886
+ * the per-item floor slightly while applying it to under a third as many rows,
887
+ * which is the trade that actually reduces noise.
888
+ *
889
+ * Both are starting points measured against `benchmarks/evaluations/c103-prose-retrieval`,
890
+ * not settled constants. Re-run it before changing either.
891
+ */
892
+ const LOOSE_LEXICAL_RRF_WEIGHT = 0.75;
893
+ /**
894
+ * How many query words the quorum check will probe, one bounded FTS query each.
895
+ *
896
+ * The check exists because a single incidental token match is not evidence, and
897
+ * this index cannot tell the difference on its own: FTS5 here has no stemming
898
+ * and does not split identifiers, so "validations" misses "validation" and
899
+ * `SessionAuthenticator` is one opaque token. The cap keeps a pathologically
900
+ * long query from turning into a pathological number of probes; beyond it the
901
+ * leading words decide, which is where the discriminating terms usually are.
902
+ */
903
+ const LEXICAL_QUORUM_TOKEN_CAP = 8;
904
+ const LOOSE_LEXICAL_LIMIT = 30;
905
+ /**
906
+ * The live ranking weights, exported so a benchmark can record which values
907
+ * produced its numbers.
908
+ *
909
+ * Comparing two ranking runs whose constants differed is not a measurement, and
910
+ * the difference is invisible unless the run writes them down. Reading them here
911
+ * rather than copying them into the benchmark keeps that record from going stale
912
+ * the first time someone tunes a weight.
913
+ */
914
+ export const RANKING_CONSTANTS = {
915
+ rrfK: RRF_K,
916
+ centralityWeight: CENTRALITY_RRF_WEIGHT,
917
+ looseLexicalWeight: LOOSE_LEXICAL_RRF_WEIGHT,
918
+ looseLexicalLimit: LOOSE_LEXICAL_LIMIT,
919
+ };
829
920
  const LAST_REBUILD_SCHEMA_VERSION = 24;
830
921
  let indexGeneration = 0;
831
922
  const resourceReachabilityCache = new Map();
@@ -1631,12 +1722,48 @@ function createLanguageMemo() {
1631
1722
  return name;
1632
1723
  };
1633
1724
  }
1725
+ /**
1726
+ * Node types a `//` line can carry across the grammars this indexes. They
1727
+ * disagree on the name, so matching one of them is not enough on its own — see
1728
+ * `isLineComment` below, which also requires the `//` prefix.
1729
+ */
1730
+ const LINE_COMMENT_NODE_TYPES = new Set(["comment", "line_comment", "hash_comment"]);
1634
1731
  /** Clean up and format comment/docstring blocks in JavaScript/TypeScript. */
1635
1732
  function getPrecedingComment(node) {
1636
1733
  let target = node;
1637
1734
  while (target.parent && target.parent.startPosition.row === target.startPosition.row) {
1638
1735
  target = target.parent;
1639
1736
  }
1737
+ // Only decorators and modifiers may sit between a doc comment and the thing
1738
+ // it documents. The walk used to step over ANY five siblings looking for a
1739
+ // comment, so a symbol with no doc comment of its own adopted whichever
1740
+ // comment preceded a nearby earlier symbol — attaching prose to a symbol it
1741
+ // was never written about, in both the embedding input and the FTS `summary`
1742
+ // column.
1743
+ //
1744
+ // That is not a corner case. Measured on this repository before the fix, 857
1745
+ // of 1,329 documented symbols under `src/` — 64% — shared summary text with
1746
+ // another symbol, and spot checks showed plain misattribution rather than
1747
+ // genuine repetition: an interface carrying a neighbouring function's
1748
+ // docstring, two unrelated types sharing a third symbol's sentence. Prose
1749
+ // retrieval matching that text then returns the wrong symbol confidently,
1750
+ // which is the same class of failure as the truncation above and strictly
1751
+ // worse: absent prose loses a hit, misattributed prose manufactures one.
1752
+ //
1753
+ // Stopping at the first non-skippable node is the whole fix. A row-adjacency
1754
+ // check was tried alongside it and removed: decorators are children of the
1755
+ // declaration in some grammars and siblings in others, so "directly above"
1756
+ // is not portable, and it silently dropped the doc comment of every
1757
+ // decorated class. The walk already refuses to cross a declaration, which is
1758
+ // what the misattribution needed.
1759
+ const SKIPPABLE_BEFORE_DOC = new Set([
1760
+ "decorator",
1761
+ "export",
1762
+ "default",
1763
+ "async",
1764
+ "abstract",
1765
+ "declare",
1766
+ ]);
1640
1767
  let prev = target.previousSibling;
1641
1768
  let count = 0;
1642
1769
  while (prev && count < 5) {
@@ -1646,6 +1773,8 @@ function getPrecedingComment(node) {
1646
1773
  prev.type === "line_comment") {
1647
1774
  break;
1648
1775
  }
1776
+ if (!SKIPPABLE_BEFORE_DOC.has(prev.type))
1777
+ return null;
1649
1778
  prev = prev.previousSibling;
1650
1779
  count++;
1651
1780
  }
@@ -1654,6 +1783,48 @@ function getPrecedingComment(node) {
1654
1783
  prev.type === "hash_comment" ||
1655
1784
  prev.type === "block_comment" ||
1656
1785
  prev.type === "line_comment")) {
1786
+ // Walk back over a run of consecutive `//` lines and rejoin them.
1787
+ //
1788
+ // tree-sitter gives every `//` line its OWN comment node, so `prev` is the
1789
+ // LAST line of a block, and the `split("\n")` below could never fire for
1790
+ // this style. Every multi-line `//` doc comment in every indexed
1791
+ // repository was therefore truncated to its final line before it reached
1792
+ // the embedding or the FTS index — silently, since a one-line summary
1793
+ // looks perfectly well-formed.
1794
+ //
1795
+ // That is a retrieval defect, not a cosmetic one: the first line of a doc
1796
+ // comment is usually the sentence that says what the thing is for, which
1797
+ // is exactly what a capability question matches on. `/* */` blocks are a
1798
+ // single node and were never affected, so the damage was invisible in any
1799
+ // codebase that preferred them.
1800
+ //
1801
+ // Adjacency is required: a blank line or any code between two comment
1802
+ // nodes ends the block, so a stray earlier comment is not absorbed.
1803
+ // Keyed on the `//` PREFIX rather than on one node type. The surrounding
1804
+ // checks accept `comment`, `line_comment`, `hash_comment` and
1805
+ // `block_comment` because grammars disagree about the name, and an
1806
+ // earlier version of this rejoin tested `type === "comment"` alone — so
1807
+ // in any grammar that calls a `//` line `line_comment`, multi-line
1808
+ // comments went on being truncated to their last line while appearing
1809
+ // fixed everywhere else. The prefix is the thing that actually decides
1810
+ // whether a node is one line of a multi-line run.
1811
+ const isLineComment = (node) => LINE_COMMENT_NODE_TYPES.has(node.type) && node.text.trim().startsWith("//");
1812
+ if (isLineComment(prev)) {
1813
+ const lines = [];
1814
+ let line = prev;
1815
+ while (line &&
1816
+ isLineComment(line) &&
1817
+ // Consecutive source lines, and nothing but whitespace between them.
1818
+ (lines.length === 0 || line.endPosition.row + 1 === prev.startPosition.row)) {
1819
+ lines.unshift(line.text.trim());
1820
+ prev = line;
1821
+ line = line.previousSibling;
1822
+ }
1823
+ return lines
1824
+ .map((entry) => entry.replace(/^\/\/+\s*/, ""))
1825
+ .join("\n")
1826
+ .trim();
1827
+ }
1657
1828
  let text = prev.text.trim();
1658
1829
  if (text.startsWith("/*")) {
1659
1830
  text = text
@@ -6988,6 +7159,34 @@ function repairPathsAffectTypeScriptDi(db, repoPath, repairPaths) {
6988
7159
  // measurements.
6989
7160
  const DEFAULT_EMBEDDING_BATCH_SIZE = 1;
6990
7161
  const MAX_EMBEDDING_BATCH_SIZE = 32;
7162
+ /**
7163
+ * Total characters embedded per symbol.
7164
+ *
7165
+ * A `EMBEDDING_SOURCE_BUDGET` capping the source body inside this was tried and
7166
+ * reverted. It looked like a win — the diluted implementation's similarity rose
7167
+ * 0.163 to 0.212 against its own fresh duplicate — but that was measured on the
7168
+ * suite's bag-of-words test embedder, and re-measuring on the real MiniLM model
7169
+ * across two body shapes, three query phrasings and both budgets showed the cap
7170
+ * never changes the winner in any of the twelve configurations. Its apparent
7171
+ * benefit came from the fixture: `fillerBody` emitted forty copies of one line,
7172
+ * and truncating identical lines removes a redundancy bag-of-words punishes and
7173
+ * a transformer largely ignores. On realistic varied code the cap is neutral to
7174
+ * slightly negative.
7175
+ *
7176
+ * The general lesson, which cost a forced re-embed to learn: a short document
7177
+ * made almost entirely of query tokens beats a long one under cosine
7178
+ * similarity, and no reweighting of a single vector's inputs changes that.
7179
+ * KNODIN-15's remedy is a non-similarity signal — graph degree, in the rank
7180
+ * fusion — not the representation.
7181
+ */
7182
+ const EMBEDDING_INPUT_BUDGET = 1000;
7183
+ /**
7184
+ * The text embedded for one symbol: what it is called, what kind of thing it
7185
+ * is, what its docstring says it does, and then its body.
7186
+ */
7187
+ function embeddingInputFor(symbol, source) {
7188
+ return `Name: ${symbol.name}\nKind: ${symbol.kind}\nSummary: ${symbol.summary || ""}\nSource:\n${source}`.slice(0, EMBEDDING_INPUT_BUDGET);
7189
+ }
6991
7190
  // A full embedding pass can touch many symbols per source file. Keep only a
6992
7191
  // bounded LRU of split lines so large repositories avoid re-reading a file for
6993
7192
  // every symbol without turning source preparation into an unbounded memory sink.
@@ -7182,10 +7381,7 @@ async function indexEmbeddings(db, repoPath, progress, signal) {
7182
7381
  throwIfAborted(signal);
7183
7382
  const prepared = missing.slice(offset, offset + batchSize).map((sym) => {
7184
7383
  const source = sourceForEmbedding(sym.filePath, sym.startLine, sym.endLine);
7185
- return {
7186
- symbol: sym,
7187
- text: `Name: ${sym.name}\nKind: ${sym.kind}\nSummary: ${sym.summary || ""}\nSource:\n${source}`.slice(0, 1000),
7188
- };
7384
+ return { symbol: sym, text: embeddingInputFor(sym, source) };
7189
7385
  });
7190
7386
  const generated = await embedPreparedBatch(prepared, onModelProgress);
7191
7387
  throwIfAborted(signal);
@@ -8004,50 +8200,6 @@ const freshnessChecks = new Map();
8004
8200
  * deterministic, rather than in wall-clock milliseconds, which is flaky.
8005
8201
  */
8006
8202
  const freshnessStats = { probes: 0, reconciles: 0, cacheHits: 0 };
8007
- /** Read HEAD without spawning Git. `null` means this is not a Git checkout;
8008
- * `undefined` means Git metadata exists but could not be resolved safely. */
8009
- function readGitHeadFast(repoPath) {
8010
- const marker = path.join(repoPath, ".git");
8011
- if (!fs.existsSync(marker))
8012
- return null;
8013
- try {
8014
- const markerStat = fs.statSync(marker);
8015
- const gitDir = markerStat.isDirectory()
8016
- ? marker
8017
- : path.resolve(repoPath, fs
8018
- .readFileSync(marker, "utf8")
8019
- .trim()
8020
- .replace(/^gitdir:\s*/, ""));
8021
- const head = fs.readFileSync(path.join(gitDir, "HEAD"), "utf8").trim();
8022
- if (/^[a-f0-9]{40}$/i.test(head))
8023
- return head.toLowerCase();
8024
- const refPrefix = "ref: ";
8025
- const ref = head.startsWith(refPrefix) ? head.slice(refPrefix.length).trim() : "";
8026
- if (!ref)
8027
- return undefined;
8028
- let commonDir = gitDir;
8029
- const commonMarker = path.join(gitDir, "commondir");
8030
- if (fs.existsSync(commonMarker))
8031
- commonDir = path.resolve(gitDir, fs.readFileSync(commonMarker, "utf8").trim());
8032
- for (const base of [gitDir, commonDir]) {
8033
- const loose = path.join(base, ref);
8034
- if (fs.existsSync(loose))
8035
- return fs.readFileSync(loose, "utf8").trim().toLowerCase();
8036
- }
8037
- const packed = path.join(commonDir, "packed-refs");
8038
- if (!fs.existsSync(packed))
8039
- return undefined;
8040
- return fs
8041
- .readFileSync(packed, "utf8")
8042
- .split("\n")
8043
- .find((line) => line.endsWith(` ${ref}`))
8044
- ?.split(" ", 1)[0]
8045
- ?.toLowerCase();
8046
- }
8047
- catch {
8048
- return undefined;
8049
- }
8050
- }
8051
8203
  /**
8052
8204
  * Paths the live watcher for `repoPath` currently has registered, flattened to
8053
8205
  * repo-relative form. Empty when no watcher is running (test mode, or after
@@ -8272,15 +8424,20 @@ async function ensureIndexFresh(repoPath, db) {
8272
8424
  // already reached its event queue. Git worktrees therefore take the bounded
8273
8425
  // porcelain proof on every answer. The long watcher lease remains safe for
8274
8426
  // non-Git directories, where the watcher is the only new-file signal.
8427
+ //
8428
+ // The lease used to be paired with a spawn-free HEAD read, so a leased
8429
+ // answer could still be invalidated by a commit. Once git-backed repos took
8430
+ // `lease = 0` that read became unreachable — it was called only when `.git`
8431
+ // was absent, which was its own first bail-out — and the term it fed was
8432
+ // inert. Both are gone (KNODIN-16); a lease is now reached only where there
8433
+ // is no HEAD to check.
8275
8434
  const gitBacked = fs.existsSync(path.join(key, ".git"));
8276
8435
  const lease = gitBacked
8277
8436
  ? 0
8278
8437
  : watchQueues.get(key)?.ready
8279
8438
  ? WATCHED_FRESHNESS_LEASE_MS
8280
8439
  : FRESHNESS_PROBE_TTL_MS;
8281
- const head = lease === WATCHED_FRESHNESS_LEASE_MS ? readGitHeadFast(key) : null;
8282
- const headUnchanged = head === null || (head !== undefined && head === getMeta(db, "lastIndexedHead"));
8283
- if (cached && Date.now() - cached.at < lease && headUnchanged) {
8440
+ if (cached && Date.now() - cached.at < lease) {
8284
8441
  freshnessStats.cacheHits++;
8285
8442
  return cached.staleness;
8286
8443
  }
@@ -14447,9 +14604,7 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
14447
14604
  .split(/\s+/)
14448
14605
  .filter((word) => word.length >= 4)
14449
14606
  .map((word) => `"${word}"`);
14450
- const runFts = (separator) => {
14451
- if (tokens.length === 0)
14452
- return [];
14607
+ const runFtsMatch = (match) => {
14453
14608
  try {
14454
14609
  const statement = repo.db.query(`
14455
14610
  SELECT symbolId FROM symbols_fts
@@ -14458,7 +14613,7 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
14458
14613
  LIMIT 100
14459
14614
  `);
14460
14615
  const ids = statement
14461
- .all(tokens.join(separator))
14616
+ .all(match)
14462
14617
  .map(({ symbolId }) => symbolId)
14463
14618
  .filter((id) => allowedIds.has(id));
14464
14619
  statement.finalize();
@@ -14469,8 +14624,43 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
14469
14624
  return [];
14470
14625
  }
14471
14626
  };
14627
+ const runFts = (separator) => tokens.length === 0 ? [] : runFtsMatch(tokens.join(separator));
14628
+ const runFtsTokens = (subset) => subset.length === 0 ? [] : runFtsMatch(subset.join(" OR "));
14472
14629
  const and = measurePerfPhaseSync("fts", () => runFts(" "));
14473
- const or = tokens.length > 1 ? measurePerfPhaseSync("fts", () => runFts(" OR ")) : and;
14630
+ // The OR set, restricted to candidates matching at least two of the
14631
+ // query's words.
14632
+ //
14633
+ // A single incidental token match is not evidence. FTS5 here has no
14634
+ // stemming and does not split identifiers, so for "session
14635
+ // authentication and token validations" the class that actually does
14636
+ // the work matches NOTHING — its docstring says "sessions" and
14637
+ // "validation" — while a handler whose docstring happens to contain
14638
+ // "session" matches once and collects the full partial-match credit.
14639
+ // Fusing that put the caller above the implementation it calls.
14640
+ //
14641
+ // Requiring a quorum keeps the signal that matters (a symbol echoing
14642
+ // several of the query's words) and drops the one that misleads. It
14643
+ // costs one bounded FTS query per token, which is why it is capped.
14644
+ const or = measurePerfPhaseSync("fts", () => {
14645
+ if (tokens.length < 2)
14646
+ return and;
14647
+ const loose = runFts(" OR ");
14648
+ if (loose.length === 0)
14649
+ return loose;
14650
+ const hits = new Map();
14651
+ for (const token of tokens.slice(0, LEXICAL_QUORUM_TOKEN_CAP)) {
14652
+ for (const id of runFtsTokens([token])) {
14653
+ hits.set(id, (hits.get(id) ?? 0) + 1);
14654
+ }
14655
+ }
14656
+ // No fallback when nothing clears the bar. An earlier version
14657
+ // returned the unfiltered set in that case, which reinstated the
14658
+ // exact failure the quorum exists to prevent — and did so
14659
+ // precisely in the small-corpus situation where one weak match is
14660
+ // most likely to be the only one. A lexical channel with nothing
14661
+ // trustworthy to say should say nothing.
14662
+ return loose.filter((id) => (hits.get(id) ?? 0) >= 2);
14663
+ });
14474
14664
  const lexicalSeeds = or
14475
14665
  .map((id) => snapshot.rowById.get(id))
14476
14666
  .filter((row) => row !== undefined)
@@ -14526,6 +14716,15 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
14526
14716
  // 1. Generate query embedding once
14527
14717
  const queryVec = await generateMeasuredEmbedding(query, true);
14528
14718
  const rankedCandidates = [];
14719
+ // Inbound reference counts for the centrality term. Computed once for the
14720
+ // federation rather than per repo, and only on the route that actually
14721
+ // fuses ranks — the structural and exact-match routes above return before
14722
+ // here and must keep their deterministic ordering.
14723
+ //
14724
+ // Not wrapped in a perf phase: the phase keys are a closed set that the
14725
+ // instrumentation contract asserts on, and adding one to measure a fix
14726
+ // would change what that contract describes.
14727
+ const referenceDegree = computeResolvedReferenceDegree(allRepos, formatPath);
14529
14728
  let totalMatches = 0;
14530
14729
  for (const repo of allRepos) {
14531
14730
  const db = repo.db;
@@ -14574,12 +14773,35 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
14574
14773
  }
14575
14774
  // Sort by semantic score descending
14576
14775
  const sortedSemantics = semanticMatches.sort((a, b) => b.score - a.score);
14577
- const semanticScoresAreTied = sortedSemantics.length > 1 &&
14578
- sortedSemantics.every((item) => item.score === sortedSemantics[0]?.score);
14579
14776
  // 3. Reuse the lexical and bounded-graph ranks computed before query inference.
14777
+ //
14778
+ // Both operators are fused, as separate channels. The selection used to
14779
+ // be `semanticScoresAreTied ? or : and`, and semantic scores are never
14780
+ // all bit-identical under a real embedding model — so the OR set was
14781
+ // computed on every query and then discarded on every query.
14782
+ //
14783
+ // That made the AND clause the entire lexical channel, and for a query
14784
+ // phrased as prose the AND clause is empty: it demands one symbol
14785
+ // containing every surviving query word, function words included.
14786
+ // Nothing matches, so `ftsPart` was zero for every candidate and
14787
+ // "hybrid search" was pure vector search with a centrality nudge. The
14788
+ // docstring's BM25 evidence existed the whole time, in the OR set that
14789
+ // was thrown away (KNODIN-30).
14790
+ //
14791
+ // Fusing both rather than falling back from one to the other avoids a
14792
+ // discontinuity triggered by corpus contents rather than query intent:
14793
+ // with a fallback, one incidental symbol matching every word would flip
14794
+ // the channel from a broad candidate set to a single row, silently.
14795
+ //
14796
+ // `and ⊆ or` by construction — same tokens, narrower operator — so an
14797
+ // AND match collects BOTH terms while an OR-only match collects one.
14798
+ // The precision bonus falls out of that containment exactly, with no
14799
+ // extra machinery to express or tune.
14580
14800
  const lexical = lexicalRanks.get(repo.path);
14581
- const baseLexical = semanticScoresAreTied ? lexical?.or : lexical?.and;
14582
- const ftsMatches = [...new Set(baseLexical ?? [])].filter((id) => allowedIds.has(id));
14801
+ const strictMatches = [...new Set(lexical?.and ?? [])].filter((id) => allowedIds.has(id));
14802
+ const looseMatches = [...new Set(lexical?.or ?? [])]
14803
+ .filter((id) => allowedIds.has(id))
14804
+ .slice(0, LOOSE_LEXICAL_LIMIT);
14583
14805
  // Rank mappings for Reciprocal Rank Fusion (RRF)
14584
14806
  const semanticRankMap = new Map();
14585
14807
  const semanticById = new Map();
@@ -14587,19 +14809,57 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
14587
14809
  semanticRankMap.set(item.id, idx + 1);
14588
14810
  semanticById.set(item.id, item);
14589
14811
  });
14590
- const ftsRankMap = new Map();
14591
- ftsMatches.forEach((id, idx) => {
14592
- ftsRankMap.set(id, idx + 1);
14812
+ const strictRankMap = new Map();
14813
+ strictMatches.forEach((id, idx) => {
14814
+ strictRankMap.set(id, idx + 1);
14593
14815
  });
14816
+ const looseRankMap = new Map();
14817
+ looseMatches.forEach((id, idx) => {
14818
+ looseRankMap.set(id, idx + 1);
14819
+ });
14820
+ // Rank the matched candidates that anything actually depends on, most
14821
+ // depended-upon first. Symbols with no inbound references are left out
14822
+ // entirely rather than ranked last — see CENTRALITY_RRF_WEIGHT.
14823
+ const centralityRankMap = new Map();
14824
+ {
14825
+ const connected = [];
14826
+ for (const id of new Set([
14827
+ ...semanticRankMap.keys(),
14828
+ ...strictRankMap.keys(),
14829
+ ...looseRankMap.keys(),
14830
+ ])) {
14831
+ const row = semanticById.get(id)?.row ?? snapshot.rowById.get(id);
14832
+ if (!row)
14833
+ continue;
14834
+ const inDegree = referenceDegree.get(`${formatPath(repo.path, row.filePath)}::${row.name}`)
14835
+ ?.inDegree ?? 0;
14836
+ if (inDegree > 0)
14837
+ connected.push({ id, inDegree });
14838
+ }
14839
+ connected.sort((a, b) => b.inDegree - a.inDegree || a.id - b.id);
14840
+ connected.forEach((item, idx) => {
14841
+ centralityRankMap.set(item.id, idx + 1);
14842
+ });
14843
+ }
14594
14844
  // Perform Reciprocal Rank Fusion (RRF)
14595
14845
  const topRrf = measurePerfPhaseSync("fusion", () => {
14596
- const allMatchedIds = new Set([...semanticRankMap.keys(), ...ftsRankMap.keys()]);
14846
+ const allMatchedIds = new Set([
14847
+ ...semanticRankMap.keys(),
14848
+ ...strictRankMap.keys(),
14849
+ ...looseRankMap.keys(),
14850
+ ]);
14597
14851
  const rrfResults = Array.from(allMatchedIds).map((id) => {
14598
14852
  const semRank = semanticRankMap.get(id);
14599
- const ftsRank = ftsRankMap.get(id);
14600
- const semPart = semRank !== undefined ? 1 / (60 + semRank) : 0;
14601
- const ftsPart = ftsRank !== undefined ? 1 / (60 + ftsRank) : 0;
14602
- return { id, rrfScore: semPart + ftsPart };
14853
+ const strictRank = strictRankMap.get(id);
14854
+ const looseRank = looseRankMap.get(id);
14855
+ const centralityRank = centralityRankMap.get(id);
14856
+ const semPart = semRank !== undefined ? 1 / (RRF_K + semRank) : 0;
14857
+ const strictPart = strictRank !== undefined ? 1 / (RRF_K + strictRank) : 0;
14858
+ const loosePart = looseRank !== undefined ? LOOSE_LEXICAL_RRF_WEIGHT * (1 / (RRF_K + looseRank)) : 0;
14859
+ const centralityPart = centralityRank !== undefined
14860
+ ? CENTRALITY_RRF_WEIGHT * (1 / (RRF_K + centralityRank))
14861
+ : 0;
14862
+ return { id, rrfScore: semPart + strictPart + loosePart + centralityPart };
14603
14863
  });
14604
14864
  // Sort by RRF score descending and take top N
14605
14865
  return rrfResults.sort((a, b) => b.rrfScore - a.rrfScore).slice(0, candidateLimit);
@@ -14775,6 +15035,7 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
14775
15035
  ...(impactOptions.mode !== "file" ? ["impact"] : []),
14776
15036
  ]);
14777
15037
  let selectedTarget;
15038
+ let targetResolution;
14778
15039
  if (target && symbolPatterns.has(pattern)) {
14779
15040
  const resolved = resolveSymbolRows(db, primary.path, target, selector);
14780
15041
  if (resolved.ambiguity)
@@ -14804,6 +15065,26 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
14804
15065
  ambiguity: resolved.ambiguity,
14805
15066
  };
14806
15067
  selectedTarget = resolved.selected;
15068
+ // Graph health and target resolution are separate axes. A healthy,
15069
+ // fresh graph answering `count: 0` for a misspelled symbol was
15070
+ // byte-identical to one answering for real dead code, and the
15071
+ // freshness stamp made the typo read as an authoritative negative
15072
+ // (KNODIN-28). Say which of the two happened.
15073
+ //
15074
+ // `resolveSymbolRows` only sees the primary database, so a symbol
15075
+ // defined in a federated repo must be checked before it is called
15076
+ // missing — otherwise cross-repo queries would report every target
15077
+ // as a typo.
15078
+ targetResolution = resolved.selected
15079
+ ? "resolved"
15080
+ : allRepos.some((r) => {
15081
+ const stmt = r.db.query("SELECT filePath FROM symbols WHERE name = ? LIMIT 1");
15082
+ const row = stmt.get(target);
15083
+ stmt.finalize();
15084
+ return row != null;
15085
+ })
15086
+ ? "resolved"
15087
+ : "not-found";
14807
15088
  }
14808
15089
  const finish = (rows, extra = {}) => {
14809
15090
  for (const row of rows) {
@@ -14822,6 +15103,7 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
14822
15103
  count: rows.length,
14823
15104
  results: rows.slice(0, cap),
14824
15105
  ...(rows.length > cap ? { hasMore: true } : {}),
15106
+ ...(targetResolution ? { targetResolution } : {}),
14825
15107
  ...extra,
14826
15108
  };
14827
15109
  };
@@ -15004,26 +15286,41 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
15004
15286
  const refs = pattern === "callers_of"
15005
15287
  ? findCallersFederated(allRepos, defRepo, targetName, 1, new Set(), defFile, selectedTarget?.identity ?? null)
15006
15288
  : findCalleesFederated(allRepos, defRepo, targetName, 1, new Set(), defFile, selectedTarget?.identity ?? null);
15007
- const seen = new Set();
15289
+ // One row per caller symbol, but carrying every call site it has.
15290
+ // Collapsing to the first line lost the rest and made `count`
15291
+ // read as a call-site count (KNODIN-29).
15292
+ const byCaller = new Map();
15008
15293
  const rows = [];
15294
+ let callSiteCount = 0;
15009
15295
  for (const r of refs) {
15010
15296
  const file = formatPath(r.repoPath || defRepo, r.filePath);
15011
15297
  const key = `${r.symbol}|${file}`;
15012
- if (seen.has(key))
15298
+ callSiteCount++;
15299
+ const existing = byCaller.get(key);
15300
+ if (existing) {
15301
+ if (r.lineNumber !== undefined && !existing.callSiteLines?.includes(r.lineNumber))
15302
+ existing.callSiteLines?.push(r.lineNumber);
15013
15303
  continue;
15014
- seen.add(key);
15304
+ }
15015
15305
  const rr = (r.repoPath ? allRepos.find((x) => x.path === r.repoPath) : primary)?.db
15016
15306
  .query("SELECT * FROM symbols WHERE name = ? AND filePath = ? LIMIT 1")
15017
15307
  .get(r.symbol, r.filePath);
15018
- rows.push({
15308
+ const row = {
15019
15309
  symbol: r.symbol,
15020
15310
  file,
15021
15311
  line: r.lineNumber,
15022
15312
  kind: r.kind ?? "call",
15313
+ callSiteLines: r.lineNumber !== undefined ? [r.lineNumber] : [],
15023
15314
  ...(rr ? { identity: symbolIdentity(r.repoPath || defRepo, rr) } : {}),
15024
- });
15315
+ };
15316
+ byCaller.set(key, row);
15317
+ rows.push(row);
15025
15318
  }
15026
- return finish(rows);
15319
+ for (const row of rows) {
15320
+ row.callSiteLines?.sort((a, b) => a - b);
15321
+ row.line = row.callSiteLines?.[0] ?? row.line;
15322
+ }
15323
+ return finish(rows, { callSiteCount });
15027
15324
  }
15028
15325
  case "imports_of": {
15029
15326
  const stmt = db.query("SELECT DISTINCT toFile, kind FROM dependencies WHERE fromFile = ?");
@@ -147,10 +147,27 @@ class PoolRun {
147
147
  return this.buckets.get(bestKey)?.pop() ?? null;
148
148
  }
149
149
  attach(slot) {
150
- slot.handle.onMessage((response) => this.onResponse(slot, response));
151
- slot.handle.onError((error) => this.onSlotFailure(slot, error.message));
152
- slot.handle.onExit((code) => {
153
- if (slot.inFlight)
150
+ // Bind these listeners to the handle they were registered on, not just to
151
+ // the slot. `PoolWorkerHandle` offers no listener removal, so a replaced
152
+ // worker's listeners stay live and still close over the slot — whose
153
+ // `.handle` is by then the replacement. A crashed worker's inevitable late
154
+ // `exit` therefore re-entered `onSlotFailure` and terminated its OWN
155
+ // replacement, cancelling the one documented retry and leaving the lane
156
+ // dead; on a one-worker pool that dropped the whole run to inline parsing
157
+ // (KNODIN-20). Mirrors the `pending.child !== child` guard the MCP
158
+ // supervisor already carries.
159
+ const handle = slot.handle;
160
+ const current = () => slot.handle === handle;
161
+ handle.onMessage((response) => {
162
+ if (current())
163
+ this.onResponse(slot, response);
164
+ });
165
+ handle.onError((error) => {
166
+ if (current())
167
+ this.onSlotFailure(slot, error.message);
168
+ });
169
+ handle.onExit((code) => {
170
+ if (current() && slot.inFlight)
154
171
  this.onSlotFailure(slot, `worker exited with code ${code}`);
155
172
  });
156
173
  }