knodin 0.10.8 → 0.12.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/bin/cli.js +41 -4
- package/dist/src/cli-model.js +1 -0
- package/dist/src/compact-structural.js +8 -3
- package/dist/src/credential-patterns.js +14 -1
- package/dist/src/engine/index.js +371 -74
- package/dist/src/engine/parse-pool.js +21 -4
- package/dist/src/engine/seal.js +122 -3
- package/dist/src/engine/sealed-open.js +125 -8
- package/dist/src/engine/text-matches.js +121 -79
- package/dist/src/graph-query-health.js +9 -3
- package/dist/src/init.js +109 -3
- package/dist/src/lifecycle-health.js +46 -3
- package/dist/src/mcp-worker-supervisor.js +10 -2
- package/dist/src/repair-progress.js +12 -0
- package/dist/src/structural-fast-path.js +64 -1
- package/dist/src/tools/knodin-tools.js +32 -9
- package/docs/releases/0.11.0.md +227 -0
- package/docs/releases/0.12.0.md +281 -0
- package/package.json +5 -2
package/dist/src/engine/index.js
CHANGED
|
@@ -826,6 +826,97 @@ function freshnessMechanismFor(repoPath, policy) {
|
|
|
826
826
|
// Bumped whenever any repo is (re)indexed, so cached community-detection results
|
|
827
827
|
// below can be invalidated cheaply instead of recomputed on every call.
|
|
828
828
|
export const KNODIN_SCHEMA_VERSION = 24;
|
|
829
|
+
/**
|
|
830
|
+
* Reciprocal Rank Fusion's smoothing constant. Was written as a bare `60` at
|
|
831
|
+
* each use site; named so the ranking formula can be read and changed in one
|
|
832
|
+
* place. Adjacent ranks differ by roughly 0.00026 at this k, which is the unit
|
|
833
|
+
* every other ranking weight has to be reasoned about in.
|
|
834
|
+
*/
|
|
835
|
+
const RRF_K = 60;
|
|
836
|
+
/**
|
|
837
|
+
* How hard graph centrality pulls on ranking, as a fraction of one full RRF
|
|
838
|
+
* component.
|
|
839
|
+
*
|
|
840
|
+
* Semantic similarity alone ranks a minutes-old duplicate above the
|
|
841
|
+
* implementation it duplicates: the duplicate is short and made almost entirely
|
|
842
|
+
* of the query's own words, while the canonical version is diluted by a real
|
|
843
|
+
* body. The searcher then reads their own new leaf as evidence that nothing
|
|
844
|
+
* existed (KNODIN-15). Community and degree data was already in the graph and
|
|
845
|
+
* already in the output; it simply was not weighted into ranking.
|
|
846
|
+
*
|
|
847
|
+
* 0.05 caps the boost at `0.05 / (RRF_K + 1)` ≈ 0.00082, about three rank
|
|
848
|
+
* positions. That is deliberately modest, because the opposite failure is
|
|
849
|
+
* equally real and lands on the same symbols: a *seam* is by definition a
|
|
850
|
+
* well-documented symbol with no callers, and those are already the hardest
|
|
851
|
+
* things to retrieve (KNODIN-28, KNODIN-30). A large centrality weight would
|
|
852
|
+
* bury them to fix this, trading one silent miss for another. So a connected
|
|
853
|
+
* implementation climbs past neighbours it is scoring near, and does not
|
|
854
|
+
* override a decisively stronger semantic match.
|
|
855
|
+
*
|
|
856
|
+
* Zero-in-degree symbols are excluded from the centrality ranking rather than
|
|
857
|
+
* ranked last, so an uncalled symbol is never *penalised* — it only forgoes a
|
|
858
|
+
* boost it has no evidence for.
|
|
859
|
+
*/
|
|
860
|
+
const CENTRALITY_RRF_WEIGHT = 0.05;
|
|
861
|
+
/**
|
|
862
|
+
* The live ranking weights, exported so a benchmark can record which values
|
|
863
|
+
* produced its numbers.
|
|
864
|
+
*
|
|
865
|
+
* Comparing two ranking runs whose constants differed is not a measurement, and
|
|
866
|
+
* the difference is invisible unless the run writes them down. Reading them here
|
|
867
|
+
* rather than copying them into the benchmark keeps that record from going stale
|
|
868
|
+
* the first time someone tunes a weight.
|
|
869
|
+
*/
|
|
870
|
+
/**
|
|
871
|
+
* How hard a partial (OR) lexical match pulls, as a fraction of one full RRF
|
|
872
|
+
* component, and how many of them are considered.
|
|
873
|
+
*
|
|
874
|
+
* The weight buys a statable property: at 0.75 a top-ranked partial match
|
|
875
|
+
* contributes `0.75 / (RRF_K + 1)` ≈ 0.0123, which is enough to lift a symbol
|
|
876
|
+
* sitting around semantic rank 90 above an unsupported semantic rank 1. So
|
|
877
|
+
* lexical evidence can rescue a symbol the embedder under-ranked, and cannot
|
|
878
|
+
* resurrect one the embedder judged irrelevant. At 1.0 the loose channel becomes
|
|
879
|
+
* a full peer of semantic similarity and wins from arbitrarily deep, which is
|
|
880
|
+
* over-correction.
|
|
881
|
+
*
|
|
882
|
+
* The limit is a separate lever and conflating the two is how this gets
|
|
883
|
+
* mistuned. RRF has a floor — the hundredth partial match still scores
|
|
884
|
+
* `w / (RRF_K + 100)` — so an unbounded channel gives a meaningful shove to a
|
|
885
|
+
* hundred symbols at once on any query full of common words. Truncating raises
|
|
886
|
+
* the per-item floor slightly while applying it to under a third as many rows,
|
|
887
|
+
* which is the trade that actually reduces noise.
|
|
888
|
+
*
|
|
889
|
+
* Both are starting points measured against `benchmarks/evaluations/c103-prose-retrieval`,
|
|
890
|
+
* not settled constants. Re-run it before changing either.
|
|
891
|
+
*/
|
|
892
|
+
const LOOSE_LEXICAL_RRF_WEIGHT = 0.75;
|
|
893
|
+
/**
|
|
894
|
+
* How many query words the quorum check will probe, one bounded FTS query each.
|
|
895
|
+
*
|
|
896
|
+
* The check exists because a single incidental token match is not evidence, and
|
|
897
|
+
* this index cannot tell the difference on its own: FTS5 here has no stemming
|
|
898
|
+
* and does not split identifiers, so "validations" misses "validation" and
|
|
899
|
+
* `SessionAuthenticator` is one opaque token. The cap keeps a pathologically
|
|
900
|
+
* long query from turning into a pathological number of probes; beyond it the
|
|
901
|
+
* leading words decide, which is where the discriminating terms usually are.
|
|
902
|
+
*/
|
|
903
|
+
const LEXICAL_QUORUM_TOKEN_CAP = 8;
|
|
904
|
+
const LOOSE_LEXICAL_LIMIT = 30;
|
|
905
|
+
/**
|
|
906
|
+
* The live ranking weights, exported so a benchmark can record which values
|
|
907
|
+
* produced its numbers.
|
|
908
|
+
*
|
|
909
|
+
* Comparing two ranking runs whose constants differed is not a measurement, and
|
|
910
|
+
* the difference is invisible unless the run writes them down. Reading them here
|
|
911
|
+
* rather than copying them into the benchmark keeps that record from going stale
|
|
912
|
+
* the first time someone tunes a weight.
|
|
913
|
+
*/
|
|
914
|
+
export const RANKING_CONSTANTS = {
|
|
915
|
+
rrfK: RRF_K,
|
|
916
|
+
centralityWeight: CENTRALITY_RRF_WEIGHT,
|
|
917
|
+
looseLexicalWeight: LOOSE_LEXICAL_RRF_WEIGHT,
|
|
918
|
+
looseLexicalLimit: LOOSE_LEXICAL_LIMIT,
|
|
919
|
+
};
|
|
829
920
|
const LAST_REBUILD_SCHEMA_VERSION = 24;
|
|
830
921
|
let indexGeneration = 0;
|
|
831
922
|
const resourceReachabilityCache = new Map();
|
|
@@ -1631,12 +1722,48 @@ function createLanguageMemo() {
|
|
|
1631
1722
|
return name;
|
|
1632
1723
|
};
|
|
1633
1724
|
}
|
|
1725
|
+
/**
|
|
1726
|
+
* Node types a `//` line can carry across the grammars this indexes. They
|
|
1727
|
+
* disagree on the name, so matching one of them is not enough on its own — see
|
|
1728
|
+
* `isLineComment` below, which also requires the `//` prefix.
|
|
1729
|
+
*/
|
|
1730
|
+
const LINE_COMMENT_NODE_TYPES = new Set(["comment", "line_comment", "hash_comment"]);
|
|
1634
1731
|
/** Clean up and format comment/docstring blocks in JavaScript/TypeScript. */
|
|
1635
1732
|
function getPrecedingComment(node) {
|
|
1636
1733
|
let target = node;
|
|
1637
1734
|
while (target.parent && target.parent.startPosition.row === target.startPosition.row) {
|
|
1638
1735
|
target = target.parent;
|
|
1639
1736
|
}
|
|
1737
|
+
// Only decorators and modifiers may sit between a doc comment and the thing
|
|
1738
|
+
// it documents. The walk used to step over ANY five siblings looking for a
|
|
1739
|
+
// comment, so a symbol with no doc comment of its own adopted whichever
|
|
1740
|
+
// comment preceded a nearby earlier symbol — attaching prose to a symbol it
|
|
1741
|
+
// was never written about, in both the embedding input and the FTS `summary`
|
|
1742
|
+
// column.
|
|
1743
|
+
//
|
|
1744
|
+
// That is not a corner case. Measured on this repository before the fix, 857
|
|
1745
|
+
// of 1,329 documented symbols under `src/` — 64% — shared summary text with
|
|
1746
|
+
// another symbol, and spot checks showed plain misattribution rather than
|
|
1747
|
+
// genuine repetition: an interface carrying a neighbouring function's
|
|
1748
|
+
// docstring, two unrelated types sharing a third symbol's sentence. Prose
|
|
1749
|
+
// retrieval matching that text then returns the wrong symbol confidently,
|
|
1750
|
+
// which is the same class of failure as the truncation above and strictly
|
|
1751
|
+
// worse: absent prose loses a hit, misattributed prose manufactures one.
|
|
1752
|
+
//
|
|
1753
|
+
// Stopping at the first non-skippable node is the whole fix. A row-adjacency
|
|
1754
|
+
// check was tried alongside it and removed: decorators are children of the
|
|
1755
|
+
// declaration in some grammars and siblings in others, so "directly above"
|
|
1756
|
+
// is not portable, and it silently dropped the doc comment of every
|
|
1757
|
+
// decorated class. The walk already refuses to cross a declaration, which is
|
|
1758
|
+
// what the misattribution needed.
|
|
1759
|
+
const SKIPPABLE_BEFORE_DOC = new Set([
|
|
1760
|
+
"decorator",
|
|
1761
|
+
"export",
|
|
1762
|
+
"default",
|
|
1763
|
+
"async",
|
|
1764
|
+
"abstract",
|
|
1765
|
+
"declare",
|
|
1766
|
+
]);
|
|
1640
1767
|
let prev = target.previousSibling;
|
|
1641
1768
|
let count = 0;
|
|
1642
1769
|
while (prev && count < 5) {
|
|
@@ -1646,6 +1773,8 @@ function getPrecedingComment(node) {
|
|
|
1646
1773
|
prev.type === "line_comment") {
|
|
1647
1774
|
break;
|
|
1648
1775
|
}
|
|
1776
|
+
if (!SKIPPABLE_BEFORE_DOC.has(prev.type))
|
|
1777
|
+
return null;
|
|
1649
1778
|
prev = prev.previousSibling;
|
|
1650
1779
|
count++;
|
|
1651
1780
|
}
|
|
@@ -1654,6 +1783,48 @@ function getPrecedingComment(node) {
|
|
|
1654
1783
|
prev.type === "hash_comment" ||
|
|
1655
1784
|
prev.type === "block_comment" ||
|
|
1656
1785
|
prev.type === "line_comment")) {
|
|
1786
|
+
// Walk back over a run of consecutive `//` lines and rejoin them.
|
|
1787
|
+
//
|
|
1788
|
+
// tree-sitter gives every `//` line its OWN comment node, so `prev` is the
|
|
1789
|
+
// LAST line of a block, and the `split("\n")` below could never fire for
|
|
1790
|
+
// this style. Every multi-line `//` doc comment in every indexed
|
|
1791
|
+
// repository was therefore truncated to its final line before it reached
|
|
1792
|
+
// the embedding or the FTS index — silently, since a one-line summary
|
|
1793
|
+
// looks perfectly well-formed.
|
|
1794
|
+
//
|
|
1795
|
+
// That is a retrieval defect, not a cosmetic one: the first line of a doc
|
|
1796
|
+
// comment is usually the sentence that says what the thing is for, which
|
|
1797
|
+
// is exactly what a capability question matches on. `/* */` blocks are a
|
|
1798
|
+
// single node and were never affected, so the damage was invisible in any
|
|
1799
|
+
// codebase that preferred them.
|
|
1800
|
+
//
|
|
1801
|
+
// Adjacency is required: a blank line or any code between two comment
|
|
1802
|
+
// nodes ends the block, so a stray earlier comment is not absorbed.
|
|
1803
|
+
// Keyed on the `//` PREFIX rather than on one node type. The surrounding
|
|
1804
|
+
// checks accept `comment`, `line_comment`, `hash_comment` and
|
|
1805
|
+
// `block_comment` because grammars disagree about the name, and an
|
|
1806
|
+
// earlier version of this rejoin tested `type === "comment"` alone — so
|
|
1807
|
+
// in any grammar that calls a `//` line `line_comment`, multi-line
|
|
1808
|
+
// comments went on being truncated to their last line while appearing
|
|
1809
|
+
// fixed everywhere else. The prefix is the thing that actually decides
|
|
1810
|
+
// whether a node is one line of a multi-line run.
|
|
1811
|
+
const isLineComment = (node) => LINE_COMMENT_NODE_TYPES.has(node.type) && node.text.trim().startsWith("//");
|
|
1812
|
+
if (isLineComment(prev)) {
|
|
1813
|
+
const lines = [];
|
|
1814
|
+
let line = prev;
|
|
1815
|
+
while (line &&
|
|
1816
|
+
isLineComment(line) &&
|
|
1817
|
+
// Consecutive source lines, and nothing but whitespace between them.
|
|
1818
|
+
(lines.length === 0 || line.endPosition.row + 1 === prev.startPosition.row)) {
|
|
1819
|
+
lines.unshift(line.text.trim());
|
|
1820
|
+
prev = line;
|
|
1821
|
+
line = line.previousSibling;
|
|
1822
|
+
}
|
|
1823
|
+
return lines
|
|
1824
|
+
.map((entry) => entry.replace(/^\/\/+\s*/, ""))
|
|
1825
|
+
.join("\n")
|
|
1826
|
+
.trim();
|
|
1827
|
+
}
|
|
1657
1828
|
let text = prev.text.trim();
|
|
1658
1829
|
if (text.startsWith("/*")) {
|
|
1659
1830
|
text = text
|
|
@@ -6988,6 +7159,34 @@ function repairPathsAffectTypeScriptDi(db, repoPath, repairPaths) {
|
|
|
6988
7159
|
// measurements.
|
|
6989
7160
|
const DEFAULT_EMBEDDING_BATCH_SIZE = 1;
|
|
6990
7161
|
const MAX_EMBEDDING_BATCH_SIZE = 32;
|
|
7162
|
+
/**
|
|
7163
|
+
* Total characters embedded per symbol.
|
|
7164
|
+
*
|
|
7165
|
+
* A `EMBEDDING_SOURCE_BUDGET` capping the source body inside this was tried and
|
|
7166
|
+
* reverted. It looked like a win — the diluted implementation's similarity rose
|
|
7167
|
+
* 0.163 to 0.212 against its own fresh duplicate — but that was measured on the
|
|
7168
|
+
* suite's bag-of-words test embedder, and re-measuring on the real MiniLM model
|
|
7169
|
+
* across two body shapes, three query phrasings and both budgets showed the cap
|
|
7170
|
+
* never changes the winner in any of the twelve configurations. Its apparent
|
|
7171
|
+
* benefit came from the fixture: `fillerBody` emitted forty copies of one line,
|
|
7172
|
+
* and truncating identical lines removes a redundancy bag-of-words punishes and
|
|
7173
|
+
* a transformer largely ignores. On realistic varied code the cap is neutral to
|
|
7174
|
+
* slightly negative.
|
|
7175
|
+
*
|
|
7176
|
+
* The general lesson, which cost a forced re-embed to learn: a short document
|
|
7177
|
+
* made almost entirely of query tokens beats a long one under cosine
|
|
7178
|
+
* similarity, and no reweighting of a single vector's inputs changes that.
|
|
7179
|
+
* KNODIN-15's remedy is a non-similarity signal — graph degree, in the rank
|
|
7180
|
+
* fusion — not the representation.
|
|
7181
|
+
*/
|
|
7182
|
+
const EMBEDDING_INPUT_BUDGET = 1000;
|
|
7183
|
+
/**
|
|
7184
|
+
* The text embedded for one symbol: what it is called, what kind of thing it
|
|
7185
|
+
* is, what its docstring says it does, and then its body.
|
|
7186
|
+
*/
|
|
7187
|
+
function embeddingInputFor(symbol, source) {
|
|
7188
|
+
return `Name: ${symbol.name}\nKind: ${symbol.kind}\nSummary: ${symbol.summary || ""}\nSource:\n${source}`.slice(0, EMBEDDING_INPUT_BUDGET);
|
|
7189
|
+
}
|
|
6991
7190
|
// A full embedding pass can touch many symbols per source file. Keep only a
|
|
6992
7191
|
// bounded LRU of split lines so large repositories avoid re-reading a file for
|
|
6993
7192
|
// every symbol without turning source preparation into an unbounded memory sink.
|
|
@@ -7182,10 +7381,7 @@ async function indexEmbeddings(db, repoPath, progress, signal) {
|
|
|
7182
7381
|
throwIfAborted(signal);
|
|
7183
7382
|
const prepared = missing.slice(offset, offset + batchSize).map((sym) => {
|
|
7184
7383
|
const source = sourceForEmbedding(sym.filePath, sym.startLine, sym.endLine);
|
|
7185
|
-
return {
|
|
7186
|
-
symbol: sym,
|
|
7187
|
-
text: `Name: ${sym.name}\nKind: ${sym.kind}\nSummary: ${sym.summary || ""}\nSource:\n${source}`.slice(0, 1000),
|
|
7188
|
-
};
|
|
7384
|
+
return { symbol: sym, text: embeddingInputFor(sym, source) };
|
|
7189
7385
|
});
|
|
7190
7386
|
const generated = await embedPreparedBatch(prepared, onModelProgress);
|
|
7191
7387
|
throwIfAborted(signal);
|
|
@@ -8004,50 +8200,6 @@ const freshnessChecks = new Map();
|
|
|
8004
8200
|
* deterministic, rather than in wall-clock milliseconds, which is flaky.
|
|
8005
8201
|
*/
|
|
8006
8202
|
const freshnessStats = { probes: 0, reconciles: 0, cacheHits: 0 };
|
|
8007
|
-
/** Read HEAD without spawning Git. `null` means this is not a Git checkout;
|
|
8008
|
-
* `undefined` means Git metadata exists but could not be resolved safely. */
|
|
8009
|
-
function readGitHeadFast(repoPath) {
|
|
8010
|
-
const marker = path.join(repoPath, ".git");
|
|
8011
|
-
if (!fs.existsSync(marker))
|
|
8012
|
-
return null;
|
|
8013
|
-
try {
|
|
8014
|
-
const markerStat = fs.statSync(marker);
|
|
8015
|
-
const gitDir = markerStat.isDirectory()
|
|
8016
|
-
? marker
|
|
8017
|
-
: path.resolve(repoPath, fs
|
|
8018
|
-
.readFileSync(marker, "utf8")
|
|
8019
|
-
.trim()
|
|
8020
|
-
.replace(/^gitdir:\s*/, ""));
|
|
8021
|
-
const head = fs.readFileSync(path.join(gitDir, "HEAD"), "utf8").trim();
|
|
8022
|
-
if (/^[a-f0-9]{40}$/i.test(head))
|
|
8023
|
-
return head.toLowerCase();
|
|
8024
|
-
const refPrefix = "ref: ";
|
|
8025
|
-
const ref = head.startsWith(refPrefix) ? head.slice(refPrefix.length).trim() : "";
|
|
8026
|
-
if (!ref)
|
|
8027
|
-
return undefined;
|
|
8028
|
-
let commonDir = gitDir;
|
|
8029
|
-
const commonMarker = path.join(gitDir, "commondir");
|
|
8030
|
-
if (fs.existsSync(commonMarker))
|
|
8031
|
-
commonDir = path.resolve(gitDir, fs.readFileSync(commonMarker, "utf8").trim());
|
|
8032
|
-
for (const base of [gitDir, commonDir]) {
|
|
8033
|
-
const loose = path.join(base, ref);
|
|
8034
|
-
if (fs.existsSync(loose))
|
|
8035
|
-
return fs.readFileSync(loose, "utf8").trim().toLowerCase();
|
|
8036
|
-
}
|
|
8037
|
-
const packed = path.join(commonDir, "packed-refs");
|
|
8038
|
-
if (!fs.existsSync(packed))
|
|
8039
|
-
return undefined;
|
|
8040
|
-
return fs
|
|
8041
|
-
.readFileSync(packed, "utf8")
|
|
8042
|
-
.split("\n")
|
|
8043
|
-
.find((line) => line.endsWith(` ${ref}`))
|
|
8044
|
-
?.split(" ", 1)[0]
|
|
8045
|
-
?.toLowerCase();
|
|
8046
|
-
}
|
|
8047
|
-
catch {
|
|
8048
|
-
return undefined;
|
|
8049
|
-
}
|
|
8050
|
-
}
|
|
8051
8203
|
/**
|
|
8052
8204
|
* Paths the live watcher for `repoPath` currently has registered, flattened to
|
|
8053
8205
|
* repo-relative form. Empty when no watcher is running (test mode, or after
|
|
@@ -8272,15 +8424,20 @@ async function ensureIndexFresh(repoPath, db) {
|
|
|
8272
8424
|
// already reached its event queue. Git worktrees therefore take the bounded
|
|
8273
8425
|
// porcelain proof on every answer. The long watcher lease remains safe for
|
|
8274
8426
|
// non-Git directories, where the watcher is the only new-file signal.
|
|
8427
|
+
//
|
|
8428
|
+
// The lease used to be paired with a spawn-free HEAD read, so a leased
|
|
8429
|
+
// answer could still be invalidated by a commit. Once git-backed repos took
|
|
8430
|
+
// `lease = 0` that read became unreachable — it was called only when `.git`
|
|
8431
|
+
// was absent, which was its own first bail-out — and the term it fed was
|
|
8432
|
+
// inert. Both are gone (KNODIN-16); a lease is now reached only where there
|
|
8433
|
+
// is no HEAD to check.
|
|
8275
8434
|
const gitBacked = fs.existsSync(path.join(key, ".git"));
|
|
8276
8435
|
const lease = gitBacked
|
|
8277
8436
|
? 0
|
|
8278
8437
|
: watchQueues.get(key)?.ready
|
|
8279
8438
|
? WATCHED_FRESHNESS_LEASE_MS
|
|
8280
8439
|
: FRESHNESS_PROBE_TTL_MS;
|
|
8281
|
-
|
|
8282
|
-
const headUnchanged = head === null || (head !== undefined && head === getMeta(db, "lastIndexedHead"));
|
|
8283
|
-
if (cached && Date.now() - cached.at < lease && headUnchanged) {
|
|
8440
|
+
if (cached && Date.now() - cached.at < lease) {
|
|
8284
8441
|
freshnessStats.cacheHits++;
|
|
8285
8442
|
return cached.staleness;
|
|
8286
8443
|
}
|
|
@@ -14447,9 +14604,7 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
|
|
|
14447
14604
|
.split(/\s+/)
|
|
14448
14605
|
.filter((word) => word.length >= 4)
|
|
14449
14606
|
.map((word) => `"${word}"`);
|
|
14450
|
-
const
|
|
14451
|
-
if (tokens.length === 0)
|
|
14452
|
-
return [];
|
|
14607
|
+
const runFtsMatch = (match) => {
|
|
14453
14608
|
try {
|
|
14454
14609
|
const statement = repo.db.query(`
|
|
14455
14610
|
SELECT symbolId FROM symbols_fts
|
|
@@ -14458,7 +14613,7 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
|
|
|
14458
14613
|
LIMIT 100
|
|
14459
14614
|
`);
|
|
14460
14615
|
const ids = statement
|
|
14461
|
-
.all(
|
|
14616
|
+
.all(match)
|
|
14462
14617
|
.map(({ symbolId }) => symbolId)
|
|
14463
14618
|
.filter((id) => allowedIds.has(id));
|
|
14464
14619
|
statement.finalize();
|
|
@@ -14469,8 +14624,43 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
|
|
|
14469
14624
|
return [];
|
|
14470
14625
|
}
|
|
14471
14626
|
};
|
|
14627
|
+
const runFts = (separator) => tokens.length === 0 ? [] : runFtsMatch(tokens.join(separator));
|
|
14628
|
+
const runFtsTokens = (subset) => subset.length === 0 ? [] : runFtsMatch(subset.join(" OR "));
|
|
14472
14629
|
const and = measurePerfPhaseSync("fts", () => runFts(" "));
|
|
14473
|
-
|
|
14630
|
+
// The OR set, restricted to candidates matching at least two of the
|
|
14631
|
+
// query's words.
|
|
14632
|
+
//
|
|
14633
|
+
// A single incidental token match is not evidence. FTS5 here has no
|
|
14634
|
+
// stemming and does not split identifiers, so for "session
|
|
14635
|
+
// authentication and token validations" the class that actually does
|
|
14636
|
+
// the work matches NOTHING — its docstring says "sessions" and
|
|
14637
|
+
// "validation" — while a handler whose docstring happens to contain
|
|
14638
|
+
// "session" matches once and collects the full partial-match credit.
|
|
14639
|
+
// Fusing that put the caller above the implementation it calls.
|
|
14640
|
+
//
|
|
14641
|
+
// Requiring a quorum keeps the signal that matters (a symbol echoing
|
|
14642
|
+
// several of the query's words) and drops the one that misleads. It
|
|
14643
|
+
// costs one bounded FTS query per token, which is why it is capped.
|
|
14644
|
+
const or = measurePerfPhaseSync("fts", () => {
|
|
14645
|
+
if (tokens.length < 2)
|
|
14646
|
+
return and;
|
|
14647
|
+
const loose = runFts(" OR ");
|
|
14648
|
+
if (loose.length === 0)
|
|
14649
|
+
return loose;
|
|
14650
|
+
const hits = new Map();
|
|
14651
|
+
for (const token of tokens.slice(0, LEXICAL_QUORUM_TOKEN_CAP)) {
|
|
14652
|
+
for (const id of runFtsTokens([token])) {
|
|
14653
|
+
hits.set(id, (hits.get(id) ?? 0) + 1);
|
|
14654
|
+
}
|
|
14655
|
+
}
|
|
14656
|
+
// No fallback when nothing clears the bar. An earlier version
|
|
14657
|
+
// returned the unfiltered set in that case, which reinstated the
|
|
14658
|
+
// exact failure the quorum exists to prevent — and did so
|
|
14659
|
+
// precisely in the small-corpus situation where one weak match is
|
|
14660
|
+
// most likely to be the only one. A lexical channel with nothing
|
|
14661
|
+
// trustworthy to say should say nothing.
|
|
14662
|
+
return loose.filter((id) => (hits.get(id) ?? 0) >= 2);
|
|
14663
|
+
});
|
|
14474
14664
|
const lexicalSeeds = or
|
|
14475
14665
|
.map((id) => snapshot.rowById.get(id))
|
|
14476
14666
|
.filter((row) => row !== undefined)
|
|
@@ -14526,6 +14716,15 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
|
|
|
14526
14716
|
// 1. Generate query embedding once
|
|
14527
14717
|
const queryVec = await generateMeasuredEmbedding(query, true);
|
|
14528
14718
|
const rankedCandidates = [];
|
|
14719
|
+
// Inbound reference counts for the centrality term. Computed once for the
|
|
14720
|
+
// federation rather than per repo, and only on the route that actually
|
|
14721
|
+
// fuses ranks — the structural and exact-match routes above return before
|
|
14722
|
+
// here and must keep their deterministic ordering.
|
|
14723
|
+
//
|
|
14724
|
+
// Not wrapped in a perf phase: the phase keys are a closed set that the
|
|
14725
|
+
// instrumentation contract asserts on, and adding one to measure a fix
|
|
14726
|
+
// would change what that contract describes.
|
|
14727
|
+
const referenceDegree = computeResolvedReferenceDegree(allRepos, formatPath);
|
|
14529
14728
|
let totalMatches = 0;
|
|
14530
14729
|
for (const repo of allRepos) {
|
|
14531
14730
|
const db = repo.db;
|
|
@@ -14574,12 +14773,35 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
|
|
|
14574
14773
|
}
|
|
14575
14774
|
// Sort by semantic score descending
|
|
14576
14775
|
const sortedSemantics = semanticMatches.sort((a, b) => b.score - a.score);
|
|
14577
|
-
const semanticScoresAreTied = sortedSemantics.length > 1 &&
|
|
14578
|
-
sortedSemantics.every((item) => item.score === sortedSemantics[0]?.score);
|
|
14579
14776
|
// 3. Reuse the lexical and bounded-graph ranks computed before query inference.
|
|
14777
|
+
//
|
|
14778
|
+
// Both operators are fused, as separate channels. The selection used to
|
|
14779
|
+
// be `semanticScoresAreTied ? or : and`, and semantic scores are never
|
|
14780
|
+
// all bit-identical under a real embedding model — so the OR set was
|
|
14781
|
+
// computed on every query and then discarded on every query.
|
|
14782
|
+
//
|
|
14783
|
+
// That made the AND clause the entire lexical channel, and for a query
|
|
14784
|
+
// phrased as prose the AND clause is empty: it demands one symbol
|
|
14785
|
+
// containing every surviving query word, function words included.
|
|
14786
|
+
// Nothing matches, so `ftsPart` was zero for every candidate and
|
|
14787
|
+
// "hybrid search" was pure vector search with a centrality nudge. The
|
|
14788
|
+
// docstring's BM25 evidence existed the whole time, in the OR set that
|
|
14789
|
+
// was thrown away (KNODIN-30).
|
|
14790
|
+
//
|
|
14791
|
+
// Fusing both rather than falling back from one to the other avoids a
|
|
14792
|
+
// discontinuity triggered by corpus contents rather than query intent:
|
|
14793
|
+
// with a fallback, one incidental symbol matching every word would flip
|
|
14794
|
+
// the channel from a broad candidate set to a single row, silently.
|
|
14795
|
+
//
|
|
14796
|
+
// `and ⊆ or` by construction — same tokens, narrower operator — so an
|
|
14797
|
+
// AND match collects BOTH terms while an OR-only match collects one.
|
|
14798
|
+
// The precision bonus falls out of that containment exactly, with no
|
|
14799
|
+
// extra machinery to express or tune.
|
|
14580
14800
|
const lexical = lexicalRanks.get(repo.path);
|
|
14581
|
-
const
|
|
14582
|
-
const
|
|
14801
|
+
const strictMatches = [...new Set(lexical?.and ?? [])].filter((id) => allowedIds.has(id));
|
|
14802
|
+
const looseMatches = [...new Set(lexical?.or ?? [])]
|
|
14803
|
+
.filter((id) => allowedIds.has(id))
|
|
14804
|
+
.slice(0, LOOSE_LEXICAL_LIMIT);
|
|
14583
14805
|
// Rank mappings for Reciprocal Rank Fusion (RRF)
|
|
14584
14806
|
const semanticRankMap = new Map();
|
|
14585
14807
|
const semanticById = new Map();
|
|
@@ -14587,19 +14809,57 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
|
|
|
14587
14809
|
semanticRankMap.set(item.id, idx + 1);
|
|
14588
14810
|
semanticById.set(item.id, item);
|
|
14589
14811
|
});
|
|
14590
|
-
const
|
|
14591
|
-
|
|
14592
|
-
|
|
14812
|
+
const strictRankMap = new Map();
|
|
14813
|
+
strictMatches.forEach((id, idx) => {
|
|
14814
|
+
strictRankMap.set(id, idx + 1);
|
|
14593
14815
|
});
|
|
14816
|
+
const looseRankMap = new Map();
|
|
14817
|
+
looseMatches.forEach((id, idx) => {
|
|
14818
|
+
looseRankMap.set(id, idx + 1);
|
|
14819
|
+
});
|
|
14820
|
+
// Rank the matched candidates that anything actually depends on, most
|
|
14821
|
+
// depended-upon first. Symbols with no inbound references are left out
|
|
14822
|
+
// entirely rather than ranked last — see CENTRALITY_RRF_WEIGHT.
|
|
14823
|
+
const centralityRankMap = new Map();
|
|
14824
|
+
{
|
|
14825
|
+
const connected = [];
|
|
14826
|
+
for (const id of new Set([
|
|
14827
|
+
...semanticRankMap.keys(),
|
|
14828
|
+
...strictRankMap.keys(),
|
|
14829
|
+
...looseRankMap.keys(),
|
|
14830
|
+
])) {
|
|
14831
|
+
const row = semanticById.get(id)?.row ?? snapshot.rowById.get(id);
|
|
14832
|
+
if (!row)
|
|
14833
|
+
continue;
|
|
14834
|
+
const inDegree = referenceDegree.get(`${formatPath(repo.path, row.filePath)}::${row.name}`)
|
|
14835
|
+
?.inDegree ?? 0;
|
|
14836
|
+
if (inDegree > 0)
|
|
14837
|
+
connected.push({ id, inDegree });
|
|
14838
|
+
}
|
|
14839
|
+
connected.sort((a, b) => b.inDegree - a.inDegree || a.id - b.id);
|
|
14840
|
+
connected.forEach((item, idx) => {
|
|
14841
|
+
centralityRankMap.set(item.id, idx + 1);
|
|
14842
|
+
});
|
|
14843
|
+
}
|
|
14594
14844
|
// Perform Reciprocal Rank Fusion (RRF)
|
|
14595
14845
|
const topRrf = measurePerfPhaseSync("fusion", () => {
|
|
14596
|
-
const allMatchedIds = new Set([
|
|
14846
|
+
const allMatchedIds = new Set([
|
|
14847
|
+
...semanticRankMap.keys(),
|
|
14848
|
+
...strictRankMap.keys(),
|
|
14849
|
+
...looseRankMap.keys(),
|
|
14850
|
+
]);
|
|
14597
14851
|
const rrfResults = Array.from(allMatchedIds).map((id) => {
|
|
14598
14852
|
const semRank = semanticRankMap.get(id);
|
|
14599
|
-
const
|
|
14600
|
-
const
|
|
14601
|
-
const
|
|
14602
|
-
|
|
14853
|
+
const strictRank = strictRankMap.get(id);
|
|
14854
|
+
const looseRank = looseRankMap.get(id);
|
|
14855
|
+
const centralityRank = centralityRankMap.get(id);
|
|
14856
|
+
const semPart = semRank !== undefined ? 1 / (RRF_K + semRank) : 0;
|
|
14857
|
+
const strictPart = strictRank !== undefined ? 1 / (RRF_K + strictRank) : 0;
|
|
14858
|
+
const loosePart = looseRank !== undefined ? LOOSE_LEXICAL_RRF_WEIGHT * (1 / (RRF_K + looseRank)) : 0;
|
|
14859
|
+
const centralityPart = centralityRank !== undefined
|
|
14860
|
+
? CENTRALITY_RRF_WEIGHT * (1 / (RRF_K + centralityRank))
|
|
14861
|
+
: 0;
|
|
14862
|
+
return { id, rrfScore: semPart + strictPart + loosePart + centralityPart };
|
|
14603
14863
|
});
|
|
14604
14864
|
// Sort by RRF score descending and take top N
|
|
14605
14865
|
return rrfResults.sort((a, b) => b.rrfScore - a.rrfScore).slice(0, candidateLimit);
|
|
@@ -14775,6 +15035,7 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
|
|
|
14775
15035
|
...(impactOptions.mode !== "file" ? ["impact"] : []),
|
|
14776
15036
|
]);
|
|
14777
15037
|
let selectedTarget;
|
|
15038
|
+
let targetResolution;
|
|
14778
15039
|
if (target && symbolPatterns.has(pattern)) {
|
|
14779
15040
|
const resolved = resolveSymbolRows(db, primary.path, target, selector);
|
|
14780
15041
|
if (resolved.ambiguity)
|
|
@@ -14804,6 +15065,26 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
|
|
|
14804
15065
|
ambiguity: resolved.ambiguity,
|
|
14805
15066
|
};
|
|
14806
15067
|
selectedTarget = resolved.selected;
|
|
15068
|
+
// Graph health and target resolution are separate axes. A healthy,
|
|
15069
|
+
// fresh graph answering `count: 0` for a misspelled symbol was
|
|
15070
|
+
// byte-identical to one answering for real dead code, and the
|
|
15071
|
+
// freshness stamp made the typo read as an authoritative negative
|
|
15072
|
+
// (KNODIN-28). Say which of the two happened.
|
|
15073
|
+
//
|
|
15074
|
+
// `resolveSymbolRows` only sees the primary database, so a symbol
|
|
15075
|
+
// defined in a federated repo must be checked before it is called
|
|
15076
|
+
// missing — otherwise cross-repo queries would report every target
|
|
15077
|
+
// as a typo.
|
|
15078
|
+
targetResolution = resolved.selected
|
|
15079
|
+
? "resolved"
|
|
15080
|
+
: allRepos.some((r) => {
|
|
15081
|
+
const stmt = r.db.query("SELECT filePath FROM symbols WHERE name = ? LIMIT 1");
|
|
15082
|
+
const row = stmt.get(target);
|
|
15083
|
+
stmt.finalize();
|
|
15084
|
+
return row != null;
|
|
15085
|
+
})
|
|
15086
|
+
? "resolved"
|
|
15087
|
+
: "not-found";
|
|
14807
15088
|
}
|
|
14808
15089
|
const finish = (rows, extra = {}) => {
|
|
14809
15090
|
for (const row of rows) {
|
|
@@ -14822,6 +15103,7 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
|
|
|
14822
15103
|
count: rows.length,
|
|
14823
15104
|
results: rows.slice(0, cap),
|
|
14824
15105
|
...(rows.length > cap ? { hasMore: true } : {}),
|
|
15106
|
+
...(targetResolution ? { targetResolution } : {}),
|
|
14825
15107
|
...extra,
|
|
14826
15108
|
};
|
|
14827
15109
|
};
|
|
@@ -15004,26 +15286,41 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
|
|
|
15004
15286
|
const refs = pattern === "callers_of"
|
|
15005
15287
|
? findCallersFederated(allRepos, defRepo, targetName, 1, new Set(), defFile, selectedTarget?.identity ?? null)
|
|
15006
15288
|
: findCalleesFederated(allRepos, defRepo, targetName, 1, new Set(), defFile, selectedTarget?.identity ?? null);
|
|
15007
|
-
|
|
15289
|
+
// One row per caller symbol, but carrying every call site it has.
|
|
15290
|
+
// Collapsing to the first line lost the rest and made `count`
|
|
15291
|
+
// read as a call-site count (KNODIN-29).
|
|
15292
|
+
const byCaller = new Map();
|
|
15008
15293
|
const rows = [];
|
|
15294
|
+
let callSiteCount = 0;
|
|
15009
15295
|
for (const r of refs) {
|
|
15010
15296
|
const file = formatPath(r.repoPath || defRepo, r.filePath);
|
|
15011
15297
|
const key = `${r.symbol}|${file}`;
|
|
15012
|
-
|
|
15298
|
+
callSiteCount++;
|
|
15299
|
+
const existing = byCaller.get(key);
|
|
15300
|
+
if (existing) {
|
|
15301
|
+
if (r.lineNumber !== undefined && !existing.callSiteLines?.includes(r.lineNumber))
|
|
15302
|
+
existing.callSiteLines?.push(r.lineNumber);
|
|
15013
15303
|
continue;
|
|
15014
|
-
|
|
15304
|
+
}
|
|
15015
15305
|
const rr = (r.repoPath ? allRepos.find((x) => x.path === r.repoPath) : primary)?.db
|
|
15016
15306
|
.query("SELECT * FROM symbols WHERE name = ? AND filePath = ? LIMIT 1")
|
|
15017
15307
|
.get(r.symbol, r.filePath);
|
|
15018
|
-
|
|
15308
|
+
const row = {
|
|
15019
15309
|
symbol: r.symbol,
|
|
15020
15310
|
file,
|
|
15021
15311
|
line: r.lineNumber,
|
|
15022
15312
|
kind: r.kind ?? "call",
|
|
15313
|
+
callSiteLines: r.lineNumber !== undefined ? [r.lineNumber] : [],
|
|
15023
15314
|
...(rr ? { identity: symbolIdentity(r.repoPath || defRepo, rr) } : {}),
|
|
15024
|
-
}
|
|
15315
|
+
};
|
|
15316
|
+
byCaller.set(key, row);
|
|
15317
|
+
rows.push(row);
|
|
15025
15318
|
}
|
|
15026
|
-
|
|
15319
|
+
for (const row of rows) {
|
|
15320
|
+
row.callSiteLines?.sort((a, b) => a - b);
|
|
15321
|
+
row.line = row.callSiteLines?.[0] ?? row.line;
|
|
15322
|
+
}
|
|
15323
|
+
return finish(rows, { callSiteCount });
|
|
15027
15324
|
}
|
|
15028
15325
|
case "imports_of": {
|
|
15029
15326
|
const stmt = db.query("SELECT DISTINCT toFile, kind FROM dependencies WHERE fromFile = ?");
|
|
@@ -147,10 +147,27 @@ class PoolRun {
|
|
|
147
147
|
return this.buckets.get(bestKey)?.pop() ?? null;
|
|
148
148
|
}
|
|
149
149
|
attach(slot) {
|
|
150
|
-
|
|
151
|
-
slot.
|
|
152
|
-
slot
|
|
153
|
-
|
|
150
|
+
// Bind these listeners to the handle they were registered on, not just to
|
|
151
|
+
// the slot. `PoolWorkerHandle` offers no listener removal, so a replaced
|
|
152
|
+
// worker's listeners stay live and still close over the slot — whose
|
|
153
|
+
// `.handle` is by then the replacement. A crashed worker's inevitable late
|
|
154
|
+
// `exit` therefore re-entered `onSlotFailure` and terminated its OWN
|
|
155
|
+
// replacement, cancelling the one documented retry and leaving the lane
|
|
156
|
+
// dead; on a one-worker pool that dropped the whole run to inline parsing
|
|
157
|
+
// (KNODIN-20). Mirrors the `pending.child !== child` guard the MCP
|
|
158
|
+
// supervisor already carries.
|
|
159
|
+
const handle = slot.handle;
|
|
160
|
+
const current = () => slot.handle === handle;
|
|
161
|
+
handle.onMessage((response) => {
|
|
162
|
+
if (current())
|
|
163
|
+
this.onResponse(slot, response);
|
|
164
|
+
});
|
|
165
|
+
handle.onError((error) => {
|
|
166
|
+
if (current())
|
|
167
|
+
this.onSlotFailure(slot, error.message);
|
|
168
|
+
});
|
|
169
|
+
handle.onExit((code) => {
|
|
170
|
+
if (current() && slot.inFlight)
|
|
154
171
|
this.onSlotFailure(slot, `worker exited with code ${code}`);
|
|
155
172
|
});
|
|
156
173
|
}
|