knodin 0.11.0 → 0.12.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/bin/cli.js +41 -4
- package/dist/src/cli-model.js +1 -0
- package/dist/src/engine/index.js +467 -75
- package/dist/src/graph-query-health.js +9 -3
- package/dist/src/init.js +81 -0
- package/dist/src/lifecycle-health.js +46 -3
- package/dist/src/output-telemetry.js +80 -1
- package/dist/src/repair-progress.js +12 -0
- package/dist/src/tools/knodin-tools.js +16 -0
- package/docs/releases/0.12.0.md +281 -0
- package/docs/releases/0.12.1.md +98 -0
- package/package.json +5 -2
package/dist/src/engine/index.js
CHANGED
|
@@ -826,6 +826,97 @@ function freshnessMechanismFor(repoPath, policy) {
|
|
|
826
826
|
// Bumped whenever any repo is (re)indexed, so cached community-detection results
|
|
827
827
|
// below can be invalidated cheaply instead of recomputed on every call.
|
|
828
828
|
export const KNODIN_SCHEMA_VERSION = 24;
|
|
829
|
+
/**
|
|
830
|
+
* Reciprocal Rank Fusion's smoothing constant. Was written as a bare `60` at
|
|
831
|
+
* each use site; named so the ranking formula can be read and changed in one
|
|
832
|
+
* place. Adjacent ranks differ by roughly 0.00026 at this k, which is the unit
|
|
833
|
+
* every other ranking weight has to be reasoned about in.
|
|
834
|
+
*/
|
|
835
|
+
const RRF_K = 60;
|
|
836
|
+
/**
|
|
837
|
+
* How hard graph centrality pulls on ranking, as a fraction of one full RRF
|
|
838
|
+
* component.
|
|
839
|
+
*
|
|
840
|
+
* Semantic similarity alone ranks a minutes-old duplicate above the
|
|
841
|
+
* implementation it duplicates: the duplicate is short and made almost entirely
|
|
842
|
+
* of the query's own words, while the canonical version is diluted by a real
|
|
843
|
+
* body. The searcher then reads their own new leaf as evidence that nothing
|
|
844
|
+
* existed (KNODIN-15). Community and degree data was already in the graph and
|
|
845
|
+
* already in the output; it simply was not weighted into ranking.
|
|
846
|
+
*
|
|
847
|
+
* 0.05 caps the boost at `0.05 / (RRF_K + 1)` ≈ 0.00082, about three rank
|
|
848
|
+
* positions. That is deliberately modest, because the opposite failure is
|
|
849
|
+
* equally real and lands on the same symbols: a *seam* is by definition a
|
|
850
|
+
* well-documented symbol with no callers, and those are already the hardest
|
|
851
|
+
* things to retrieve (KNODIN-28, KNODIN-30). A large centrality weight would
|
|
852
|
+
* bury them to fix this, trading one silent miss for another. So a connected
|
|
853
|
+
* implementation climbs past neighbours it is scoring near, and does not
|
|
854
|
+
* override a decisively stronger semantic match.
|
|
855
|
+
*
|
|
856
|
+
* Zero-in-degree symbols are excluded from the centrality ranking rather than
|
|
857
|
+
* ranked last, so an uncalled symbol is never *penalised* — it only forgoes a
|
|
858
|
+
* boost it has no evidence for.
|
|
859
|
+
*/
|
|
860
|
+
const CENTRALITY_RRF_WEIGHT = 0.05;
|
|
861
|
+
/**
|
|
862
|
+
* The live ranking weights, exported so a benchmark can record which values
|
|
863
|
+
* produced its numbers.
|
|
864
|
+
*
|
|
865
|
+
* Comparing two ranking runs whose constants differed is not a measurement, and
|
|
866
|
+
* the difference is invisible unless the run writes them down. Reading them here
|
|
867
|
+
* rather than copying them into the benchmark keeps that record from going stale
|
|
868
|
+
* the first time someone tunes a weight.
|
|
869
|
+
*/
|
|
870
|
+
/**
|
|
871
|
+
* How hard a partial (OR) lexical match pulls, as a fraction of one full RRF
|
|
872
|
+
* component, and how many of them are considered.
|
|
873
|
+
*
|
|
874
|
+
* The weight buys a statable property: at 0.75 a top-ranked partial match
|
|
875
|
+
* contributes `0.75 / (RRF_K + 1)` ≈ 0.0123, which is enough to lift a symbol
|
|
876
|
+
* sitting around semantic rank 90 above an unsupported semantic rank 1. So
|
|
877
|
+
* lexical evidence can rescue a symbol the embedder under-ranked, and cannot
|
|
878
|
+
* resurrect one the embedder judged irrelevant. At 1.0 the loose channel becomes
|
|
879
|
+
* a full peer of semantic similarity and wins from arbitrarily deep, which is
|
|
880
|
+
* over-correction.
|
|
881
|
+
*
|
|
882
|
+
* The limit is a separate lever and conflating the two is how this gets
|
|
883
|
+
* mistuned. RRF has a floor — the hundredth partial match still scores
|
|
884
|
+
* `w / (RRF_K + 100)` — so an unbounded channel gives a meaningful shove to a
|
|
885
|
+
* hundred symbols at once on any query full of common words. Truncating raises
|
|
886
|
+
* the per-item floor slightly while applying it to under a third as many rows,
|
|
887
|
+
* which is the trade that actually reduces noise.
|
|
888
|
+
*
|
|
889
|
+
* Both are starting points measured against `benchmarks/evaluations/c103-prose-retrieval`,
|
|
890
|
+
* not settled constants. Re-run it before changing either.
|
|
891
|
+
*/
|
|
892
|
+
const LOOSE_LEXICAL_RRF_WEIGHT = 0.75;
|
|
893
|
+
/**
|
|
894
|
+
* How many query words the quorum check will probe, one bounded FTS query each.
|
|
895
|
+
*
|
|
896
|
+
* The check exists because a single incidental token match is not evidence, and
|
|
897
|
+
* this index cannot tell the difference on its own: FTS5 here has no stemming
|
|
898
|
+
* and does not split identifiers, so "validations" misses "validation" and
|
|
899
|
+
* `SessionAuthenticator` is one opaque token. The cap keeps a pathologically
|
|
900
|
+
* long query from turning into a pathological number of probes; beyond it the
|
|
901
|
+
* leading words decide, which is where the discriminating terms usually are.
|
|
902
|
+
*/
|
|
903
|
+
const LEXICAL_QUORUM_TOKEN_CAP = 8;
|
|
904
|
+
const LOOSE_LEXICAL_LIMIT = 30;
|
|
905
|
+
/**
|
|
906
|
+
* The live ranking weights, exported so a benchmark can record which values
|
|
907
|
+
* produced its numbers.
|
|
908
|
+
*
|
|
909
|
+
* Comparing two ranking runs whose constants differed is not a measurement, and
|
|
910
|
+
* the difference is invisible unless the run writes them down. Reading them here
|
|
911
|
+
* rather than copying them into the benchmark keeps that record from going stale
|
|
912
|
+
* the first time someone tunes a weight.
|
|
913
|
+
*/
|
|
914
|
+
export const RANKING_CONSTANTS = {
|
|
915
|
+
rrfK: RRF_K,
|
|
916
|
+
centralityWeight: CENTRALITY_RRF_WEIGHT,
|
|
917
|
+
looseLexicalWeight: LOOSE_LEXICAL_RRF_WEIGHT,
|
|
918
|
+
looseLexicalLimit: LOOSE_LEXICAL_LIMIT,
|
|
919
|
+
};
|
|
829
920
|
const LAST_REBUILD_SCHEMA_VERSION = 24;
|
|
830
921
|
let indexGeneration = 0;
|
|
831
922
|
const resourceReachabilityCache = new Map();
|
|
@@ -1631,12 +1722,76 @@ function createLanguageMemo() {
|
|
|
1631
1722
|
return name;
|
|
1632
1723
|
};
|
|
1633
1724
|
}
|
|
1725
|
+
/**
|
|
1726
|
+
* Node types a `//` line can carry across the grammars this indexes. They
|
|
1727
|
+
* disagree on the name, so matching one of them is not enough on its own — see
|
|
1728
|
+
* `isLineComment` below, which also requires the `//` prefix.
|
|
1729
|
+
*/
|
|
1730
|
+
const LINE_COMMENT_NODE_TYPES = new Set(["comment", "line_comment", "hash_comment"]);
|
|
1731
|
+
/**
|
|
1732
|
+
* Repo-relative files that reference `symbol` and are test files, read from raw
|
|
1733
|
+
* reference rows rather than from resolved callers.
|
|
1734
|
+
*
|
|
1735
|
+
* Resolved callers cannot answer this. A call inside an anonymous callback --
|
|
1736
|
+
* `it("...", () => { target() })`, which is how essentially all test code is
|
|
1737
|
+
* written -- has no enclosing NAMED symbol, so it is stored with a null
|
|
1738
|
+
* `callerSymbol` and every caller-resolution path drops it. Measured on this
|
|
1739
|
+
* repository: 29,158 of 31,467 references from test files (92.7%) have a null
|
|
1740
|
+
* caller, against 390 of 18,399 (2.1%) from production files. So a coverage
|
|
1741
|
+
* question answered from callers is roughly 93% blind in exactly the files it is
|
|
1742
|
+
* asking about, and reported `untested: true` for 642 symbols that have tests
|
|
1743
|
+
* (KNODIN-36).
|
|
1744
|
+
*
|
|
1745
|
+
* The reference row itself carries `callerFile`, which is all this question
|
|
1746
|
+
* needs. Attributing the call to the file instead would be the other repair, and
|
|
1747
|
+
* `call-graph.spec.ts` deliberately forbids it: a file path must never appear as
|
|
1748
|
+
* a caller. So the fix belongs here, at the question, not in the graph.
|
|
1749
|
+
*
|
|
1750
|
+
* Shared by `explain`'s `untested`, `tests_for`, and `detectKnowledgeGaps` so the
|
|
1751
|
+
* definition of "covered by a test" cannot drift between them.
|
|
1752
|
+
*/
|
|
1753
|
+
function testCallerFilesFor(db, symbol, relativeFile) {
|
|
1754
|
+
const statement = db.query('SELECT DISTINCT callerFile FROM "references" WHERE calleeSymbol = ? AND (calleeFile = ? OR calleeFile IS NULL)');
|
|
1755
|
+
const rows = statement.all(symbol, relativeFile);
|
|
1756
|
+
statement.finalize();
|
|
1757
|
+
return rows.map((row) => row.callerFile).filter((file) => isTestFilePath(file));
|
|
1758
|
+
}
|
|
1634
1759
|
/** Clean up and format comment/docstring blocks in JavaScript/TypeScript. */
|
|
1635
1760
|
function getPrecedingComment(node) {
|
|
1636
1761
|
let target = node;
|
|
1637
1762
|
while (target.parent && target.parent.startPosition.row === target.startPosition.row) {
|
|
1638
1763
|
target = target.parent;
|
|
1639
1764
|
}
|
|
1765
|
+
// Only decorators and modifiers may sit between a doc comment and the thing
|
|
1766
|
+
// it documents. The walk used to step over ANY five siblings looking for a
|
|
1767
|
+
// comment, so a symbol with no doc comment of its own adopted whichever
|
|
1768
|
+
// comment preceded a nearby earlier symbol — attaching prose to a symbol it
|
|
1769
|
+
// was never written about, in both the embedding input and the FTS `summary`
|
|
1770
|
+
// column.
|
|
1771
|
+
//
|
|
1772
|
+
// That is not a corner case. Measured on this repository before the fix, 857
|
|
1773
|
+
// of 1,329 documented symbols under `src/` — 64% — shared summary text with
|
|
1774
|
+
// another symbol, and spot checks showed plain misattribution rather than
|
|
1775
|
+
// genuine repetition: an interface carrying a neighbouring function's
|
|
1776
|
+
// docstring, two unrelated types sharing a third symbol's sentence. Prose
|
|
1777
|
+
// retrieval matching that text then returns the wrong symbol confidently,
|
|
1778
|
+
// which is the same class of failure as the truncation above and strictly
|
|
1779
|
+
// worse: absent prose loses a hit, misattributed prose manufactures one.
|
|
1780
|
+
//
|
|
1781
|
+
// Stopping at the first non-skippable node is the whole fix. A row-adjacency
|
|
1782
|
+
// check was tried alongside it and removed: decorators are children of the
|
|
1783
|
+
// declaration in some grammars and siblings in others, so "directly above"
|
|
1784
|
+
// is not portable, and it silently dropped the doc comment of every
|
|
1785
|
+
// decorated class. The walk already refuses to cross a declaration, which is
|
|
1786
|
+
// what the misattribution needed.
|
|
1787
|
+
const SKIPPABLE_BEFORE_DOC = new Set([
|
|
1788
|
+
"decorator",
|
|
1789
|
+
"export",
|
|
1790
|
+
"default",
|
|
1791
|
+
"async",
|
|
1792
|
+
"abstract",
|
|
1793
|
+
"declare",
|
|
1794
|
+
]);
|
|
1640
1795
|
let prev = target.previousSibling;
|
|
1641
1796
|
let count = 0;
|
|
1642
1797
|
while (prev && count < 5) {
|
|
@@ -1646,6 +1801,8 @@ function getPrecedingComment(node) {
|
|
|
1646
1801
|
prev.type === "line_comment") {
|
|
1647
1802
|
break;
|
|
1648
1803
|
}
|
|
1804
|
+
if (!SKIPPABLE_BEFORE_DOC.has(prev.type))
|
|
1805
|
+
return null;
|
|
1649
1806
|
prev = prev.previousSibling;
|
|
1650
1807
|
count++;
|
|
1651
1808
|
}
|
|
@@ -1654,6 +1811,48 @@ function getPrecedingComment(node) {
|
|
|
1654
1811
|
prev.type === "hash_comment" ||
|
|
1655
1812
|
prev.type === "block_comment" ||
|
|
1656
1813
|
prev.type === "line_comment")) {
|
|
1814
|
+
// Walk back over a run of consecutive `//` lines and rejoin them.
|
|
1815
|
+
//
|
|
1816
|
+
// tree-sitter gives every `//` line its OWN comment node, so `prev` is the
|
|
1817
|
+
// LAST line of a block, and the `split("\n")` below could never fire for
|
|
1818
|
+
// this style. Every multi-line `//` doc comment in every indexed
|
|
1819
|
+
// repository was therefore truncated to its final line before it reached
|
|
1820
|
+
// the embedding or the FTS index — silently, since a one-line summary
|
|
1821
|
+
// looks perfectly well-formed.
|
|
1822
|
+
//
|
|
1823
|
+
// That is a retrieval defect, not a cosmetic one: the first line of a doc
|
|
1824
|
+
// comment is usually the sentence that says what the thing is for, which
|
|
1825
|
+
// is exactly what a capability question matches on. `/* */` blocks are a
|
|
1826
|
+
// single node and were never affected, so the damage was invisible in any
|
|
1827
|
+
// codebase that preferred them.
|
|
1828
|
+
//
|
|
1829
|
+
// Adjacency is required: a blank line or any code between two comment
|
|
1830
|
+
// nodes ends the block, so a stray earlier comment is not absorbed.
|
|
1831
|
+
// Keyed on the `//` PREFIX rather than on one node type. The surrounding
|
|
1832
|
+
// checks accept `comment`, `line_comment`, `hash_comment` and
|
|
1833
|
+
// `block_comment` because grammars disagree about the name, and an
|
|
1834
|
+
// earlier version of this rejoin tested `type === "comment"` alone — so
|
|
1835
|
+
// in any grammar that calls a `//` line `line_comment`, multi-line
|
|
1836
|
+
// comments went on being truncated to their last line while appearing
|
|
1837
|
+
// fixed everywhere else. The prefix is the thing that actually decides
|
|
1838
|
+
// whether a node is one line of a multi-line run.
|
|
1839
|
+
const isLineComment = (node) => LINE_COMMENT_NODE_TYPES.has(node.type) && node.text.trim().startsWith("//");
|
|
1840
|
+
if (isLineComment(prev)) {
|
|
1841
|
+
const lines = [];
|
|
1842
|
+
let line = prev;
|
|
1843
|
+
while (line &&
|
|
1844
|
+
isLineComment(line) &&
|
|
1845
|
+
// Consecutive source lines, and nothing but whitespace between them.
|
|
1846
|
+
(lines.length === 0 || line.endPosition.row + 1 === prev.startPosition.row)) {
|
|
1847
|
+
lines.unshift(line.text.trim());
|
|
1848
|
+
prev = line;
|
|
1849
|
+
line = line.previousSibling;
|
|
1850
|
+
}
|
|
1851
|
+
return lines
|
|
1852
|
+
.map((entry) => entry.replace(/^\/\/+\s*/, ""))
|
|
1853
|
+
.join("\n")
|
|
1854
|
+
.trim();
|
|
1855
|
+
}
|
|
1657
1856
|
let text = prev.text.trim();
|
|
1658
1857
|
if (text.startsWith("/*")) {
|
|
1659
1858
|
text = text
|
|
@@ -6988,6 +7187,34 @@ function repairPathsAffectTypeScriptDi(db, repoPath, repairPaths) {
|
|
|
6988
7187
|
// measurements.
|
|
6989
7188
|
const DEFAULT_EMBEDDING_BATCH_SIZE = 1;
|
|
6990
7189
|
const MAX_EMBEDDING_BATCH_SIZE = 32;
|
|
7190
|
+
/**
|
|
7191
|
+
* Total characters embedded per symbol.
|
|
7192
|
+
*
|
|
7193
|
+
* A `EMBEDDING_SOURCE_BUDGET` capping the source body inside this was tried and
|
|
7194
|
+
* reverted. It looked like a win — the diluted implementation's similarity rose
|
|
7195
|
+
* 0.163 to 0.212 against its own fresh duplicate — but that was measured on the
|
|
7196
|
+
* suite's bag-of-words test embedder, and re-measuring on the real MiniLM model
|
|
7197
|
+
* across two body shapes, three query phrasings and both budgets showed the cap
|
|
7198
|
+
* never changes the winner in any of the twelve configurations. Its apparent
|
|
7199
|
+
* benefit came from the fixture: `fillerBody` emitted forty copies of one line,
|
|
7200
|
+
* and truncating identical lines removes a redundancy bag-of-words punishes and
|
|
7201
|
+
* a transformer largely ignores. On realistic varied code the cap is neutral to
|
|
7202
|
+
* slightly negative.
|
|
7203
|
+
*
|
|
7204
|
+
* The general lesson, which cost a forced re-embed to learn: a short document
|
|
7205
|
+
* made almost entirely of query tokens beats a long one under cosine
|
|
7206
|
+
* similarity, and no reweighting of a single vector's inputs changes that.
|
|
7207
|
+
* KNODIN-15's remedy is a non-similarity signal — graph degree, in the rank
|
|
7208
|
+
* fusion — not the representation.
|
|
7209
|
+
*/
|
|
7210
|
+
const EMBEDDING_INPUT_BUDGET = 1000;
|
|
7211
|
+
/**
|
|
7212
|
+
* The text embedded for one symbol: what it is called, what kind of thing it
|
|
7213
|
+
* is, what its docstring says it does, and then its body.
|
|
7214
|
+
*/
|
|
7215
|
+
function embeddingInputFor(symbol, source) {
|
|
7216
|
+
return `Name: ${symbol.name}\nKind: ${symbol.kind}\nSummary: ${symbol.summary || ""}\nSource:\n${source}`.slice(0, EMBEDDING_INPUT_BUDGET);
|
|
7217
|
+
}
|
|
6991
7218
|
// A full embedding pass can touch many symbols per source file. Keep only a
|
|
6992
7219
|
// bounded LRU of split lines so large repositories avoid re-reading a file for
|
|
6993
7220
|
// every symbol without turning source preparation into an unbounded memory sink.
|
|
@@ -7182,10 +7409,7 @@ async function indexEmbeddings(db, repoPath, progress, signal) {
|
|
|
7182
7409
|
throwIfAborted(signal);
|
|
7183
7410
|
const prepared = missing.slice(offset, offset + batchSize).map((sym) => {
|
|
7184
7411
|
const source = sourceForEmbedding(sym.filePath, sym.startLine, sym.endLine);
|
|
7185
|
-
return {
|
|
7186
|
-
symbol: sym,
|
|
7187
|
-
text: `Name: ${sym.name}\nKind: ${sym.kind}\nSummary: ${sym.summary || ""}\nSource:\n${source}`.slice(0, 1000),
|
|
7188
|
-
};
|
|
7412
|
+
return { symbol: sym, text: embeddingInputFor(sym, source) };
|
|
7189
7413
|
});
|
|
7190
7414
|
const generated = await embedPreparedBatch(prepared, onModelProgress);
|
|
7191
7415
|
throwIfAborted(signal);
|
|
@@ -8004,50 +8228,6 @@ const freshnessChecks = new Map();
|
|
|
8004
8228
|
* deterministic, rather than in wall-clock milliseconds, which is flaky.
|
|
8005
8229
|
*/
|
|
8006
8230
|
const freshnessStats = { probes: 0, reconciles: 0, cacheHits: 0 };
|
|
8007
|
-
/** Read HEAD without spawning Git. `null` means this is not a Git checkout;
|
|
8008
|
-
* `undefined` means Git metadata exists but could not be resolved safely. */
|
|
8009
|
-
function readGitHeadFast(repoPath) {
|
|
8010
|
-
const marker = path.join(repoPath, ".git");
|
|
8011
|
-
if (!fs.existsSync(marker))
|
|
8012
|
-
return null;
|
|
8013
|
-
try {
|
|
8014
|
-
const markerStat = fs.statSync(marker);
|
|
8015
|
-
const gitDir = markerStat.isDirectory()
|
|
8016
|
-
? marker
|
|
8017
|
-
: path.resolve(repoPath, fs
|
|
8018
|
-
.readFileSync(marker, "utf8")
|
|
8019
|
-
.trim()
|
|
8020
|
-
.replace(/^gitdir:\s*/, ""));
|
|
8021
|
-
const head = fs.readFileSync(path.join(gitDir, "HEAD"), "utf8").trim();
|
|
8022
|
-
if (/^[a-f0-9]{40}$/i.test(head))
|
|
8023
|
-
return head.toLowerCase();
|
|
8024
|
-
const refPrefix = "ref: ";
|
|
8025
|
-
const ref = head.startsWith(refPrefix) ? head.slice(refPrefix.length).trim() : "";
|
|
8026
|
-
if (!ref)
|
|
8027
|
-
return undefined;
|
|
8028
|
-
let commonDir = gitDir;
|
|
8029
|
-
const commonMarker = path.join(gitDir, "commondir");
|
|
8030
|
-
if (fs.existsSync(commonMarker))
|
|
8031
|
-
commonDir = path.resolve(gitDir, fs.readFileSync(commonMarker, "utf8").trim());
|
|
8032
|
-
for (const base of [gitDir, commonDir]) {
|
|
8033
|
-
const loose = path.join(base, ref);
|
|
8034
|
-
if (fs.existsSync(loose))
|
|
8035
|
-
return fs.readFileSync(loose, "utf8").trim().toLowerCase();
|
|
8036
|
-
}
|
|
8037
|
-
const packed = path.join(commonDir, "packed-refs");
|
|
8038
|
-
if (!fs.existsSync(packed))
|
|
8039
|
-
return undefined;
|
|
8040
|
-
return fs
|
|
8041
|
-
.readFileSync(packed, "utf8")
|
|
8042
|
-
.split("\n")
|
|
8043
|
-
.find((line) => line.endsWith(` ${ref}`))
|
|
8044
|
-
?.split(" ", 1)[0]
|
|
8045
|
-
?.toLowerCase();
|
|
8046
|
-
}
|
|
8047
|
-
catch {
|
|
8048
|
-
return undefined;
|
|
8049
|
-
}
|
|
8050
|
-
}
|
|
8051
8231
|
/**
|
|
8052
8232
|
* Paths the live watcher for `repoPath` currently has registered, flattened to
|
|
8053
8233
|
* repo-relative form. Empty when no watcher is running (test mode, or after
|
|
@@ -8272,15 +8452,20 @@ async function ensureIndexFresh(repoPath, db) {
|
|
|
8272
8452
|
// already reached its event queue. Git worktrees therefore take the bounded
|
|
8273
8453
|
// porcelain proof on every answer. The long watcher lease remains safe for
|
|
8274
8454
|
// non-Git directories, where the watcher is the only new-file signal.
|
|
8455
|
+
//
|
|
8456
|
+
// The lease used to be paired with a spawn-free HEAD read, so a leased
|
|
8457
|
+
// answer could still be invalidated by a commit. Once git-backed repos took
|
|
8458
|
+
// `lease = 0` that read became unreachable — it was called only when `.git`
|
|
8459
|
+
// was absent, which was its own first bail-out — and the term it fed was
|
|
8460
|
+
// inert. Both are gone (KNODIN-16); a lease is now reached only where there
|
|
8461
|
+
// is no HEAD to check.
|
|
8275
8462
|
const gitBacked = fs.existsSync(path.join(key, ".git"));
|
|
8276
8463
|
const lease = gitBacked
|
|
8277
8464
|
? 0
|
|
8278
8465
|
: watchQueues.get(key)?.ready
|
|
8279
8466
|
? WATCHED_FRESHNESS_LEASE_MS
|
|
8280
8467
|
: FRESHNESS_PROBE_TTL_MS;
|
|
8281
|
-
|
|
8282
|
-
const headUnchanged = head === null || (head !== undefined && head === getMeta(db, "lastIndexedHead"));
|
|
8283
|
-
if (cached && Date.now() - cached.at < lease && headUnchanged) {
|
|
8468
|
+
if (cached && Date.now() - cached.at < lease) {
|
|
8284
8469
|
freshnessStats.cacheHits++;
|
|
8285
8470
|
return cached.staleness;
|
|
8286
8471
|
}
|
|
@@ -12292,7 +12477,20 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
|
|
|
12292
12477
|
})
|
|
12293
12478
|
: [];
|
|
12294
12479
|
const allBlastFiles = Array.from(new Set(allCallers.map((c) => c.filePath)));
|
|
12295
|
-
|
|
12480
|
+
// Asked of raw reference rows, not of resolved callers: a test that calls
|
|
12481
|
+
// this symbol from inside an anonymous `it(...)` callback has no caller
|
|
12482
|
+
// symbol and is invisible to `allCallers` (KNODIN-36). `untested: true`
|
|
12483
|
+
// is a positive assertion with no zero-count to invite doubt, so it has
|
|
12484
|
+
// to be answered from the evidence that actually exists.
|
|
12485
|
+
// Three values, not two. Without the defining repository's database this
|
|
12486
|
+
// question cannot be answered, and the previous fallback -- guessing from
|
|
12487
|
+
// resolved callers -- is precisely the collapse that made this field
|
|
12488
|
+
// wrong in the first place: it turned "cannot tell" into a confident
|
|
12489
|
+
// `true`. `untested` is now simply absent when the evidence is absent,
|
|
12490
|
+
// so a caller reading it gets a fact or nothing, never a guess.
|
|
12491
|
+
const untested = targetDb
|
|
12492
|
+
? testCallerFilesFor(targetDb, primaryDef.name, primaryDef.filePath).length === 0
|
|
12493
|
+
: undefined;
|
|
12296
12494
|
const minimal = detailLevel === "minimal";
|
|
12297
12495
|
const cap = minimal ? 25 : 200;
|
|
12298
12496
|
const truncated = allCallers.length > cap ||
|
|
@@ -14447,9 +14645,7 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
|
|
|
14447
14645
|
.split(/\s+/)
|
|
14448
14646
|
.filter((word) => word.length >= 4)
|
|
14449
14647
|
.map((word) => `"${word}"`);
|
|
14450
|
-
const
|
|
14451
|
-
if (tokens.length === 0)
|
|
14452
|
-
return [];
|
|
14648
|
+
const runFtsMatch = (match) => {
|
|
14453
14649
|
try {
|
|
14454
14650
|
const statement = repo.db.query(`
|
|
14455
14651
|
SELECT symbolId FROM symbols_fts
|
|
@@ -14458,7 +14654,7 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
|
|
|
14458
14654
|
LIMIT 100
|
|
14459
14655
|
`);
|
|
14460
14656
|
const ids = statement
|
|
14461
|
-
.all(
|
|
14657
|
+
.all(match)
|
|
14462
14658
|
.map(({ symbolId }) => symbolId)
|
|
14463
14659
|
.filter((id) => allowedIds.has(id));
|
|
14464
14660
|
statement.finalize();
|
|
@@ -14469,8 +14665,43 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
|
|
|
14469
14665
|
return [];
|
|
14470
14666
|
}
|
|
14471
14667
|
};
|
|
14668
|
+
const runFts = (separator) => tokens.length === 0 ? [] : runFtsMatch(tokens.join(separator));
|
|
14669
|
+
const runFtsTokens = (subset) => subset.length === 0 ? [] : runFtsMatch(subset.join(" OR "));
|
|
14472
14670
|
const and = measurePerfPhaseSync("fts", () => runFts(" "));
|
|
14473
|
-
|
|
14671
|
+
// The OR set, restricted to candidates matching at least two of the
|
|
14672
|
+
// query's words.
|
|
14673
|
+
//
|
|
14674
|
+
// A single incidental token match is not evidence. FTS5 here has no
|
|
14675
|
+
// stemming and does not split identifiers, so for "session
|
|
14676
|
+
// authentication and token validations" the class that actually does
|
|
14677
|
+
// the work matches NOTHING — its docstring says "sessions" and
|
|
14678
|
+
// "validation" — while a handler whose docstring happens to contain
|
|
14679
|
+
// "session" matches once and collects the full partial-match credit.
|
|
14680
|
+
// Fusing that put the caller above the implementation it calls.
|
|
14681
|
+
//
|
|
14682
|
+
// Requiring a quorum keeps the signal that matters (a symbol echoing
|
|
14683
|
+
// several of the query's words) and drops the one that misleads. It
|
|
14684
|
+
// costs one bounded FTS query per token, which is why it is capped.
|
|
14685
|
+
const or = measurePerfPhaseSync("fts", () => {
|
|
14686
|
+
if (tokens.length < 2)
|
|
14687
|
+
return and;
|
|
14688
|
+
const loose = runFts(" OR ");
|
|
14689
|
+
if (loose.length === 0)
|
|
14690
|
+
return loose;
|
|
14691
|
+
const hits = new Map();
|
|
14692
|
+
for (const token of tokens.slice(0, LEXICAL_QUORUM_TOKEN_CAP)) {
|
|
14693
|
+
for (const id of runFtsTokens([token])) {
|
|
14694
|
+
hits.set(id, (hits.get(id) ?? 0) + 1);
|
|
14695
|
+
}
|
|
14696
|
+
}
|
|
14697
|
+
// No fallback when nothing clears the bar. An earlier version
|
|
14698
|
+
// returned the unfiltered set in that case, which reinstated the
|
|
14699
|
+
// exact failure the quorum exists to prevent — and did so
|
|
14700
|
+
// precisely in the small-corpus situation where one weak match is
|
|
14701
|
+
// most likely to be the only one. A lexical channel with nothing
|
|
14702
|
+
// trustworthy to say should say nothing.
|
|
14703
|
+
return loose.filter((id) => (hits.get(id) ?? 0) >= 2);
|
|
14704
|
+
});
|
|
14474
14705
|
const lexicalSeeds = or
|
|
14475
14706
|
.map((id) => snapshot.rowById.get(id))
|
|
14476
14707
|
.filter((row) => row !== undefined)
|
|
@@ -14526,6 +14757,15 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
|
|
|
14526
14757
|
// 1. Generate query embedding once
|
|
14527
14758
|
const queryVec = await generateMeasuredEmbedding(query, true);
|
|
14528
14759
|
const rankedCandidates = [];
|
|
14760
|
+
// Inbound reference counts for the centrality term. Computed once for the
|
|
14761
|
+
// federation rather than per repo, and only on the route that actually
|
|
14762
|
+
// fuses ranks — the structural and exact-match routes above return before
|
|
14763
|
+
// here and must keep their deterministic ordering.
|
|
14764
|
+
//
|
|
14765
|
+
// Not wrapped in a perf phase: the phase keys are a closed set that the
|
|
14766
|
+
// instrumentation contract asserts on, and adding one to measure a fix
|
|
14767
|
+
// would change what that contract describes.
|
|
14768
|
+
const referenceDegree = computeResolvedReferenceDegree(allRepos, formatPath);
|
|
14529
14769
|
let totalMatches = 0;
|
|
14530
14770
|
for (const repo of allRepos) {
|
|
14531
14771
|
const db = repo.db;
|
|
@@ -14574,12 +14814,35 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
|
|
|
14574
14814
|
}
|
|
14575
14815
|
// Sort by semantic score descending
|
|
14576
14816
|
const sortedSemantics = semanticMatches.sort((a, b) => b.score - a.score);
|
|
14577
|
-
const semanticScoresAreTied = sortedSemantics.length > 1 &&
|
|
14578
|
-
sortedSemantics.every((item) => item.score === sortedSemantics[0]?.score);
|
|
14579
14817
|
// 3. Reuse the lexical and bounded-graph ranks computed before query inference.
|
|
14818
|
+
//
|
|
14819
|
+
// Both operators are fused, as separate channels. The selection used to
|
|
14820
|
+
// be `semanticScoresAreTied ? or : and`, and semantic scores are never
|
|
14821
|
+
// all bit-identical under a real embedding model — so the OR set was
|
|
14822
|
+
// computed on every query and then discarded on every query.
|
|
14823
|
+
//
|
|
14824
|
+
// That made the AND clause the entire lexical channel, and for a query
|
|
14825
|
+
// phrased as prose the AND clause is empty: it demands one symbol
|
|
14826
|
+
// containing every surviving query word, function words included.
|
|
14827
|
+
// Nothing matches, so `ftsPart` was zero for every candidate and
|
|
14828
|
+
// "hybrid search" was pure vector search with a centrality nudge. The
|
|
14829
|
+
// docstring's BM25 evidence existed the whole time, in the OR set that
|
|
14830
|
+
// was thrown away (KNODIN-30).
|
|
14831
|
+
//
|
|
14832
|
+
// Fusing both rather than falling back from one to the other avoids a
|
|
14833
|
+
// discontinuity triggered by corpus contents rather than query intent:
|
|
14834
|
+
// with a fallback, one incidental symbol matching every word would flip
|
|
14835
|
+
// the channel from a broad candidate set to a single row, silently.
|
|
14836
|
+
//
|
|
14837
|
+
// `and ⊆ or` by construction — same tokens, narrower operator — so an
|
|
14838
|
+
// AND match collects BOTH terms while an OR-only match collects one.
|
|
14839
|
+
// The precision bonus falls out of that containment exactly, with no
|
|
14840
|
+
// extra machinery to express or tune.
|
|
14580
14841
|
const lexical = lexicalRanks.get(repo.path);
|
|
14581
|
-
const
|
|
14582
|
-
const
|
|
14842
|
+
const strictMatches = [...new Set(lexical?.and ?? [])].filter((id) => allowedIds.has(id));
|
|
14843
|
+
const looseMatches = [...new Set(lexical?.or ?? [])]
|
|
14844
|
+
.filter((id) => allowedIds.has(id))
|
|
14845
|
+
.slice(0, LOOSE_LEXICAL_LIMIT);
|
|
14583
14846
|
// Rank mappings for Reciprocal Rank Fusion (RRF)
|
|
14584
14847
|
const semanticRankMap = new Map();
|
|
14585
14848
|
const semanticById = new Map();
|
|
@@ -14587,19 +14850,57 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
|
|
|
14587
14850
|
semanticRankMap.set(item.id, idx + 1);
|
|
14588
14851
|
semanticById.set(item.id, item);
|
|
14589
14852
|
});
|
|
14590
|
-
const
|
|
14591
|
-
|
|
14592
|
-
|
|
14853
|
+
const strictRankMap = new Map();
|
|
14854
|
+
strictMatches.forEach((id, idx) => {
|
|
14855
|
+
strictRankMap.set(id, idx + 1);
|
|
14856
|
+
});
|
|
14857
|
+
const looseRankMap = new Map();
|
|
14858
|
+
looseMatches.forEach((id, idx) => {
|
|
14859
|
+
looseRankMap.set(id, idx + 1);
|
|
14593
14860
|
});
|
|
14861
|
+
// Rank the matched candidates that anything actually depends on, most
|
|
14862
|
+
// depended-upon first. Symbols with no inbound references are left out
|
|
14863
|
+
// entirely rather than ranked last — see CENTRALITY_RRF_WEIGHT.
|
|
14864
|
+
const centralityRankMap = new Map();
|
|
14865
|
+
{
|
|
14866
|
+
const connected = [];
|
|
14867
|
+
for (const id of new Set([
|
|
14868
|
+
...semanticRankMap.keys(),
|
|
14869
|
+
...strictRankMap.keys(),
|
|
14870
|
+
...looseRankMap.keys(),
|
|
14871
|
+
])) {
|
|
14872
|
+
const row = semanticById.get(id)?.row ?? snapshot.rowById.get(id);
|
|
14873
|
+
if (!row)
|
|
14874
|
+
continue;
|
|
14875
|
+
const inDegree = referenceDegree.get(`${formatPath(repo.path, row.filePath)}::${row.name}`)
|
|
14876
|
+
?.inDegree ?? 0;
|
|
14877
|
+
if (inDegree > 0)
|
|
14878
|
+
connected.push({ id, inDegree });
|
|
14879
|
+
}
|
|
14880
|
+
connected.sort((a, b) => b.inDegree - a.inDegree || a.id - b.id);
|
|
14881
|
+
connected.forEach((item, idx) => {
|
|
14882
|
+
centralityRankMap.set(item.id, idx + 1);
|
|
14883
|
+
});
|
|
14884
|
+
}
|
|
14594
14885
|
// Perform Reciprocal Rank Fusion (RRF)
|
|
14595
14886
|
const topRrf = measurePerfPhaseSync("fusion", () => {
|
|
14596
|
-
const allMatchedIds = new Set([
|
|
14887
|
+
const allMatchedIds = new Set([
|
|
14888
|
+
...semanticRankMap.keys(),
|
|
14889
|
+
...strictRankMap.keys(),
|
|
14890
|
+
...looseRankMap.keys(),
|
|
14891
|
+
]);
|
|
14597
14892
|
const rrfResults = Array.from(allMatchedIds).map((id) => {
|
|
14598
14893
|
const semRank = semanticRankMap.get(id);
|
|
14599
|
-
const
|
|
14600
|
-
const
|
|
14601
|
-
const
|
|
14602
|
-
|
|
14894
|
+
const strictRank = strictRankMap.get(id);
|
|
14895
|
+
const looseRank = looseRankMap.get(id);
|
|
14896
|
+
const centralityRank = centralityRankMap.get(id);
|
|
14897
|
+
const semPart = semRank !== undefined ? 1 / (RRF_K + semRank) : 0;
|
|
14898
|
+
const strictPart = strictRank !== undefined ? 1 / (RRF_K + strictRank) : 0;
|
|
14899
|
+
const loosePart = looseRank !== undefined ? LOOSE_LEXICAL_RRF_WEIGHT * (1 / (RRF_K + looseRank)) : 0;
|
|
14900
|
+
const centralityPart = centralityRank !== undefined
|
|
14901
|
+
? CENTRALITY_RRF_WEIGHT * (1 / (RRF_K + centralityRank))
|
|
14902
|
+
: 0;
|
|
14903
|
+
return { id, rrfScore: semPart + strictPart + loosePart + centralityPart };
|
|
14603
14904
|
});
|
|
14604
14905
|
// Sort by RRF score descending and take top N
|
|
14605
14906
|
return rrfResults.sort((a, b) => b.rrfScore - a.rrfScore).slice(0, candidateLimit);
|
|
@@ -14775,6 +15076,7 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
|
|
|
14775
15076
|
...(impactOptions.mode !== "file" ? ["impact"] : []),
|
|
14776
15077
|
]);
|
|
14777
15078
|
let selectedTarget;
|
|
15079
|
+
let targetResolution;
|
|
14778
15080
|
if (target && symbolPatterns.has(pattern)) {
|
|
14779
15081
|
const resolved = resolveSymbolRows(db, primary.path, target, selector);
|
|
14780
15082
|
if (resolved.ambiguity)
|
|
@@ -14804,6 +15106,26 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
|
|
|
14804
15106
|
ambiguity: resolved.ambiguity,
|
|
14805
15107
|
};
|
|
14806
15108
|
selectedTarget = resolved.selected;
|
|
15109
|
+
// Graph health and target resolution are separate axes. A healthy,
|
|
15110
|
+
// fresh graph answering `count: 0` for a misspelled symbol was
|
|
15111
|
+
// byte-identical to one answering for real dead code, and the
|
|
15112
|
+
// freshness stamp made the typo read as an authoritative negative
|
|
15113
|
+
// (KNODIN-28). Say which of the two happened.
|
|
15114
|
+
//
|
|
15115
|
+
// `resolveSymbolRows` only sees the primary database, so a symbol
|
|
15116
|
+
// defined in a federated repo must be checked before it is called
|
|
15117
|
+
// missing — otherwise cross-repo queries would report every target
|
|
15118
|
+
// as a typo.
|
|
15119
|
+
targetResolution = resolved.selected
|
|
15120
|
+
? "resolved"
|
|
15121
|
+
: allRepos.some((r) => {
|
|
15122
|
+
const stmt = r.db.query("SELECT filePath FROM symbols WHERE name = ? LIMIT 1");
|
|
15123
|
+
const row = stmt.get(target);
|
|
15124
|
+
stmt.finalize();
|
|
15125
|
+
return row != null;
|
|
15126
|
+
})
|
|
15127
|
+
? "resolved"
|
|
15128
|
+
: "not-found";
|
|
14807
15129
|
}
|
|
14808
15130
|
const finish = (rows, extra = {}) => {
|
|
14809
15131
|
for (const row of rows) {
|
|
@@ -14822,6 +15144,7 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
|
|
|
14822
15144
|
count: rows.length,
|
|
14823
15145
|
results: rows.slice(0, cap),
|
|
14824
15146
|
...(rows.length > cap ? { hasMore: true } : {}),
|
|
15147
|
+
...(targetResolution ? { targetResolution } : {}),
|
|
14825
15148
|
...extra,
|
|
14826
15149
|
};
|
|
14827
15150
|
};
|
|
@@ -15004,26 +15327,41 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
|
|
|
15004
15327
|
const refs = pattern === "callers_of"
|
|
15005
15328
|
? findCallersFederated(allRepos, defRepo, targetName, 1, new Set(), defFile, selectedTarget?.identity ?? null)
|
|
15006
15329
|
: findCalleesFederated(allRepos, defRepo, targetName, 1, new Set(), defFile, selectedTarget?.identity ?? null);
|
|
15007
|
-
|
|
15330
|
+
// One row per caller symbol, but carrying every call site it has.
|
|
15331
|
+
// Collapsing to the first line lost the rest and made `count`
|
|
15332
|
+
// read as a call-site count (KNODIN-29).
|
|
15333
|
+
const byCaller = new Map();
|
|
15008
15334
|
const rows = [];
|
|
15335
|
+
let callSiteCount = 0;
|
|
15009
15336
|
for (const r of refs) {
|
|
15010
15337
|
const file = formatPath(r.repoPath || defRepo, r.filePath);
|
|
15011
15338
|
const key = `${r.symbol}|${file}`;
|
|
15012
|
-
|
|
15339
|
+
callSiteCount++;
|
|
15340
|
+
const existing = byCaller.get(key);
|
|
15341
|
+
if (existing) {
|
|
15342
|
+
if (r.lineNumber !== undefined && !existing.callSiteLines?.includes(r.lineNumber))
|
|
15343
|
+
existing.callSiteLines?.push(r.lineNumber);
|
|
15013
15344
|
continue;
|
|
15014
|
-
|
|
15345
|
+
}
|
|
15015
15346
|
const rr = (r.repoPath ? allRepos.find((x) => x.path === r.repoPath) : primary)?.db
|
|
15016
15347
|
.query("SELECT * FROM symbols WHERE name = ? AND filePath = ? LIMIT 1")
|
|
15017
15348
|
.get(r.symbol, r.filePath);
|
|
15018
|
-
|
|
15349
|
+
const row = {
|
|
15019
15350
|
symbol: r.symbol,
|
|
15020
15351
|
file,
|
|
15021
15352
|
line: r.lineNumber,
|
|
15022
15353
|
kind: r.kind ?? "call",
|
|
15354
|
+
callSiteLines: r.lineNumber !== undefined ? [r.lineNumber] : [],
|
|
15023
15355
|
...(rr ? { identity: symbolIdentity(r.repoPath || defRepo, rr) } : {}),
|
|
15024
|
-
}
|
|
15356
|
+
};
|
|
15357
|
+
byCaller.set(key, row);
|
|
15358
|
+
rows.push(row);
|
|
15025
15359
|
}
|
|
15026
|
-
|
|
15360
|
+
for (const row of rows) {
|
|
15361
|
+
row.callSiteLines?.sort((a, b) => a - b);
|
|
15362
|
+
row.line = row.callSiteLines?.[0] ?? row.line;
|
|
15363
|
+
}
|
|
15364
|
+
return finish(rows, { callSiteCount });
|
|
15027
15365
|
}
|
|
15028
15366
|
case "imports_of": {
|
|
15029
15367
|
const stmt = db.query("SELECT DISTINCT toFile, kind FROM dependencies WHERE fromFile = ?");
|
|
@@ -15162,6 +15500,60 @@ export function createEngine(openPolicy = DEFAULT_ENGINE_OPEN_POLICY) {
|
|
|
15162
15500
|
seen.add(key);
|
|
15163
15501
|
rows.push({ symbol: r.symbol, file, line: r.lineNumber });
|
|
15164
15502
|
}
|
|
15503
|
+
// Resolved callers miss almost every real test. A call inside an
|
|
15504
|
+
// anonymous `it(...)` callback has no enclosing named symbol, so it
|
|
15505
|
+
// carries a null `callerSymbol` and `findCallersFederated` drops it
|
|
15506
|
+
// (92.7% of this repository's test-file references, against 2.1% of
|
|
15507
|
+
// production ones). Reading the raw rows recovers them (KNODIN-36).
|
|
15508
|
+
//
|
|
15509
|
+
// These rows carry no caller name because there genuinely is none,
|
|
15510
|
+
// and inventing one -- the file path, or the enclosing `describe`
|
|
15511
|
+
// title -- would put a non-symbol in a `symbol` field that
|
|
15512
|
+
// `call-graph.spec.ts` explicitly forbids. `symbol` is therefore
|
|
15513
|
+
// omitted and the file and line carry the answer, which is what the
|
|
15514
|
+
// question asked for anyway: which test covers this.
|
|
15515
|
+
for (const repo of allRepos) {
|
|
15516
|
+
if (defFile && repo.path !== defRepo)
|
|
15517
|
+
continue;
|
|
15518
|
+
// MIN(line) grouped by file, not DISTINCT(file, line). The rows are
|
|
15519
|
+
// deduped by file below, so a plain DISTINCT would leave SQLite free
|
|
15520
|
+
// to hand back whichever line its query plan reached first when a
|
|
15521
|
+
// test file references the symbol more than once -- a value that can
|
|
15522
|
+
// change between runs and plans, which is a flaky test waiting to be
|
|
15523
|
+
// written. The first reference is both stable and the more useful
|
|
15524
|
+
// one to report.
|
|
15525
|
+
// With no defining file the file filter is dropped rather than bound
|
|
15526
|
+
// to "", which would degrade the clause to `calleeFile IS NULL` and
|
|
15527
|
+
// silently discard every reference that does carry a file -- turning
|
|
15528
|
+
// "I do not know where this is defined" into "it is defined nowhere",
|
|
15529
|
+
// and under-reporting coverage for exactly the symbols we know least
|
|
15530
|
+
// about (ADR 008). findCallersFederated omits the filter in the same
|
|
15531
|
+
// situation; these two must not disagree.
|
|
15532
|
+
const anonymous = defFile
|
|
15533
|
+
? (() => {
|
|
15534
|
+
const stmt = repo.db.query('SELECT callerFile, MIN(line) AS line FROM "references" WHERE calleeSymbol = ? AND (calleeFile = ? OR calleeFile IS NULL) AND callerSymbol IS NULL GROUP BY callerFile ORDER BY callerFile');
|
|
15535
|
+
const out = stmt.all(target, defFile);
|
|
15536
|
+
stmt.finalize();
|
|
15537
|
+
return out;
|
|
15538
|
+
})()
|
|
15539
|
+
: (() => {
|
|
15540
|
+
const stmt = repo.db.query('SELECT callerFile, MIN(line) AS line FROM "references" WHERE calleeSymbol = ? AND callerSymbol IS NULL GROUP BY callerFile ORDER BY callerFile');
|
|
15541
|
+
const out = stmt.all(target);
|
|
15542
|
+
stmt.finalize();
|
|
15543
|
+
return out;
|
|
15544
|
+
})();
|
|
15545
|
+
for (const row of anonymous) {
|
|
15546
|
+
const file = formatPath(repo.path, row.callerFile);
|
|
15547
|
+
if (!isTestFilePath(file))
|
|
15548
|
+
continue;
|
|
15549
|
+
const key = `|${file}`;
|
|
15550
|
+
if (seen.has(key))
|
|
15551
|
+
continue;
|
|
15552
|
+
seen.add(key);
|
|
15553
|
+
rows.push({ file, line: row.line });
|
|
15554
|
+
}
|
|
15555
|
+
}
|
|
15556
|
+
rows.sort((a, b) => (a.file ?? "").localeCompare(b.file ?? "") || (a.line ?? 0) - (b.line ?? 0));
|
|
15165
15557
|
return finish(rows);
|
|
15166
15558
|
}
|
|
15167
15559
|
case "rename_preview": {
|