@mrciphersmith/keryx 0.2.82 → 0.2.84
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -1
- package/dist/cli.js +28857 -19477
- package/dist/core.js +25751 -0
- package/package.json +15 -2
- package/src/gdgraph/dangling.ts +204 -0
- package/src/gdgraph/find.ts +529 -37
- package/src/gdgraph/repomap.ts +140 -12
- package/src/gdgraph/staleness.ts +22 -9
- package/src/gdgraph/symbol.ts +45 -6
- package/src/gdgraph/treesitter/extract.ts +153 -5
- package/src/gdgraph/wiki-layer.ts +32 -1
- package/src/gdskills/bundled/skills/core/reviewer-skill-creator/SKILL.md +29 -0
- package/src/gdskills/bundled/skills/orchestration/code-verifier/SKILL.codex.md +1 -1
- package/src/gdskills/bundled/skills/orchestration/code-verifier/SKILL.cursor.md +1 -1
- package/src/gdskills/bundled/skills/orchestration/code-verifier/SKILL.md +1 -1
- package/src/gdskills/bundled/skills/orchestration/code-verifier/SKILL.opencode.md +1 -1
- package/src/gdskills/bundled/skills/orchestration/code-verifier/SKILL.zed.md +1 -1
- package/src/gdskills/bundled/skills/orchestration/flow-orchestrator/SKILL.md +34 -6
- package/src/gdskills/bundled/skills/orchestration/job-orchestrator/SKILL.codex.md +1 -1
- package/src/gdskills/bundled/skills/orchestration/job-orchestrator/SKILL.cursor.md +1 -1
- package/src/gdskills/bundled/skills/orchestration/job-orchestrator/SKILL.md +1 -1
- package/src/gdskills/bundled/skills/orchestration/job-orchestrator/SKILL.opencode.md +1 -1
- package/src/gdskills/bundled/skills/orchestration/job-orchestrator/SKILL.zed.md +1 -1
- package/src/gdskills/bundled/skills/quality/security-audit/SKILL.codex.md +52 -5
- package/src/gdskills/bundled/skills/quality/security-audit/SKILL.cursor.md +52 -5
- package/src/gdskills/bundled/skills/quality/security-audit/SKILL.md +52 -5
- package/src/gdgraph/affected.test.ts +0 -133
- package/src/gdgraph/build-integrity.test.ts +0 -193
- package/src/gdgraph/build-lang.test.ts +0 -406
- package/src/gdgraph/build.test.ts +0 -120
- package/src/gdgraph/config.test.ts +0 -47
- package/src/gdgraph/core-sources.test.ts +0 -99
- package/src/gdgraph/fallback.test.ts +0 -153
- package/src/gdgraph/find.test.ts +0 -78
- package/src/gdgraph/import-kind.test.ts +0 -525
- package/src/gdgraph/path.test.ts +0 -56
- package/src/gdgraph/repomap.test.ts +0 -110
- package/src/gdgraph/service.test.ts +0 -89
- package/src/gdgraph/staleness.test.ts +0 -208
- package/src/gdgraph/symbol.test.ts +0 -89
- package/src/gdgraph/symbols-capability.test.ts +0 -138
- package/src/gdgraph/treesitter/adapter.test.ts +0 -496
- package/src/gdgraph/treesitter/extract.test.ts +0 -278
- package/src/gdgraph/treesitter/no-treesitter-import.test.ts +0 -51
- package/src/gdgraph/treesitter/real-grammar-fixture.test.ts +0 -71
- package/src/gdgraph/treesitter/resolve-calls.test.ts +0 -38
- package/src/gdgraph/wiki-layer-no-git.test.ts +0 -73
- package/src/gdgraph/wiki-layer.test.ts +0 -211
package/src/gdgraph/find.ts
CHANGED
|
@@ -1,17 +1,84 @@
|
|
|
1
1
|
import type { GraphData } from "./types";
|
|
2
|
+
import {
|
|
3
|
+
MAX_NEXT_ACTIONS,
|
|
4
|
+
RETRIEVAL_NEXT_ACTIONS,
|
|
5
|
+
type RetrievalCode,
|
|
6
|
+
} from "../lib/retrieval-codes";
|
|
2
7
|
|
|
3
8
|
// Seed-file search over the file-level graph — the "find files about X" primitive
|
|
4
9
|
// keryx lacked (agents kept mis-reaching for `gdgraph query "<nl>"`, which only
|
|
5
|
-
// does cycles/orphans). Deterministic and offline: rank file nodes by how
|
|
6
|
-
// query
|
|
7
|
-
// (dependents)
|
|
10
|
+
// does cycles/orphans). Deterministic and offline: rank file nodes by how much
|
|
11
|
+
// each matched query term actually NARROWS this corpus, with a basename boost,
|
|
12
|
+
// tie-broken by fan-in (dependents). The result is a short seed list to feed
|
|
8
13
|
// into `gdgraph affected <file>`.
|
|
14
|
+
//
|
|
15
|
+
// WHY TERM WEIGHTING, AND WHAT IT REPLACED (AC6 / AFC-M04, flow 235)
|
|
16
|
+
//
|
|
17
|
+
// The first version scored `matched.length * 10 + baseHits * 5`: every matched
|
|
18
|
+
// term was worth the same. Measured against the live index on 2026-09-08,
|
|
19
|
+
// `keryx gdgraph find "wiki search ranking by section"` returned sixteen files
|
|
20
|
+
// tied at score 15, led by `src/memory/search.ts` (dependents 12) — while
|
|
21
|
+
// `src/wiki/ask.ts`, the one file the question was about, scored 10 and fell
|
|
22
|
+
// off the end of the 20-item list.
|
|
23
|
+
//
|
|
24
|
+
// The cause was not that fan-in is consulted. It was that a hit on `search` —
|
|
25
|
+
// a term in nearly every path in this repository, and therefore carrying no
|
|
26
|
+
// information — counted exactly as much as a hit on `wiki`, which carries a
|
|
27
|
+
// lot. Once the scores mass-tied, global popularity chose the answer. That is
|
|
28
|
+
// "важный связанный код уступает нерелевантной глобальной популярности"
|
|
29
|
+
// verbatim.
|
|
30
|
+
//
|
|
31
|
+
// So a term is now weighted by its inverse document frequency OVER THIS GRAPH:
|
|
32
|
+
// how few file paths contain it. A term present in every path scores zero and
|
|
33
|
+
// contributes nothing ("Zero-score не добавляет ballast", specification.md §5)
|
|
34
|
+
// — and a file whose only hits are such terms is skipped with a `continue`, not
|
|
35
|
+
// a `break`, so dropping ballast never truncates the scan.
|
|
36
|
+
//
|
|
37
|
+
// Fan-in stays exactly where it belongs: a TIE-BREAK between two files the
|
|
38
|
+
// query cannot otherwise separate. It is never added into the score, because a
|
|
39
|
+
// file's global importance is not evidence that it answers this question.
|
|
40
|
+
//
|
|
41
|
+
// No new search engine is introduced (AFC-M04 is explicit about that): this is
|
|
42
|
+
// the same path-and-symbol lexical match it always was, with the term weights
|
|
43
|
+
// it should have had.
|
|
44
|
+
|
|
45
|
+
/**
|
|
46
|
+
* A term present in more than this fraction of the corpus is "ubiquitous": it
|
|
47
|
+
* cannot discriminate, so a candidate matching only such terms is
|
|
48
|
+
* `insufficient-evidence` rather than a real hit.
|
|
49
|
+
*
|
|
50
|
+
* 0.5 is not chosen freshly here — it is `UBIQUITY_FRACTION` from
|
|
51
|
+
* `src/wiki/section-index.ts`, kept identical on purpose so the two retrieval
|
|
52
|
+
* surfaces call the same query "too weak to be evidence" at the same threshold.
|
|
53
|
+
*/
|
|
54
|
+
export const UBIQUITY_FRACTION = 0.5;
|
|
55
|
+
|
|
56
|
+
/**
|
|
57
|
+
* Below this many documents, document frequency is not evidence of anything.
|
|
58
|
+
*
|
|
59
|
+
* In a four-file corpus "this term is in three of them" says nothing about the
|
|
60
|
+
* term — and in a one-file corpus every term is trivially "in 100% of the
|
|
61
|
+
* corpus", which would weight the only real answer to zero and report
|
|
62
|
+
* `no-match` for a graph that plainly contains the file. So a corpus smaller
|
|
63
|
+
* than this keeps the original flat scoring (`matched × 10 + basename × 5`)
|
|
64
|
+
* byte for byte, and declares no term ubiquitous.
|
|
65
|
+
*/
|
|
66
|
+
export const MIN_WEIGHTED_CORPUS = 12;
|
|
9
67
|
|
|
10
68
|
export interface FindResult {
|
|
11
69
|
path: string;
|
|
70
|
+
/** A ranking score. Not a probability, not a percentage (specification.md §3). */
|
|
12
71
|
score: number;
|
|
13
72
|
matched: string[];
|
|
73
|
+
/**
|
|
74
|
+
* The subset of `matched` that actually narrows this corpus. This is the
|
|
75
|
+
* evidence: a candidate whose `discriminating` is empty matched only on
|
|
76
|
+
* noise.
|
|
77
|
+
*/
|
|
78
|
+
discriminating: string[];
|
|
14
79
|
dependents: number;
|
|
80
|
+
/** Why this candidate is here, in one line — computed once, not re-derived by each renderer. */
|
|
81
|
+
reason: string;
|
|
15
82
|
}
|
|
16
83
|
|
|
17
84
|
export interface SymbolFindResult {
|
|
@@ -22,6 +89,8 @@ export interface SymbolFindResult {
|
|
|
22
89
|
startLine: number;
|
|
23
90
|
score: number;
|
|
24
91
|
matched: string[];
|
|
92
|
+
discriminating: string[];
|
|
93
|
+
reason: string;
|
|
25
94
|
}
|
|
26
95
|
|
|
27
96
|
const STOP = new Set([
|
|
@@ -40,68 +109,491 @@ export function tokenize(query: string): string[] {
|
|
|
40
109
|
];
|
|
41
110
|
}
|
|
42
111
|
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
112
|
+
/**
|
|
113
|
+
* How much a term narrows a corpus of `total` documents, `df` of which contain
|
|
114
|
+
* it. Zero when the term is in every document — it distinguishes nothing there,
|
|
115
|
+
* so it must contribute nothing.
|
|
116
|
+
*/
|
|
117
|
+
function termWeight(df: number, total: number): number {
|
|
118
|
+
if (total === 0 || df <= 0) {
|
|
119
|
+
return 0;
|
|
120
|
+
}
|
|
121
|
+
if (total < MIN_WEIGHTED_CORPUS) {
|
|
122
|
+
return 1;
|
|
123
|
+
}
|
|
124
|
+
return Math.log((total + 1) / (df + 1));
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
/**
|
|
128
|
+
* Where a new word starts inside an identifier: index 0, after any separator,
|
|
129
|
+
* at a camelCase hump, at the tail of an acronym (`XMLHttp` → `Http`), and at a
|
|
130
|
+
* letter→digit transition.
|
|
131
|
+
*
|
|
132
|
+
* WHY THIS EXISTS (AC6, flow 235)
|
|
133
|
+
*
|
|
134
|
+
* `findSymbols` matched a query term against the whole lowercased name with
|
|
135
|
+
* `includes`, which lets a term hide in the middle of an unrelated identifier.
|
|
136
|
+
* Measured at the real command line on 2026-09-08:
|
|
137
|
+
* `keryx gdgraph find "kubernetes helm chart"` — a query about nothing in this
|
|
138
|
+
* repository — returned `DispatchArtifactRef`, because "chart" sits inside
|
|
139
|
+
* "dispat-CHART-ifact". That is a nonsense query rendered as a successful
|
|
140
|
+
* search, which is precisely the failure AC5 forbids and AC6's "explainable
|
|
141
|
+
* candidates" rules out: no reader of that name would say it contains the word.
|
|
142
|
+
*/
|
|
143
|
+
function wordStarts(name: string): number[] {
|
|
144
|
+
const isUpper = (c: string): boolean => c >= "A" && c <= "Z";
|
|
145
|
+
const isLower = (c: string): boolean => c >= "a" && c <= "z";
|
|
146
|
+
const isDigit = (c: string): boolean => c >= "0" && c <= "9";
|
|
147
|
+
const isSeparator = (c: string): boolean => !isUpper(c) && !isLower(c) && !isDigit(c);
|
|
148
|
+
|
|
149
|
+
const starts = [0];
|
|
150
|
+
for (let i = 1; i < name.length; i += 1) {
|
|
151
|
+
const previous = name[i - 1] ?? "";
|
|
152
|
+
const current = name[i] ?? "";
|
|
153
|
+
const next = name[i + 1] ?? "";
|
|
154
|
+
if (
|
|
155
|
+
isSeparator(previous) ||
|
|
156
|
+
(isUpper(current) && !isUpper(previous)) ||
|
|
157
|
+
(isUpper(current) && isUpper(previous) && isLower(next)) ||
|
|
158
|
+
(isDigit(current) && !isDigit(previous))
|
|
159
|
+
) {
|
|
160
|
+
starts.push(i);
|
|
161
|
+
}
|
|
162
|
+
}
|
|
163
|
+
return starts;
|
|
164
|
+
}
|
|
165
|
+
|
|
166
|
+
/**
|
|
167
|
+
* Does `term` (already lowercased) occur in `name` at a word boundary?
|
|
168
|
+
*
|
|
169
|
+
* A prefix of the whole name still counts, so `find "clonePipeline"` keeps
|
|
170
|
+
* returning `clonePipelineDeep`.
|
|
171
|
+
*/
|
|
172
|
+
function matchesAtWordBoundary(name: string, term: string): boolean {
|
|
173
|
+
const lower = name.toLowerCase();
|
|
174
|
+
if (lower.length !== name.length) {
|
|
175
|
+
// A locale-folding name (non-ASCII) shifts the indices; fall back rather
|
|
176
|
+
// than reporting a boundary that may not be one.
|
|
177
|
+
return lower.includes(term);
|
|
178
|
+
}
|
|
179
|
+
for (const start of wordStarts(name)) {
|
|
180
|
+
if (lower.startsWith(term, start)) {
|
|
181
|
+
return true;
|
|
182
|
+
}
|
|
183
|
+
}
|
|
184
|
+
return false;
|
|
185
|
+
}
|
|
186
|
+
|
|
187
|
+
function documentFrequencies(
|
|
188
|
+
terms: string[],
|
|
189
|
+
documents: string[],
|
|
190
|
+
matches: (document: string, term: string) => boolean = (document, term) =>
|
|
191
|
+
document.includes(term),
|
|
192
|
+
): Map<string, number> {
|
|
193
|
+
const frequencies = new Map<string, number>();
|
|
194
|
+
for (const term of terms) {
|
|
195
|
+
let count = 0;
|
|
196
|
+
for (const document of documents) {
|
|
197
|
+
if (matches(document, term)) {
|
|
198
|
+
count += 1;
|
|
199
|
+
}
|
|
200
|
+
}
|
|
201
|
+
frequencies.set(term, count);
|
|
202
|
+
}
|
|
203
|
+
return frequencies;
|
|
204
|
+
}
|
|
205
|
+
|
|
206
|
+
function ubiquitousOf(
|
|
207
|
+
terms: string[],
|
|
208
|
+
frequencies: Map<string, number>,
|
|
209
|
+
total: number,
|
|
210
|
+
): string[] {
|
|
211
|
+
if (total < MIN_WEIGHTED_CORPUS) {
|
|
212
|
+
return [];
|
|
213
|
+
}
|
|
214
|
+
return terms.filter((term) => (frequencies.get(term) ?? 0) / total > UBIQUITY_FRACTION);
|
|
215
|
+
}
|
|
216
|
+
|
|
217
|
+
function reasonFor(
|
|
218
|
+
matched: string[],
|
|
219
|
+
discriminating: string[],
|
|
220
|
+
boost: string | undefined,
|
|
221
|
+
dependents: number | undefined,
|
|
222
|
+
): string {
|
|
223
|
+
const parts = [`matched ${matched.join(", ")}`];
|
|
224
|
+
parts.push(
|
|
225
|
+
discriminating.length > 0
|
|
226
|
+
? `narrowing on ${discriminating.join(", ")}`
|
|
227
|
+
: "no narrowing term — every hit is corpus-wide",
|
|
228
|
+
);
|
|
229
|
+
if (boost) {
|
|
230
|
+
parts.push(boost);
|
|
231
|
+
}
|
|
232
|
+
if (dependents !== undefined) {
|
|
233
|
+
// Named as a tie-break in the reason itself, because "this file has a lot
|
|
234
|
+
// of dependents" is exactly the wrong thing for a reader to mistake for
|
|
235
|
+
// evidence that it answers the question.
|
|
236
|
+
parts.push(`fan-in ${dependents} (tie-break only)`);
|
|
237
|
+
}
|
|
238
|
+
return parts.join("; ");
|
|
239
|
+
}
|
|
240
|
+
|
|
241
|
+
/**
|
|
242
|
+
* What is on the PAGE, when that differs from what the scan found.
|
|
243
|
+
*
|
|
244
|
+
* THE THIRD AND FOURTH INSTANCES (flow 235, T15)
|
|
245
|
+
*
|
|
246
|
+
* Separating matching from ranking (see the note below) stopped a display
|
|
247
|
+
* decision from choosing a `code`. It did not stop one from writing the PROSE.
|
|
248
|
+
* Reproduced on a 100-file corpus with 40 genuine matches:
|
|
249
|
+
*
|
|
250
|
+
* code: ok
|
|
251
|
+
* reason: "20 files and 0 symbols matched." // 40 did
|
|
252
|
+
*
|
|
253
|
+
* and, through the MCP boundary, `fileLimit: 0` produced `"0 files and 0
|
|
254
|
+
* symbols matched."` — a page size rendered as a fact about the corpus, at
|
|
255
|
+
* `code: ok`. The same shape sat in `insufficient-evidence`, whose tail chose
|
|
256
|
+
* between "the ranking below is not evidence" and "Every match scored zero" by
|
|
257
|
+
* asking how long the PAGE was: at `fileLimit: 0` it asserted every match
|
|
258
|
+
* scored zero while sixty matches scored ~5.
|
|
259
|
+
*
|
|
260
|
+
* The rule that closes both: a reason may state what MATCHED only from the
|
|
261
|
+
* scan, and what is SHOWN only as a separate, explicitly-labelled clause. A
|
|
262
|
+
* reader must always be able to tell "40 matched, showing 20" from "20
|
|
263
|
+
* matched"; the two can never be collapsed into one number again.
|
|
264
|
+
*/
|
|
265
|
+
function displayNote(
|
|
266
|
+
matchedFiles: number,
|
|
267
|
+
matchedSymbols: number,
|
|
268
|
+
shownFiles: number,
|
|
269
|
+
shownSymbols: number,
|
|
270
|
+
): string {
|
|
271
|
+
const cut: string[] = [];
|
|
272
|
+
if (shownFiles !== matchedFiles) {
|
|
273
|
+
cut.push(`${shownFiles} of ${matchedFiles} files`);
|
|
274
|
+
}
|
|
275
|
+
if (shownSymbols !== matchedSymbols) {
|
|
276
|
+
cut.push(`${shownSymbols} of ${matchedSymbols} symbols`);
|
|
277
|
+
}
|
|
278
|
+
if (cut.length === 0) {
|
|
279
|
+
return "";
|
|
280
|
+
}
|
|
281
|
+
return (
|
|
282
|
+
` Showing ${cut.join(" and ")} — the rest are below the display limit or scored zero.` +
|
|
283
|
+
" That is a ranking decision about this page, not a claim about the corpus."
|
|
284
|
+
);
|
|
285
|
+
}
|
|
286
|
+
|
|
287
|
+
// MATCHING vs RANKING — WHY THESE ARE TWO STEPS (flow 235, T14)
|
|
288
|
+
//
|
|
289
|
+
// They used to be one loop, and that produced a false statement about the
|
|
290
|
+
// world. Reproduced at the real command line on 2026-09-08, in a temp project
|
|
291
|
+
// of 14 files ALL under `alpha/`, after a real `keryx gdgraph build`:
|
|
292
|
+
//
|
|
293
|
+
// $ keryx gdgraph find "alpha" --json
|
|
294
|
+
// { "code": "no-match",
|
|
295
|
+
// "reason": "the search completed over 14 indexed files; no path or symbol
|
|
296
|
+
// contains any of: alpha.",
|
|
297
|
+
// "ubiquitousTerms": ["alpha"], "files": [] } exit=0
|
|
298
|
+
//
|
|
299
|
+
// Every one of the fourteen paths contains `alpha`. The payload even said so —
|
|
300
|
+
// `ubiquitousTerms: ["alpha"]` sat in the same object as a reason denying it.
|
|
301
|
+
//
|
|
302
|
+
// The mechanism: `alpha` is in ALL 14 documents, so `termWeight` returns
|
|
303
|
+
// `log(15/15) = 0`, every file scored 0, and the ballast filter dropped every
|
|
304
|
+
// one of them. `findCandidates` then read the empty list as "nothing matched"
|
|
305
|
+
// and made a claim about the corpus that the corpus contradicts — at exit 0,
|
|
306
|
+
// sending the caller off to a text search they did not need.
|
|
307
|
+
//
|
|
308
|
+
// Dropping zero-weight ballast is a RANKING decision: it says "this file is not
|
|
309
|
+
// worth showing you", never "this file does not exist". So the scan below
|
|
310
|
+
// collects every file that matched, score included and nothing filtered, and
|
|
311
|
+
// ranking is a separate step applied only to what gets DISPLAYED. The
|
|
312
|
+
// classifier reads the scan, never the ranked page — which is also why slicing
|
|
313
|
+
// to `limit` can no longer decide a code either.
|
|
314
|
+
|
|
315
|
+
interface FileScan {
|
|
316
|
+
/** Every file whose path contains ≥1 query term. Unfiltered, unsorted, unsliced. */
|
|
317
|
+
readonly matches: FindResult[];
|
|
318
|
+
readonly ubiquitous: string[];
|
|
319
|
+
readonly corpusSize: number;
|
|
320
|
+
}
|
|
321
|
+
|
|
322
|
+
function scanFiles(graph: GraphData, query: string): FileScan {
|
|
323
|
+
const terms = tokenize(query);
|
|
324
|
+
const files = graph.nodes.filter((node) => node.kind === "file");
|
|
325
|
+
const paths = files.map((node) => node.path.toLowerCase());
|
|
326
|
+
const frequencies = documentFrequencies(terms, paths);
|
|
327
|
+
const ubiquitous = ubiquitousOf(terms, frequencies, paths.length);
|
|
328
|
+
const ubiquitousSet = new Set(ubiquitous);
|
|
329
|
+
const scan = { ubiquitous, corpusSize: files.length };
|
|
330
|
+
|
|
331
|
+
if (terms.length === 0) {
|
|
332
|
+
return { matches: [], ...scan };
|
|
333
|
+
}
|
|
334
|
+
|
|
335
|
+
// fan-in (dependents) per node id: how many edges point at it.
|
|
336
|
+
const fanIn = new Map<string, number>();
|
|
337
|
+
for (const edge of graph.edges) {
|
|
338
|
+
fanIn.set(edge.to, (fanIn.get(edge.to) ?? 0) + 1);
|
|
339
|
+
}
|
|
340
|
+
|
|
341
|
+
const matches: FindResult[] = [];
|
|
342
|
+
for (const node of files) {
|
|
343
|
+
const lowerPath = node.path.toLowerCase();
|
|
344
|
+
const base = lowerPath.split("/").pop() ?? lowerPath;
|
|
345
|
+
const matched = terms.filter((t) => lowerPath.includes(t));
|
|
346
|
+
if (matched.length === 0) {
|
|
347
|
+
continue;
|
|
348
|
+
}
|
|
349
|
+
const basenameHits = matched.filter((t) => base.includes(t));
|
|
350
|
+
let score = 0;
|
|
351
|
+
for (const term of matched) {
|
|
352
|
+
const weight = termWeight(frequencies.get(term) ?? 0, paths.length);
|
|
353
|
+
// The basename boost is weighted too: a filename hit on a term that
|
|
354
|
+
// narrows nothing is still worth nothing.
|
|
355
|
+
score += weight * 10 + (basenameHits.includes(term) ? weight * 5 : 0);
|
|
356
|
+
}
|
|
357
|
+
const dependents = fanIn.get(node.id) ?? 0;
|
|
358
|
+
const discriminating = matched.filter((term) => !ubiquitousSet.has(term));
|
|
359
|
+
matches.push({
|
|
360
|
+
path: node.path,
|
|
361
|
+
score,
|
|
362
|
+
matched,
|
|
363
|
+
discriminating,
|
|
364
|
+
dependents,
|
|
365
|
+
reason: reasonFor(
|
|
366
|
+
matched,
|
|
367
|
+
discriminating,
|
|
368
|
+
basenameHits.length > 0 ? `in the filename (${basenameHits.join(", ")})` : undefined,
|
|
369
|
+
dependents,
|
|
370
|
+
),
|
|
371
|
+
});
|
|
372
|
+
}
|
|
373
|
+
return { matches, ...scan };
|
|
374
|
+
}
|
|
375
|
+
|
|
376
|
+
/**
|
|
377
|
+
* The ranking half: drop zero-score ballast, order, and cut to `limit`.
|
|
378
|
+
*
|
|
379
|
+
* Every decision here is about what is worth SHOWING. None of it may reach the
|
|
380
|
+
* classifier — see the note above `scanFiles`.
|
|
381
|
+
*/
|
|
382
|
+
function rankFiles(matches: readonly FindResult[], limit = 20): FindResult[] {
|
|
383
|
+
return matches
|
|
384
|
+
.filter((match) => match.score > 0)
|
|
385
|
+
.sort(
|
|
386
|
+
// Score first, fan-in ONLY as a tie-break: global popularity may separate
|
|
387
|
+
// two files the query cannot, and may never outrank the query itself.
|
|
388
|
+
(a, b) => b.score - a.score || b.dependents - a.dependents || a.path.localeCompare(b.path),
|
|
389
|
+
)
|
|
390
|
+
.slice(0, limit);
|
|
391
|
+
}
|
|
392
|
+
|
|
393
|
+
// Match symbol nodes by name — the precise half of `find` when the symbol layer
|
|
394
|
+
// is active. Exact name match is boosted so `find "clonePipeline"` returns the
|
|
395
|
+
// definition, not just path hits. Term weighting is computed over symbol NAMES
|
|
396
|
+
// (their own corpus), not over paths: a term common among file paths can still
|
|
397
|
+
// be a precise symbol name and vice versa.
|
|
398
|
+
function scanSymbols(graph: GraphData, query: string): SymbolFindResult[] {
|
|
47
399
|
const terms = tokenize(query);
|
|
48
400
|
const symbols = graph.symbols ?? [];
|
|
49
401
|
if (terms.length === 0 || symbols.length === 0) {
|
|
50
402
|
return [];
|
|
51
403
|
}
|
|
52
404
|
|
|
53
|
-
|
|
405
|
+
// Frequencies are counted with the SAME predicate the match uses, so a term
|
|
406
|
+
// cannot be rare by one rule and common by another.
|
|
407
|
+
const names = symbols.map((symbol) => symbol.name);
|
|
408
|
+
const frequencies = documentFrequencies(terms, names, matchesAtWordBoundary);
|
|
409
|
+
const ubiquitous = new Set(ubiquitousOf(terms, frequencies, names.length));
|
|
410
|
+
|
|
411
|
+
const matches: SymbolFindResult[] = [];
|
|
54
412
|
for (const symbol of symbols) {
|
|
55
413
|
const nameLower = symbol.name.toLowerCase();
|
|
56
|
-
const matched = terms.filter((t) =>
|
|
414
|
+
const matched = terms.filter((t) => matchesAtWordBoundary(symbol.name, t));
|
|
57
415
|
if (matched.length === 0) {
|
|
58
416
|
continue;
|
|
59
417
|
}
|
|
60
418
|
const exactBonus = terms.some((t) => nameLower === t) ? 20 : 0;
|
|
61
|
-
|
|
419
|
+
let score = exactBonus;
|
|
420
|
+
for (const term of matched) {
|
|
421
|
+
score += termWeight(frequencies.get(term) ?? 0, names.length) * 10;
|
|
422
|
+
}
|
|
423
|
+
const discriminating = matched.filter((term) => !ubiquitous.has(term));
|
|
424
|
+
matches.push({
|
|
62
425
|
id: symbol.id,
|
|
63
426
|
name: symbol.name,
|
|
64
427
|
kind: symbol.kind,
|
|
65
428
|
path: symbol.path,
|
|
66
429
|
startLine: symbol.startLine,
|
|
67
|
-
score
|
|
430
|
+
score,
|
|
68
431
|
matched,
|
|
432
|
+
discriminating,
|
|
433
|
+
reason: reasonFor(
|
|
434
|
+
matched,
|
|
435
|
+
discriminating,
|
|
436
|
+
exactBonus > 0 ? "exact symbol name" : undefined,
|
|
437
|
+
undefined,
|
|
438
|
+
),
|
|
69
439
|
});
|
|
70
440
|
}
|
|
71
|
-
|
|
72
|
-
|
|
441
|
+
return matches;
|
|
442
|
+
}
|
|
443
|
+
|
|
444
|
+
/** Ranking only — a name whose hits are all corpus-wide, with no exact match, is not worth showing. */
|
|
445
|
+
function rankSymbols(matches: readonly SymbolFindResult[], limit = 15): SymbolFindResult[] {
|
|
446
|
+
return matches
|
|
447
|
+
.filter((match) => match.score > 0)
|
|
448
|
+
.sort(
|
|
449
|
+
(a, b) => b.score - a.score || a.path.localeCompare(b.path) || a.startLine - b.startLine,
|
|
450
|
+
)
|
|
451
|
+
.slice(0, limit);
|
|
452
|
+
}
|
|
453
|
+
|
|
454
|
+
export function findSymbols(graph: GraphData, query: string, limit = 15): SymbolFindResult[] {
|
|
455
|
+
return rankSymbols(scanSymbols(graph, query), limit);
|
|
73
456
|
}
|
|
74
457
|
|
|
75
458
|
export function findNodes(graph: GraphData, query: string, limit = 20): FindResult[] {
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
459
|
+
return rankFiles(scanFiles(graph, query).matches, limit);
|
|
460
|
+
}
|
|
461
|
+
|
|
462
|
+
/**
|
|
463
|
+
* `find` with an AC5 outcome code attached (AFC-M03).
|
|
464
|
+
*
|
|
465
|
+
* The measured defect: `keryx gdgraph find` printed "No files or symbols
|
|
466
|
+
* matched" for a genuine no-match AND for a directory with no graph at all,
|
|
467
|
+
* and printed a confident ranked list for a query whose only matching term was
|
|
468
|
+
* in most of the corpus. Four different situations, one indistinguishable
|
|
469
|
+
* answer, all at exit 0.
|
|
470
|
+
*
|
|
471
|
+
* The classification order matters and is not arbitrary: an index that cannot
|
|
472
|
+
* answer must be reported before "nothing matched", because "nothing matched"
|
|
473
|
+
* is a claim about the corpus and an empty index has no standing to make it.
|
|
474
|
+
*
|
|
475
|
+
* For the same reason (T14) every test below reads the SCAN — the full set of
|
|
476
|
+
* files and symbols that matched — and never `foundFiles`/`foundSymbols`, which
|
|
477
|
+
* are the ranked, ballast-free, `limit`-sliced page meant for a reader. A
|
|
478
|
+
* display decision must never be able to turn a corpus that contains your terms
|
|
479
|
+
* into a `no-match` that says it does not.
|
|
480
|
+
*
|
|
481
|
+
* T15 extends that from the conditions to the REASONS. `foundFiles`/
|
|
482
|
+
* `foundSymbols` may appear in a reason only through `displayNote`, which
|
|
483
|
+
* labels the number as a page size; every count a reason states as a fact about
|
|
484
|
+
* the corpus comes from the scan. `find-display-truth.test.ts` pins this.
|
|
485
|
+
*/
|
|
486
|
+
export interface FindOutcome {
|
|
487
|
+
readonly code: RetrievalCode;
|
|
488
|
+
readonly reason: string;
|
|
489
|
+
readonly nextActions: readonly string[];
|
|
490
|
+
readonly files: FindResult[];
|
|
491
|
+
readonly symbols: SymbolFindResult[];
|
|
492
|
+
readonly queryTerms: string[];
|
|
493
|
+
readonly ubiquitousTerms: string[];
|
|
494
|
+
}
|
|
495
|
+
|
|
496
|
+
export interface FindOptions {
|
|
497
|
+
readonly fileLimit?: number;
|
|
498
|
+
readonly symbolLimit?: number;
|
|
499
|
+
}
|
|
500
|
+
|
|
501
|
+
export function findCandidates(
|
|
502
|
+
graph: GraphData,
|
|
503
|
+
query: string,
|
|
504
|
+
options: FindOptions = {},
|
|
505
|
+
): FindOutcome {
|
|
506
|
+
const queryTerms = tokenize(query);
|
|
507
|
+
const scan = scanFiles(graph, query);
|
|
508
|
+
const symbolMatches = scanSymbols(graph, query);
|
|
509
|
+
const ubiquitousTerms = scan.ubiquitous;
|
|
510
|
+
|
|
511
|
+
const empty = { files: [], symbols: [], queryTerms, ubiquitousTerms };
|
|
512
|
+
|
|
513
|
+
if (query.trim().length === 0) {
|
|
514
|
+
return outcome("invalid-input", "no query was given — `find` needs at least one term.", empty);
|
|
515
|
+
}
|
|
516
|
+
if (scan.corpusSize === 0) {
|
|
517
|
+
return outcome(
|
|
518
|
+
"index-incomplete",
|
|
519
|
+
"the graph index holds no file nodes — it was never built here, or its storage is unreadable. " +
|
|
520
|
+
"This is not a statement about whether the code exists.",
|
|
521
|
+
empty,
|
|
522
|
+
);
|
|
523
|
+
}
|
|
524
|
+
if (queryTerms.length === 0) {
|
|
525
|
+
// Aligned word-for-word with the wiki lane's classifier
|
|
526
|
+
// (`src/wiki/ask.ts`) so the same situation gets the same code and the
|
|
527
|
+
// same explanation on both retrieval surfaces.
|
|
528
|
+
return outcome(
|
|
529
|
+
"no-match",
|
|
530
|
+
"the query carries no content terms — every token is a stop word or shorter than two characters.",
|
|
531
|
+
empty,
|
|
532
|
+
);
|
|
79
533
|
}
|
|
80
534
|
|
|
81
|
-
|
|
82
|
-
const
|
|
83
|
-
|
|
84
|
-
|
|
535
|
+
const foundFiles = rankFiles(scan.matches, options.fileLimit);
|
|
536
|
+
const foundSymbols = rankSymbols(symbolMatches, options.symbolLimit);
|
|
537
|
+
const found = { files: foundFiles, symbols: foundSymbols, queryTerms, ubiquitousTerms };
|
|
538
|
+
|
|
539
|
+
const matchCount = scan.matches.length + symbolMatches.length;
|
|
540
|
+
if (matchCount === 0) {
|
|
541
|
+
return outcome(
|
|
542
|
+
"no-match",
|
|
543
|
+
`the search completed over ${scan.corpusSize} indexed files; no path or symbol contains any of: ${queryTerms.join(", ")}.`,
|
|
544
|
+
found,
|
|
545
|
+
);
|
|
85
546
|
}
|
|
86
547
|
|
|
87
|
-
const
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
const
|
|
99
|
-
|
|
100
|
-
|
|
548
|
+
const note = displayNote(
|
|
549
|
+
scan.matches.length,
|
|
550
|
+
symbolMatches.length,
|
|
551
|
+
foundFiles.length,
|
|
552
|
+
foundSymbols.length,
|
|
553
|
+
);
|
|
554
|
+
|
|
555
|
+
const anyDiscriminating =
|
|
556
|
+
scan.matches.some((file) => file.discriminating.length > 0) ||
|
|
557
|
+
symbolMatches.some((symbol) => symbol.discriminating.length > 0);
|
|
558
|
+
if (!anyDiscriminating) {
|
|
559
|
+
const noise = ubiquitousTerms.join(", ") || queryTerms.join(", ");
|
|
560
|
+
// "Did anything score?" is a question about the SCAN. Asking the page
|
|
561
|
+
// instead (`foundFiles.length + foundSymbols.length > 0`) let `fileLimit: 0`
|
|
562
|
+
// turn sixty scoring matches into "Every match scored zero" — see
|
|
563
|
+
// `displayNote`.
|
|
564
|
+
const anyScored =
|
|
565
|
+
scan.matches.some((file) => file.score > 0) ||
|
|
566
|
+
symbolMatches.some((symbol) => symbol.score > 0);
|
|
567
|
+
return outcome(
|
|
568
|
+
"insufficient-evidence",
|
|
569
|
+
`${matchCount} candidates matched, every one of them only on ${noise} — ` +
|
|
570
|
+
"a term that appears across this corpus and so narrows nothing" +
|
|
571
|
+
(anyScored
|
|
572
|
+
? ", which is why the ranking below is not evidence for this question."
|
|
573
|
+
: ". Every match scored zero, so no ranking is shown; this is not a claim " +
|
|
574
|
+
"that the corpus lacks your terms — it is that they cannot separate anything in it.") +
|
|
575
|
+
note,
|
|
576
|
+
found,
|
|
577
|
+
);
|
|
101
578
|
}
|
|
102
579
|
|
|
103
|
-
|
|
104
|
-
|
|
580
|
+
return outcome(
|
|
581
|
+
"ok",
|
|
582
|
+
`${scan.matches.length} files and ${symbolMatches.length} symbols matched.` + note,
|
|
583
|
+
found,
|
|
105
584
|
);
|
|
106
|
-
|
|
585
|
+
}
|
|
586
|
+
|
|
587
|
+
function outcome(
|
|
588
|
+
code: RetrievalCode,
|
|
589
|
+
reason: string,
|
|
590
|
+
rest: {
|
|
591
|
+
files: FindResult[];
|
|
592
|
+
symbols: SymbolFindResult[];
|
|
593
|
+
queryTerms: string[];
|
|
594
|
+
ubiquitousTerms: string[];
|
|
595
|
+
},
|
|
596
|
+
): FindOutcome {
|
|
597
|
+
const nextActions = RETRIEVAL_NEXT_ACTIONS[code].slice(0, MAX_NEXT_ACTIONS);
|
|
598
|
+
return { code, reason, nextActions, ...rest };
|
|
107
599
|
}
|