@mrciphersmith/keryx 0.2.83 → 0.2.84

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. package/dist/cli.js +28463 -19413
  2. package/dist/core.js +25751 -0
  3. package/package.json +15 -2
  4. package/src/gdgraph/dangling.ts +204 -0
  5. package/src/gdgraph/find.ts +529 -37
  6. package/src/gdgraph/repomap.ts +140 -12
  7. package/src/gdgraph/staleness.ts +22 -9
  8. package/src/gdgraph/symbol.ts +45 -6
  9. package/src/gdgraph/treesitter/extract.ts +153 -5
  10. package/src/gdgraph/wiki-layer.ts +32 -1
  11. package/src/gdskills/bundled/skills/core/reviewer-skill-creator/SKILL.md +29 -0
  12. package/src/gdskills/bundled/skills/orchestration/code-verifier/SKILL.codex.md +1 -1
  13. package/src/gdskills/bundled/skills/orchestration/code-verifier/SKILL.cursor.md +1 -1
  14. package/src/gdskills/bundled/skills/orchestration/code-verifier/SKILL.md +1 -1
  15. package/src/gdskills/bundled/skills/orchestration/code-verifier/SKILL.opencode.md +1 -1
  16. package/src/gdskills/bundled/skills/orchestration/code-verifier/SKILL.zed.md +1 -1
  17. package/src/gdskills/bundled/skills/orchestration/flow-orchestrator/SKILL.md +34 -6
  18. package/src/gdskills/bundled/skills/orchestration/job-orchestrator/SKILL.codex.md +1 -1
  19. package/src/gdskills/bundled/skills/orchestration/job-orchestrator/SKILL.cursor.md +1 -1
  20. package/src/gdskills/bundled/skills/orchestration/job-orchestrator/SKILL.md +1 -1
  21. package/src/gdskills/bundled/skills/orchestration/job-orchestrator/SKILL.opencode.md +1 -1
  22. package/src/gdskills/bundled/skills/orchestration/job-orchestrator/SKILL.zed.md +1 -1
  23. package/src/gdskills/bundled/skills/quality/security-audit/SKILL.codex.md +52 -5
  24. package/src/gdskills/bundled/skills/quality/security-audit/SKILL.cursor.md +52 -5
  25. package/src/gdskills/bundled/skills/quality/security-audit/SKILL.md +52 -5
  26. package/src/gdgraph/affected.test.ts +0 -133
  27. package/src/gdgraph/build-integrity.test.ts +0 -193
  28. package/src/gdgraph/build-lang.test.ts +0 -406
  29. package/src/gdgraph/build.test.ts +0 -120
  30. package/src/gdgraph/config.test.ts +0 -47
  31. package/src/gdgraph/core-sources.test.ts +0 -99
  32. package/src/gdgraph/fallback.test.ts +0 -153
  33. package/src/gdgraph/find.test.ts +0 -78
  34. package/src/gdgraph/import-kind.test.ts +0 -525
  35. package/src/gdgraph/path.test.ts +0 -56
  36. package/src/gdgraph/repomap.test.ts +0 -110
  37. package/src/gdgraph/service.test.ts +0 -89
  38. package/src/gdgraph/staleness.test.ts +0 -208
  39. package/src/gdgraph/symbol.test.ts +0 -89
  40. package/src/gdgraph/symbols-capability.test.ts +0 -138
  41. package/src/gdgraph/treesitter/adapter.test.ts +0 -496
  42. package/src/gdgraph/treesitter/extract.test.ts +0 -278
  43. package/src/gdgraph/treesitter/no-treesitter-import.test.ts +0 -51
  44. package/src/gdgraph/treesitter/real-grammar-fixture.test.ts +0 -71
  45. package/src/gdgraph/treesitter/resolve-calls.test.ts +0 -38
  46. package/src/gdgraph/wiki-layer-no-git.test.ts +0 -73
  47. package/src/gdgraph/wiki-layer.test.ts +0 -211
@@ -1,17 +1,84 @@
1
1
  import type { GraphData } from "./types";
2
+ import {
3
+ MAX_NEXT_ACTIONS,
4
+ RETRIEVAL_NEXT_ACTIONS,
5
+ type RetrievalCode,
6
+ } from "../lib/retrieval-codes";
2
7
 
3
8
  // Seed-file search over the file-level graph — the "find files about X" primitive
4
9
  // keryx lacked (agents kept mis-reaching for `gdgraph query "<nl>"`, which only
5
- // does cycles/orphans). Deterministic and offline: rank file nodes by how many
6
- // query terms match their path, with a basename boost, tie-broken by fan-in
7
- // (dependents) as an importance proxy. The result is a short seed list to feed
10
+ // does cycles/orphans). Deterministic and offline: rank file nodes by how much
11
+ // each matched query term actually NARROWS this corpus, with a basename boost,
12
+ // tie-broken by fan-in (dependents). The result is a short seed list to feed
8
13
  // into `gdgraph affected <file>`.
14
+ //
15
+ // WHY TERM WEIGHTING, AND WHAT IT REPLACED (AC6 / AFC-M04, flow 235)
16
+ //
17
+ // The first version scored `matched.length * 10 + baseHits * 5`: every matched
18
+ // term was worth the same. Measured against the live index on 2026-09-08,
19
+ // `keryx gdgraph find "wiki search ranking by section"` returned sixteen files
20
+ // tied at score 15, led by `src/memory/search.ts` (dependents 12) — while
21
+ // `src/wiki/ask.ts`, the one file the question was about, scored 10 and fell
22
+ // off the end of the 20-item list.
23
+ //
24
+ // The cause was not that fan-in is consulted. It was that a hit on `search` —
25
+ // a term in nearly every path in this repository, and therefore carrying no
26
+ // information — counted exactly as much as a hit on `wiki`, which carries a
27
+ // lot. Once the scores mass-tied, global popularity chose the answer. That is
28
+ // "важный связанный код уступает нерелевантной глобальной популярности"
29
+ // verbatim.
30
+ //
31
+ // So a term is now weighted by its inverse document frequency OVER THIS GRAPH:
32
+ // how few file paths contain it. A term present in every path scores zero and
33
+ // contributes nothing ("Zero-score не добавляет ballast", specification.md §5)
34
+ // — and a file whose only hits are such terms is skipped with a `continue`, not
35
+ // a `break`, so dropping ballast never truncates the scan.
36
+ //
37
+ // Fan-in stays exactly where it belongs: a TIE-BREAK between two files the
38
+ // query cannot otherwise separate. It is never added into the score, because a
39
+ // file's global importance is not evidence that it answers this question.
40
+ //
41
+ // No new search engine is introduced (AFC-M04 is explicit about that): this is
42
+ // the same path-and-symbol lexical match it always was, with the term weights
43
+ // it should have had.
44
+
45
+ /**
46
+ * A term present in more than this fraction of the corpus is "ubiquitous": it
47
+ * cannot discriminate, so a candidate matching only such terms is
48
+ * `insufficient-evidence` rather than a real hit.
49
+ *
50
+ * 0.5 is not chosen freshly here — it is `UBIQUITY_FRACTION` from
51
+ * `src/wiki/section-index.ts`, kept identical on purpose so the two retrieval
52
+ * surfaces call the same query "too weak to be evidence" at the same threshold.
53
+ */
54
+ export const UBIQUITY_FRACTION = 0.5;
55
+
56
+ /**
57
+ * Below this many documents, document frequency is not evidence of anything.
58
+ *
59
+ * In a four-file corpus "this term is in three of them" says nothing about the
60
+ * term — and in a one-file corpus every term is trivially "in 100% of the
61
+ * corpus", which would weight the only real answer to zero and report
62
+ * `no-match` for a graph that plainly contains the file. So a corpus smaller
63
+ * than this keeps the original flat scoring (`matched × 10 + basename × 5`)
64
+ * byte for byte, and declares no term ubiquitous.
65
+ */
66
+ export const MIN_WEIGHTED_CORPUS = 12;
9
67
 
10
68
  export interface FindResult {
11
69
  path: string;
70
+ /** A ranking score. Not a probability, not a percentage (specification.md §3). */
12
71
  score: number;
13
72
  matched: string[];
73
+ /**
74
+ * The subset of `matched` that actually narrows this corpus. This is the
75
+ * evidence: a candidate whose `discriminating` is empty matched only on
76
+ * noise.
77
+ */
78
+ discriminating: string[];
14
79
  dependents: number;
80
+ /** Why this candidate is here, in one line — computed once, not re-derived by each renderer. */
81
+ reason: string;
15
82
  }
16
83
 
17
84
  export interface SymbolFindResult {
@@ -22,6 +89,8 @@ export interface SymbolFindResult {
22
89
  startLine: number;
23
90
  score: number;
24
91
  matched: string[];
92
+ discriminating: string[];
93
+ reason: string;
25
94
  }
26
95
 
27
96
  const STOP = new Set([
@@ -40,68 +109,491 @@ export function tokenize(query: string): string[] {
40
109
  ];
41
110
  }
42
111
 
43
- // Rank symbol nodes by name match — the precise half of `find` when the symbol
44
- // layer is active. Exact name match is boosted so `find "clonePipeline"` returns
45
- // the definition, not just path hits.
46
- export function findSymbols(graph: GraphData, query: string, limit = 15): SymbolFindResult[] {
112
+ /**
113
+ * How much a term narrows a corpus of `total` documents, `df` of which contain
114
+ * it. Zero when the term is in every document — it distinguishes nothing there,
115
+ * so it must contribute nothing.
116
+ */
117
+ function termWeight(df: number, total: number): number {
118
+ if (total === 0 || df <= 0) {
119
+ return 0;
120
+ }
121
+ if (total < MIN_WEIGHTED_CORPUS) {
122
+ return 1;
123
+ }
124
+ return Math.log((total + 1) / (df + 1));
125
+ }
126
+
127
+ /**
128
+ * Where a new word starts inside an identifier: index 0, after any separator,
129
+ * at a camelCase hump, at the tail of an acronym (`XMLHttp` → `Http`), and at a
130
+ * letter→digit transition.
131
+ *
132
+ * WHY THIS EXISTS (AC6, flow 235)
133
+ *
134
+ * `findSymbols` matched a query term against the whole lowercased name with
135
+ * `includes`, which lets a term hide in the middle of an unrelated identifier.
136
+ * Measured at the real command line on 2026-09-08:
137
+ * `keryx gdgraph find "kubernetes helm chart"` — a query about nothing in this
138
+ * repository — returned `DispatchArtifactRef`, because "chart" sits inside
139
+ * "dispat-CHART-ifact". That is a nonsense query rendered as a successful
140
+ * search, which is precisely the failure AC5 forbids and AC6's "explainable
141
+ * candidates" rules out: no reader of that name would say it contains the word.
142
+ */
143
+ function wordStarts(name: string): number[] {
144
+ const isUpper = (c: string): boolean => c >= "A" && c <= "Z";
145
+ const isLower = (c: string): boolean => c >= "a" && c <= "z";
146
+ const isDigit = (c: string): boolean => c >= "0" && c <= "9";
147
+ const isSeparator = (c: string): boolean => !isUpper(c) && !isLower(c) && !isDigit(c);
148
+
149
+ const starts = [0];
150
+ for (let i = 1; i < name.length; i += 1) {
151
+ const previous = name[i - 1] ?? "";
152
+ const current = name[i] ?? "";
153
+ const next = name[i + 1] ?? "";
154
+ if (
155
+ isSeparator(previous) ||
156
+ (isUpper(current) && !isUpper(previous)) ||
157
+ (isUpper(current) && isUpper(previous) && isLower(next)) ||
158
+ (isDigit(current) && !isDigit(previous))
159
+ ) {
160
+ starts.push(i);
161
+ }
162
+ }
163
+ return starts;
164
+ }
165
+
166
+ /**
167
+ * Does `term` (already lowercased) occur in `name` at a word boundary?
168
+ *
169
+ * A prefix of the whole name still counts, so `find "clonePipeline"` keeps
170
+ * returning `clonePipelineDeep`.
171
+ */
172
+ function matchesAtWordBoundary(name: string, term: string): boolean {
173
+ const lower = name.toLowerCase();
174
+ if (lower.length !== name.length) {
175
+ // A locale-folding name (non-ASCII) shifts the indices; fall back rather
176
+ // than reporting a boundary that may not be one.
177
+ return lower.includes(term);
178
+ }
179
+ for (const start of wordStarts(name)) {
180
+ if (lower.startsWith(term, start)) {
181
+ return true;
182
+ }
183
+ }
184
+ return false;
185
+ }
186
+
187
+ function documentFrequencies(
188
+ terms: string[],
189
+ documents: string[],
190
+ matches: (document: string, term: string) => boolean = (document, term) =>
191
+ document.includes(term),
192
+ ): Map<string, number> {
193
+ const frequencies = new Map<string, number>();
194
+ for (const term of terms) {
195
+ let count = 0;
196
+ for (const document of documents) {
197
+ if (matches(document, term)) {
198
+ count += 1;
199
+ }
200
+ }
201
+ frequencies.set(term, count);
202
+ }
203
+ return frequencies;
204
+ }
205
+
206
+ function ubiquitousOf(
207
+ terms: string[],
208
+ frequencies: Map<string, number>,
209
+ total: number,
210
+ ): string[] {
211
+ if (total < MIN_WEIGHTED_CORPUS) {
212
+ return [];
213
+ }
214
+ return terms.filter((term) => (frequencies.get(term) ?? 0) / total > UBIQUITY_FRACTION);
215
+ }
216
+
217
+ function reasonFor(
218
+ matched: string[],
219
+ discriminating: string[],
220
+ boost: string | undefined,
221
+ dependents: number | undefined,
222
+ ): string {
223
+ const parts = [`matched ${matched.join(", ")}`];
224
+ parts.push(
225
+ discriminating.length > 0
226
+ ? `narrowing on ${discriminating.join(", ")}`
227
+ : "no narrowing term — every hit is corpus-wide",
228
+ );
229
+ if (boost) {
230
+ parts.push(boost);
231
+ }
232
+ if (dependents !== undefined) {
233
+ // Named as a tie-break in the reason itself, because "this file has a lot
234
+ // of dependents" is exactly the wrong thing for a reader to mistake for
235
+ // evidence that it answers the question.
236
+ parts.push(`fan-in ${dependents} (tie-break only)`);
237
+ }
238
+ return parts.join("; ");
239
+ }
240
+
241
+ /**
242
+ * What is on the PAGE, when that differs from what the scan found.
243
+ *
244
+ * THE THIRD AND FOURTH INSTANCES (flow 235, T15)
245
+ *
246
+ * Separating matching from ranking (see the note below) stopped a display
247
+ * decision from choosing a `code`. It did not stop one from writing the PROSE.
248
+ * Reproduced on a 100-file corpus with 40 genuine matches:
249
+ *
250
+ * code: ok
251
+ * reason: "20 files and 0 symbols matched." // 40 did
252
+ *
253
+ * and, through the MCP boundary, `fileLimit: 0` produced `"0 files and 0
254
+ * symbols matched."` — a page size rendered as a fact about the corpus, at
255
+ * `code: ok`. The same shape sat in `insufficient-evidence`, whose tail chose
256
+ * between "the ranking below is not evidence" and "Every match scored zero" by
257
+ * asking how long the PAGE was: at `fileLimit: 0` it asserted every match
258
+ * scored zero while sixty matches scored ~5.
259
+ *
260
+ * The rule that closes both: a reason may state what MATCHED only from the
261
+ * scan, and what is SHOWN only as a separate, explicitly-labelled clause. A
262
+ * reader must always be able to tell "40 matched, showing 20" from "20
263
+ * matched"; the two can never be collapsed into one number again.
264
+ */
265
+ function displayNote(
266
+ matchedFiles: number,
267
+ matchedSymbols: number,
268
+ shownFiles: number,
269
+ shownSymbols: number,
270
+ ): string {
271
+ const cut: string[] = [];
272
+ if (shownFiles !== matchedFiles) {
273
+ cut.push(`${shownFiles} of ${matchedFiles} files`);
274
+ }
275
+ if (shownSymbols !== matchedSymbols) {
276
+ cut.push(`${shownSymbols} of ${matchedSymbols} symbols`);
277
+ }
278
+ if (cut.length === 0) {
279
+ return "";
280
+ }
281
+ return (
282
+ ` Showing ${cut.join(" and ")} — the rest are below the display limit or scored zero.` +
283
+ " That is a ranking decision about this page, not a claim about the corpus."
284
+ );
285
+ }
286
+
287
+ // MATCHING vs RANKING — WHY THESE ARE TWO STEPS (flow 235, T14)
288
+ //
289
+ // They used to be one loop, and that produced a false statement about the
290
+ // world. Reproduced at the real command line on 2026-09-08, in a temp project
291
+ // of 14 files ALL under `alpha/`, after a real `keryx gdgraph build`:
292
+ //
293
+ // $ keryx gdgraph find "alpha" --json
294
+ // { "code": "no-match",
295
+ // "reason": "the search completed over 14 indexed files; no path or symbol
296
+ // contains any of: alpha.",
297
+ // "ubiquitousTerms": ["alpha"], "files": [] } exit=0
298
+ //
299
+ // Every one of the fourteen paths contains `alpha`. The payload even said so —
300
+ // `ubiquitousTerms: ["alpha"]` sat in the same object as a reason denying it.
301
+ //
302
+ // The mechanism: `alpha` is in ALL 14 documents, so `termWeight` returns
303
+ // `log(15/15) = 0`, every file scored 0, and the ballast filter dropped every
304
+ // one of them. `findCandidates` then read the empty list as "nothing matched"
305
+ // and made a claim about the corpus that the corpus contradicts — at exit 0,
306
+ // sending the caller off to a text search they did not need.
307
+ //
308
+ // Dropping zero-weight ballast is a RANKING decision: it says "this file is not
309
+ // worth showing you", never "this file does not exist". So the scan below
310
+ // collects every file that matched, score included and nothing filtered, and
311
+ // ranking is a separate step applied only to what gets DISPLAYED. The
312
+ // classifier reads the scan, never the ranked page — which is also why slicing
313
+ // to `limit` can no longer decide a code either.
314
+
315
+ interface FileScan {
316
+ /** Every file whose path contains ≥1 query term. Unfiltered, unsorted, unsliced. */
317
+ readonly matches: FindResult[];
318
+ readonly ubiquitous: string[];
319
+ readonly corpusSize: number;
320
+ }
321
+
322
+ function scanFiles(graph: GraphData, query: string): FileScan {
323
+ const terms = tokenize(query);
324
+ const files = graph.nodes.filter((node) => node.kind === "file");
325
+ const paths = files.map((node) => node.path.toLowerCase());
326
+ const frequencies = documentFrequencies(terms, paths);
327
+ const ubiquitous = ubiquitousOf(terms, frequencies, paths.length);
328
+ const ubiquitousSet = new Set(ubiquitous);
329
+ const scan = { ubiquitous, corpusSize: files.length };
330
+
331
+ if (terms.length === 0) {
332
+ return { matches: [], ...scan };
333
+ }
334
+
335
+ // fan-in (dependents) per node id: how many edges point at it.
336
+ const fanIn = new Map<string, number>();
337
+ for (const edge of graph.edges) {
338
+ fanIn.set(edge.to, (fanIn.get(edge.to) ?? 0) + 1);
339
+ }
340
+
341
+ const matches: FindResult[] = [];
342
+ for (const node of files) {
343
+ const lowerPath = node.path.toLowerCase();
344
+ const base = lowerPath.split("/").pop() ?? lowerPath;
345
+ const matched = terms.filter((t) => lowerPath.includes(t));
346
+ if (matched.length === 0) {
347
+ continue;
348
+ }
349
+ const basenameHits = matched.filter((t) => base.includes(t));
350
+ let score = 0;
351
+ for (const term of matched) {
352
+ const weight = termWeight(frequencies.get(term) ?? 0, paths.length);
353
+ // The basename boost is weighted too: a filename hit on a term that
354
+ // narrows nothing is still worth nothing.
355
+ score += weight * 10 + (basenameHits.includes(term) ? weight * 5 : 0);
356
+ }
357
+ const dependents = fanIn.get(node.id) ?? 0;
358
+ const discriminating = matched.filter((term) => !ubiquitousSet.has(term));
359
+ matches.push({
360
+ path: node.path,
361
+ score,
362
+ matched,
363
+ discriminating,
364
+ dependents,
365
+ reason: reasonFor(
366
+ matched,
367
+ discriminating,
368
+ basenameHits.length > 0 ? `in the filename (${basenameHits.join(", ")})` : undefined,
369
+ dependents,
370
+ ),
371
+ });
372
+ }
373
+ return { matches, ...scan };
374
+ }
375
+
376
+ /**
377
+ * The ranking half: drop zero-score ballast, order, and cut to `limit`.
378
+ *
379
+ * Every decision here is about what is worth SHOWING. None of it may reach the
380
+ * classifier — see the note above `scanFiles`.
381
+ */
382
+ function rankFiles(matches: readonly FindResult[], limit = 20): FindResult[] {
383
+ return matches
384
+ .filter((match) => match.score > 0)
385
+ .sort(
386
+ // Score first, fan-in ONLY as a tie-break: global popularity may separate
387
+ // two files the query cannot, and may never outrank the query itself.
388
+ (a, b) => b.score - a.score || b.dependents - a.dependents || a.path.localeCompare(b.path),
389
+ )
390
+ .slice(0, limit);
391
+ }
392
+
393
+ // Match symbol nodes by name — the precise half of `find` when the symbol layer
394
+ // is active. Exact name match is boosted so `find "clonePipeline"` returns the
395
+ // definition, not just path hits. Term weighting is computed over symbol NAMES
396
+ // (their own corpus), not over paths: a term common among file paths can still
397
+ // be a precise symbol name and vice versa.
398
+ function scanSymbols(graph: GraphData, query: string): SymbolFindResult[] {
47
399
  const terms = tokenize(query);
48
400
  const symbols = graph.symbols ?? [];
49
401
  if (terms.length === 0 || symbols.length === 0) {
50
402
  return [];
51
403
  }
52
404
 
53
- const results: SymbolFindResult[] = [];
405
+ // Frequencies are counted with the SAME predicate the match uses, so a term
406
+ // cannot be rare by one rule and common by another.
407
+ const names = symbols.map((symbol) => symbol.name);
408
+ const frequencies = documentFrequencies(terms, names, matchesAtWordBoundary);
409
+ const ubiquitous = new Set(ubiquitousOf(terms, frequencies, names.length));
410
+
411
+ const matches: SymbolFindResult[] = [];
54
412
  for (const symbol of symbols) {
55
413
  const nameLower = symbol.name.toLowerCase();
56
- const matched = terms.filter((t) => nameLower.includes(t));
414
+ const matched = terms.filter((t) => matchesAtWordBoundary(symbol.name, t));
57
415
  if (matched.length === 0) {
58
416
  continue;
59
417
  }
60
418
  const exactBonus = terms.some((t) => nameLower === t) ? 20 : 0;
61
- results.push({
419
+ let score = exactBonus;
420
+ for (const term of matched) {
421
+ score += termWeight(frequencies.get(term) ?? 0, names.length) * 10;
422
+ }
423
+ const discriminating = matched.filter((term) => !ubiquitous.has(term));
424
+ matches.push({
62
425
  id: symbol.id,
63
426
  name: symbol.name,
64
427
  kind: symbol.kind,
65
428
  path: symbol.path,
66
429
  startLine: symbol.startLine,
67
- score: matched.length * 10 + exactBonus,
430
+ score,
68
431
  matched,
432
+ discriminating,
433
+ reason: reasonFor(
434
+ matched,
435
+ discriminating,
436
+ exactBonus > 0 ? "exact symbol name" : undefined,
437
+ undefined,
438
+ ),
69
439
  });
70
440
  }
71
- results.sort((a, b) => b.score - a.score || a.path.localeCompare(b.path) || a.startLine - b.startLine);
72
- return results.slice(0, limit);
441
+ return matches;
442
+ }
443
+
444
+ /** Ranking only — a name whose hits are all corpus-wide, with no exact match, is not worth showing. */
445
+ function rankSymbols(matches: readonly SymbolFindResult[], limit = 15): SymbolFindResult[] {
446
+ return matches
447
+ .filter((match) => match.score > 0)
448
+ .sort(
449
+ (a, b) => b.score - a.score || a.path.localeCompare(b.path) || a.startLine - b.startLine,
450
+ )
451
+ .slice(0, limit);
452
+ }
453
+
454
+ export function findSymbols(graph: GraphData, query: string, limit = 15): SymbolFindResult[] {
455
+ return rankSymbols(scanSymbols(graph, query), limit);
73
456
  }
74
457
 
75
458
  export function findNodes(graph: GraphData, query: string, limit = 20): FindResult[] {
76
- const terms = tokenize(query);
77
- if (terms.length === 0) {
78
- return [];
459
+ return rankFiles(scanFiles(graph, query).matches, limit);
460
+ }
461
+
462
+ /**
463
+ * `find` with an AC5 outcome code attached (AFC-M03).
464
+ *
465
+ * The measured defect: `keryx gdgraph find` printed "No files or symbols
466
+ * matched" for a genuine no-match AND for a directory with no graph at all,
467
+ * and printed a confident ranked list for a query whose only matching term was
468
+ * in most of the corpus. Four different situations, one indistinguishable
469
+ * answer, all at exit 0.
470
+ *
471
+ * The classification order matters and is not arbitrary: an index that cannot
472
+ * answer must be reported before "nothing matched", because "nothing matched"
473
+ * is a claim about the corpus and an empty index has no standing to make it.
474
+ *
475
+ * For the same reason (T14) every test below reads the SCAN — the full set of
476
+ * files and symbols that matched — and never `foundFiles`/`foundSymbols`, which
477
+ * are the ranked, ballast-free, `limit`-sliced page meant for a reader. A
478
+ * display decision must never be able to turn a corpus that contains your terms
479
+ * into a `no-match` that says it does not.
480
+ *
481
+ * T15 extends that from the conditions to the REASONS. `foundFiles`/
482
+ * `foundSymbols` may appear in a reason only through `displayNote`, which
483
+ * labels the number as a page size; every count a reason states as a fact about
484
+ * the corpus comes from the scan. `find-display-truth.test.ts` pins this.
485
+ */
486
+ export interface FindOutcome {
487
+ readonly code: RetrievalCode;
488
+ readonly reason: string;
489
+ readonly nextActions: readonly string[];
490
+ readonly files: FindResult[];
491
+ readonly symbols: SymbolFindResult[];
492
+ readonly queryTerms: string[];
493
+ readonly ubiquitousTerms: string[];
494
+ }
495
+
496
+ export interface FindOptions {
497
+ readonly fileLimit?: number;
498
+ readonly symbolLimit?: number;
499
+ }
500
+
501
+ export function findCandidates(
502
+ graph: GraphData,
503
+ query: string,
504
+ options: FindOptions = {},
505
+ ): FindOutcome {
506
+ const queryTerms = tokenize(query);
507
+ const scan = scanFiles(graph, query);
508
+ const symbolMatches = scanSymbols(graph, query);
509
+ const ubiquitousTerms = scan.ubiquitous;
510
+
511
+ const empty = { files: [], symbols: [], queryTerms, ubiquitousTerms };
512
+
513
+ if (query.trim().length === 0) {
514
+ return outcome("invalid-input", "no query was given — `find` needs at least one term.", empty);
515
+ }
516
+ if (scan.corpusSize === 0) {
517
+ return outcome(
518
+ "index-incomplete",
519
+ "the graph index holds no file nodes — it was never built here, or its storage is unreadable. " +
520
+ "This is not a statement about whether the code exists.",
521
+ empty,
522
+ );
523
+ }
524
+ if (queryTerms.length === 0) {
525
+ // Aligned word-for-word with the wiki lane's classifier
526
+ // (`src/wiki/ask.ts`) so the same situation gets the same code and the
527
+ // same explanation on both retrieval surfaces.
528
+ return outcome(
529
+ "no-match",
530
+ "the query carries no content terms — every token is a stop word or shorter than two characters.",
531
+ empty,
532
+ );
79
533
  }
80
534
 
81
- // fan-in (dependents) per node id: how many edges point at it.
82
- const fanIn = new Map<string, number>();
83
- for (const edge of graph.edges) {
84
- fanIn.set(edge.to, (fanIn.get(edge.to) ?? 0) + 1);
535
+ const foundFiles = rankFiles(scan.matches, options.fileLimit);
536
+ const foundSymbols = rankSymbols(symbolMatches, options.symbolLimit);
537
+ const found = { files: foundFiles, symbols: foundSymbols, queryTerms, ubiquitousTerms };
538
+
539
+ const matchCount = scan.matches.length + symbolMatches.length;
540
+ if (matchCount === 0) {
541
+ return outcome(
542
+ "no-match",
543
+ `the search completed over ${scan.corpusSize} indexed files; no path or symbol contains any of: ${queryTerms.join(", ")}.`,
544
+ found,
545
+ );
85
546
  }
86
547
 
87
- const results: FindResult[] = [];
88
- for (const node of graph.nodes) {
89
- if (node.kind !== "file") {
90
- continue;
91
- }
92
- const lowerPath = node.path.toLowerCase();
93
- const base = lowerPath.split("/").pop() ?? lowerPath;
94
- const matched = terms.filter((t) => lowerPath.includes(t));
95
- if (matched.length === 0) {
96
- continue;
97
- }
98
- const baseHits = matched.filter((t) => base.includes(t)).length;
99
- const score = matched.length * 10 + baseHits * 5;
100
- results.push({ path: node.path, score, matched, dependents: fanIn.get(node.id) ?? 0 });
548
+ const note = displayNote(
549
+ scan.matches.length,
550
+ symbolMatches.length,
551
+ foundFiles.length,
552
+ foundSymbols.length,
553
+ );
554
+
555
+ const anyDiscriminating =
556
+ scan.matches.some((file) => file.discriminating.length > 0) ||
557
+ symbolMatches.some((symbol) => symbol.discriminating.length > 0);
558
+ if (!anyDiscriminating) {
559
+ const noise = ubiquitousTerms.join(", ") || queryTerms.join(", ");
560
+ // "Did anything score?" is a question about the SCAN. Asking the page
561
+ // instead (`foundFiles.length + foundSymbols.length > 0`) let `fileLimit: 0`
562
+ // turn sixty scoring matches into "Every match scored zero" — see
563
+ // `displayNote`.
564
+ const anyScored =
565
+ scan.matches.some((file) => file.score > 0) ||
566
+ symbolMatches.some((symbol) => symbol.score > 0);
567
+ return outcome(
568
+ "insufficient-evidence",
569
+ `${matchCount} candidates matched, every one of them only on ${noise} — ` +
570
+ "a term that appears across this corpus and so narrows nothing" +
571
+ (anyScored
572
+ ? ", which is why the ranking below is not evidence for this question."
573
+ : ". Every match scored zero, so no ranking is shown; this is not a claim " +
574
+ "that the corpus lacks your terms — it is that they cannot separate anything in it.") +
575
+ note,
576
+ found,
577
+ );
101
578
  }
102
579
 
103
- results.sort(
104
- (a, b) => b.score - a.score || b.dependents - a.dependents || a.path.localeCompare(b.path),
580
+ return outcome(
581
+ "ok",
582
+ `${scan.matches.length} files and ${symbolMatches.length} symbols matched.` + note,
583
+ found,
105
584
  );
106
- return results.slice(0, limit);
585
+ }
586
+
587
+ function outcome(
588
+ code: RetrievalCode,
589
+ reason: string,
590
+ rest: {
591
+ files: FindResult[];
592
+ symbols: SymbolFindResult[];
593
+ queryTerms: string[];
594
+ ubiquitousTerms: string[];
595
+ },
596
+ ): FindOutcome {
597
+ const nextActions = RETRIEVAL_NEXT_ACTIONS[code].slice(0, MAX_NEXT_ACTIONS);
598
+ return { code, reason, nextActions, ...rest };
107
599
  }