@selesai/code 0.13.37 → 0.13.39

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -15,7 +15,7 @@
15
15
  */
16
16
  import { execFile } from "node:child_process";
17
17
  import { closeSync, openSync, readSync, realpathSync, statSync } from "node:fs";
18
- import { basename, isAbsolute, normalize, relative, resolve } from "node:path";
18
+ import { basename, isAbsolute, normalize, relative, resolve, sep } from "node:path";
19
19
  import { StringEnum } from "@earendil-works/pi-ai";
20
20
  import {
21
21
  createLocalBashOperations,
@@ -36,7 +36,10 @@ import {
36
36
  UNTRUSTED_MATERIAL_FOCUS,
37
37
  warnJevUnavailableOnce,
38
38
  type JevAbstainReason,
39
+ type JevAnswers,
40
+ type JevFailureDiagnostic,
39
41
  } from "./jev/decisions.ts";
42
+ import { isScriptPath, keywordWindows, loadTypeScript, type SourceUnit, splitSourceUnits } from "./jev-find-source.ts";
40
43
 
41
44
  /** The tool's registered name. */
42
45
  export const ASK_JEV_TOOL = "ask_jev";
@@ -342,29 +345,68 @@ export function renderAskAnswers(input: {
342
345
  }
343
346
 
344
347
  // ---------------------------------------------------------------------------
345
- // jev_find: ripgrep gathers candidates, Jev judges each one
348
+ // jev_find: ripgrep gathers files; Jev narrows directories, then files, then source units
346
349
  // ---------------------------------------------------------------------------
347
350
 
348
351
  /** The file finder's registered name. */
349
352
  export const JEV_FIND_TOOL = "jev_find";
350
353
  /** ponytail: 16 nouls per request (the JevPDF batch size); raise once Jev is measured on larger blocks. */
351
354
  export const FIND_BATCH = 16;
352
- /** Three batches, sent in parallel, so a find costs one Jev round trip. */
353
- export const MAX_FIND_CANDIDATES = 48;
355
+ /** ponytail: a set this small (three parallel batches) is judged file by file, without directory descent. */
356
+ export const FIND_DIRECT_FILES = 48;
357
+ /** ponytail: at most 128 files judged per call (eight batches), best word/path score first. */
358
+ export const FIND_MAX_JUDGED_FILES = 128;
359
+ /** ponytail: hard cap of 12 Jev requests per call, retries included; raise once latency and cost are measured. */
360
+ export const FIND_MAX_REQUESTS = 12;
361
+ /** ponytail: directory descent spends at most 4 of those requests (64 directory questions over all levels). */
362
+ export const FIND_DIR_REQUESTS = 4;
363
+ /** ponytail: 2 requests are held back for the source-unit stage (32 units). */
364
+ export const FIND_UNIT_REQUESTS = 2;
365
+ /** ponytail: about 16 KB of verbatim source per call; the rest stays reachable through the reading leads. */
366
+ export const FIND_MAX_SOURCE_BYTES = 16 * 1024;
367
+ /** ponytail: descend at most 4 directory levels below the search root. */
368
+ export const FIND_MAX_DEPTH = 4;
369
+ /** ponytail: keep a directory at 0.35 or above: pruning drops every file under it, so it errs toward keeping. */
370
+ export const FIND_DIR_KEEP = 0.35;
371
+ /** ponytail: at most 6 candidate units per relevant file, each shown as at most 60 lines. */
372
+ export const FIND_UNITS_PER_FILE = 6;
373
+ export const FIND_UNIT_MAX_LINES = 60;
374
+ const FIND_DIR_FILE_NAMES = 15;
375
+ const FIND_DIR_HIT_LINES = 3;
354
376
  const FIND_MATCH_LINES = 6;
355
377
  const FIND_DEFAULT_LIMIT = 8;
356
378
  const FIND_MIN_RELEVANCE = 0.5;
357
- /** Room each batch keeps for the questions and JSON scaffolding. */
379
+ /** Files that still get keyword-window source when Jev could not rank them. */
380
+ const FIND_FALLBACK_SOURCE_FILES = 3;
381
+ /** The asker's question is clipped to this before it is repeated in every request. */
382
+ const FIND_REQUEST_BYTES = 2 * 1024;
383
+ /** Room each file batch keeps for the questions and JSON scaffolding when sizing snippets. */
358
384
  const FIND_QUESTION_RESERVE = 6 * 1024;
385
+ /** Lines of a long unit shown above its best keyword line. */
386
+ const FIND_FOCUS_LEAD = 10;
387
+
388
+ /** One file ripgrep offered. */
389
+ export interface FindFile {
390
+ path: string;
391
+ /** Matched line numbers (pattern mode). */
392
+ lines: number[];
393
+ /** Matched lines as `L12: text` (pattern mode). */
394
+ hits: string[];
395
+ /** Question words in the path, plus matched lines (pattern mode) or question words in the content. */
396
+ score: number;
397
+ }
359
398
 
360
399
  export interface FindCandidate {
361
400
  path: string;
362
- /** Matched line numbers (pattern mode only). */
401
+ /** Matched line numbers, or the best keyword lines, best first. */
363
402
  lines: number[];
364
- /** What Jev reads: the matched lines, or the head of the file. */
403
+ /** What Jev reads: the matched lines, or the file's best keyword lines, or its head. */
365
404
  snippet: string;
366
405
  }
367
406
 
407
+ /** Sends one decisions request. The tool binds it to `askJevAnswers`; tests pass a fake. */
408
+ export type JevFindAsk = (payload: Record<string, unknown>) => Promise<JevAnswers>;
409
+
368
410
  /** Run ripgrep with an argument list (no shell). Exit 1 is "no matches"; an overfull buffer keeps what arrived. */
369
411
  function runRg(bin: string, args: string[], cwd: string, signal: AbortSignal | undefined): Promise<string> {
370
412
  return new Promise((done, fail) => {
@@ -414,7 +456,7 @@ export function excerpt(text: string, words: readonly string[], maxBytes: number
414
456
  if (best.length === 0) return { lines: [], snippet: clip(text, maxBytes).text };
415
457
  const inOrder = [...best].sort((a, b) => a.index - b.index);
416
458
  const snippet = inOrder.map((hit) => `L${hit.index + 1}: ${rows[hit.index].trim().slice(0, 240)}`).join("\n");
417
- // Pointers lead with the best-matching lines.
459
+ // Leads start at the best-matching lines.
418
460
  return { lines: best.map((hit) => hit.index + 1), snippet: clip(snippet, maxBytes).text };
419
461
  }
420
462
 
@@ -424,22 +466,32 @@ function pathOverlap(path: string, words: readonly string[]): number {
424
466
  return parts.filter((part) => words.some((word) => word.includes(part) || part.includes(word))).length;
425
467
  }
426
468
 
469
+ const byScore = (a: FindFile, b: FindFile) => b.score - a.score || a.path.localeCompare(b.path);
470
+
471
+ export interface FindGatherOptions {
472
+ rg: string;
473
+ cwd: string;
474
+ /** Search root relative to `cwd`; "." for the working directory. */
475
+ root: string;
476
+ question: string;
477
+ pattern?: string;
478
+ glob?: string;
479
+ ignoreCase?: boolean;
480
+ }
481
+
427
482
  /**
428
- * Candidate files under `root`, most promising first. With `pattern`, the files ripgrep matches
429
- * (most matches first) with their matched lines as the snippet; without it, every file ripgrep
430
- * lists (question words in the path first) with the head of the file as the snippet. Respects
431
- * .gitignore, and never offers a file `askPath` would refuse.
483
+ * Every file under `root` worth judging, most promising first. With `pattern`, the files ripgrep
484
+ * matches with their matched lines; without it, every file ripgrep lists, scored by question words
485
+ * in the path and (for a set too large to judge file by file) in the content, one fixed-string
486
+ * ripgrep per word, in parallel. Respects .gitignore, and never offers a file `askPath` would refuse.
432
487
  */
433
- export async function findCandidates(
434
- options: { rg: string; cwd: string; root: string; question: string; pattern?: string; glob?: string; ignoreCase?: boolean },
435
- snippetBytes: number,
436
- signal: AbortSignal | undefined,
437
- ): Promise<{ candidates: FindCandidate[]; total: number }> {
488
+ export async function gatherFindFiles(options: FindGatherOptions, signal: AbortSignal | undefined): Promise<FindFile[]> {
438
489
  // Always name the root: with no path and a piped stdin, rg searches stdin and hangs.
439
490
  // Paths come back as `./src/x.ts`, so they are normalized below.
440
491
  const root = ["--", options.root];
441
492
  const filters = [...(options.glob ? ["--glob", options.glob] : []), ...(options.ignoreCase ? ["-i"] : [])];
442
493
  const allowed = (path: string) => "full" in askPath(path, options.cwd);
494
+ const words = questionWords(options.question);
443
495
  if (options.pattern) {
444
496
  const out = await runRg(
445
497
  options.rg,
@@ -459,17 +511,11 @@ export async function findCandidates(
459
511
  entry.text.push(`L${match[1]}: ${match[2].trim()}`);
460
512
  byFile.set(path, entry);
461
513
  }
462
- const files = [...byFile].filter(([path]) => allowed(path)).sort((a, b) => b[1].lines.length - a[1].lines.length);
463
- return {
464
- total: files.length,
465
- candidates: files.slice(0, MAX_FIND_CANDIDATES).map(([path, entry]) => ({
466
- path,
467
- lines: entry.lines,
468
- snippet: clip(entry.text.join("\n"), snippetBytes).text,
469
- })),
470
- };
514
+ return [...byFile]
515
+ .filter(([path]) => allowed(path))
516
+ .map(([path, entry]) => ({ path, lines: entry.lines, hits: entry.text, score: entry.lines.length + pathOverlap(path, words) }))
517
+ .sort(byScore);
471
518
  }
472
- const words = questionWords(options.question);
473
519
  const out = await runRg(options.rg, ["--files", "--color=never", ...filters, ...root], options.cwd, signal);
474
520
  const listed = out
475
521
  .split("\n")
@@ -477,9 +523,7 @@ export async function findCandidates(
477
523
  .map((path) => normalize(path))
478
524
  .filter(allowed);
479
525
  const score = new Map(listed.map((path) => [path, pathOverlap(path, words)]));
480
- if (listed.length > MAX_FIND_CANDIDATES && words.length > 0) {
481
- // More files than Jev judges: a file scores once per distinct question word its content holds,
482
- // one fixed-string ripgrep per word, in parallel.
526
+ if (listed.length > FIND_DIRECT_FILES && words.length > 0) {
483
527
  const hits = await Promise.all(
484
528
  words.map((word) =>
485
529
  runRg(options.rg, ["-l", "-i", "-F", "--color=never", ...filters, "-e", word, ...root], options.cwd, signal),
@@ -493,93 +537,742 @@ export async function findCandidates(
493
537
  }
494
538
  }
495
539
  }
496
- const files = listed.map((path) => ({ path })).sort((a, b) => (score.get(b.path) ?? 0) - (score.get(a.path) ?? 0));
497
- const candidates: FindCandidate[] = [];
498
- for (const { path } of files.slice(0, MAX_FIND_CANDIDATES)) {
499
- const file = readBounded(resolve(options.cwd, path), FIND_SCAN_BYTES);
500
- if ("error" in file) continue;
501
- candidates.push({ path, ...excerpt(file.text, words, snippetBytes) });
502
- }
503
- return { candidates, total: files.length };
540
+ return listed.map((path) => ({ path, lines: [], hits: [], score: score.get(path) ?? 0 })).sort(byScore);
541
+ }
542
+
543
+ // Judging ------------------------------------------------------------------
544
+
545
+ /** One noul: a keyed piece of state and the question asked about it. */
546
+ export interface JudgeItem {
547
+ key: string;
548
+ text: string;
549
+ question: string;
550
+ }
551
+
552
+ export interface JudgeStage {
553
+ request: string;
554
+ /** Where the items sit in `state` (`directories`, `files`, `units`). */
555
+ stateKey: string;
556
+ criteria: { true: string; false: string };
504
557
  }
505
558
 
506
- /** One batch as a decisions request: the shared snippets as state, one noul per file. */
507
- export function buildFindPayload(question: string, batch: readonly FindCandidate[]): Record<string, unknown> {
559
+ /** One request: the items at `items` (indexes into the stage's list), one noul `q<i>` each. */
560
+ export function buildJudgePayload(stage: JudgeStage, items: readonly JudgeItem[]): Record<string, unknown> {
508
561
  return buildAskPayload(
509
- batch.map((candidate, index) => ({
510
- name: `f${index}`,
511
- type: "noul" as const,
512
- instructions: `Is the file "${candidate.path}" relevant to the request, judged by its excerpt in state.files?`,
513
- criteria: {
514
- true: "The excerpt shows this file implements, defines, configures, or directly answers the request.",
515
- false: "The file only mentions related words, or is about something else.",
516
- },
517
- })),
518
- { request: question, files: Object.fromEntries(batch.map((candidate) => [candidate.path, candidate.snippet])) },
562
+ items.map((item, index) => ({ name: `q${index}`, type: "noul" as const, instructions: item.question, criteria: stage.criteria })),
563
+ { request: stage.request, [stage.stateKey]: Object.fromEntries(items.map((item) => [item.key, item.text])) },
519
564
  );
520
565
  }
521
566
 
522
567
  /**
523
- * The batch's request, its snippets shrunk until the serialized request fits `maxBytes`: code
524
- * grows under JSON escaping, so a byte budget per snippet alone does not guarantee a fit.
568
+ * The stage's items as requests that each serialize within `maxBytes`: FIND_BATCH items per
569
+ * request, their texts shrunk together until the request fits (code grows under JSON escaping, so a
570
+ * byte budget per text alone does not guarantee a fit). When even empty texts do not fit — the
571
+ * questions, paths, and scaffolding alone are too large — the batch is halved, and an item that
572
+ * cannot fit alone is dropped. No request this returns can come back as `overflow`.
573
+ */
574
+ export function fitJudgePayloads(
575
+ stage: JudgeStage,
576
+ items: readonly JudgeItem[],
577
+ maxBytes: number,
578
+ ): { payloads: Array<{ payload: Record<string, unknown>; items: number[] }>; dropped: number[] } {
579
+ const payloads: Array<{ payload: Record<string, unknown>; items: number[] }> = [];
580
+ const dropped: number[] = [];
581
+ const place = (indexes: number[]): void => {
582
+ let limit = Math.max(0, ...indexes.map((index) => Buffer.byteLength(items[index].text, "utf-8")));
583
+ for (;;) {
584
+ const payload = buildJudgePayload(
585
+ stage,
586
+ indexes.map((index) => {
587
+ const text = clip(items[index].text, limit);
588
+ return { ...items[index], text: text.clipped && limit > 0 ? `${text.text}${TRUNCATION_MARKER}` : text.text };
589
+ }),
590
+ );
591
+ if (serializeJevRequest(payload, maxBytes) !== undefined) {
592
+ payloads.push({ payload, items: indexes });
593
+ return;
594
+ }
595
+ if (limit > 0) {
596
+ limit = limit < 64 ? 0 : Math.floor(limit * 0.75);
597
+ continue;
598
+ }
599
+ if (indexes.length === 1) {
600
+ dropped.push(indexes[0]);
601
+ return;
602
+ }
603
+ const half = Math.ceil(indexes.length / 2);
604
+ place(indexes.slice(0, half));
605
+ place(indexes.slice(half));
606
+ return;
607
+ }
608
+ };
609
+ for (let start = 0; start < items.length; start += FIND_BATCH) {
610
+ place(Array.from({ length: Math.min(FIND_BATCH, items.length - start) }, (_, offset) => start + offset));
611
+ }
612
+ return { payloads, dropped };
613
+ }
614
+
615
+ /**
616
+ * The call's Jev requests: every stage sends its batches in parallel (one round trip), retries a
617
+ * `transport` failure once, and never exceeds FIND_MAX_REQUESTS in total, retries included.
525
618
  */
526
- export function fitFindPayload(question: string, batch: readonly FindCandidate[], maxBytes: number): Record<string, unknown> {
527
- let limit = Math.max(0, ...batch.map((candidate) => Buffer.byteLength(candidate.snippet, "utf-8")));
528
- for (;;) {
529
- const payload = buildFindPayload(
530
- question,
531
- batch.map((candidate) => ({ ...candidate, snippet: clip(candidate.snippet, limit).text })),
619
+ export class FindRequests {
620
+ sent = 0;
621
+ retried = 0;
622
+ readonly diagnostics: JevFailureDiagnostic[] = [];
623
+ private readonly ask: JevFindAsk;
624
+ readonly max: number;
625
+ constructor(ask: JevFindAsk, max = FIND_MAX_REQUESTS) {
626
+ this.ask = ask;
627
+ this.max = max;
628
+ }
629
+
630
+ get left(): number {
631
+ return this.max - this.sent;
632
+ }
633
+
634
+ /**
635
+ * Send up to `sendCap` payloads in parallel, then retry the transport failures within `retryCap`
636
+ * total requests for this stage. A payload past the caps comes back undefined (not sent).
637
+ */
638
+ async sendAll(
639
+ payloads: readonly Record<string, unknown>[],
640
+ sendCap: number,
641
+ retryCap = sendCap,
642
+ ): Promise<Array<JevAnswers | undefined>> {
643
+ const count = Math.max(0, Math.min(payloads.length, sendCap, this.left));
644
+ this.sent += count;
645
+ const results: Array<JevAnswers | undefined> = await Promise.all(
646
+ payloads.slice(0, count).map((payload) => this.ask(payload)),
532
647
  );
533
- if (limit < 64 || serializeJevRequest(payload, maxBytes) !== undefined) return payload;
534
- limit = Math.floor(limit * 0.75);
648
+ const retry = results
649
+ .map((result, index) => ({ result, index }))
650
+ .filter(({ result }) => result?.failure === "transport")
651
+ .slice(0, Math.max(0, Math.min(retryCap - count, this.left)));
652
+ this.sent += retry.length;
653
+ this.retried += retry.length;
654
+ const again = await Promise.all(retry.map(({ index }) => this.ask(payloads[index])));
655
+ retry.forEach(({ index }, k) => {
656
+ results[index] = again[k];
657
+ });
658
+ for (const result of results) {
659
+ const diagnostic = result?.diagnostic;
660
+ if (diagnostic && !this.diagnostics.some((item) => item.kind === diagnostic.kind && item.httpStatus === diagnostic.httpStatus)) {
661
+ this.diagnostics.push(diagnostic);
662
+ }
663
+ }
664
+ return [...results, ...Array.from({ length: payloads.length - count }, () => undefined)];
535
665
  }
536
666
  }
537
667
 
538
- /** Pointers only: path, matched lines, relevance. File contents never reach the agent. */
539
- export function renderFindResults(input: {
540
- ranked: ReadonlyArray<FindCandidate & { relevance?: number }>;
541
- total: number;
542
- judged: number;
543
- limit: number;
668
+ interface JudgeOutcome {
669
+ /** Jev's noul per item; undefined when it was not answered. */
670
+ scores: Array<number | undefined>;
671
+ /** Why unanswered items went unanswered: a request failure, `request cap`, or `too large`. */
672
+ reasons: string[];
673
+ }
674
+
675
+ /** Ask one noul per item, batched and fitted, within the stage's request caps. */
676
+ async function judgeNouls(
677
+ requests: FindRequests,
678
+ stage: JudgeStage,
679
+ items: readonly JudgeItem[],
680
+ maxBytes: number,
681
+ caps: { send: number; retry: number },
682
+ ): Promise<JudgeOutcome> {
683
+ const scores: Array<number | undefined> = items.map(() => undefined);
684
+ const reasons: string[] = [];
685
+ if (items.length === 0) return { scores, reasons };
686
+ const fitted = fitJudgePayloads(stage, items, maxBytes);
687
+ if (fitted.dropped.length > 0) reasons.push("too large");
688
+ const results = await requests.sendAll(
689
+ fitted.payloads.map((entry) => entry.payload),
690
+ caps.send,
691
+ caps.retry,
692
+ );
693
+ fitted.payloads.forEach((entry, p) => {
694
+ const result = results[p];
695
+ if (!result) {
696
+ reasons.push("request cap");
697
+ return;
698
+ }
699
+ if (result.failure) reasons.push(result.failure);
700
+ const answers = result.answers ?? {};
701
+ entry.items.forEach((index, k) => {
702
+ const answer = answers[`q${k}`];
703
+ if (isRecord(answer) && typeof answer.noul === "number") scores[index] = answer.noul;
704
+ });
705
+ });
706
+ return { scores, reasons: [...new Set(reasons)] };
707
+ }
708
+
709
+ // Directory descent --------------------------------------------------------
710
+
711
+ const DIRECTORY_CRITERIA = {
712
+ true: "Its name, file names, or keyword hits suggest it holds code that implements, defines, or configures what the request asks about.",
713
+ false: "It clearly holds something unrelated to the request.",
714
+ };
715
+
716
+ /** The directory one level below `dir` that holds `path`, or undefined when `path` sits directly in `dir`. */
717
+ export function childDirectory(path: string, dir: string): string | undefined {
718
+ const rest = dir === "." ? path : path.slice(dir.length + 1);
719
+ const cut = rest.indexOf(sep);
720
+ if (cut < 0) return undefined;
721
+ return dir === "." ? rest.slice(0, cut) : `${dir}${sep}${rest.slice(0, cut)}`;
722
+ }
723
+
724
+ /** What Jev reads of a directory: its size, its most promising file names, and a few keyword hits. */
725
+ function describeDirectory(dir: string, files: readonly FindFile[], hitLines: (file: FindFile) => string[]): string {
726
+ const ordered = [...files].sort(byScore);
727
+ const local = (path: string) => path.slice(dir.length + 1);
728
+ const names = ordered.slice(0, FIND_DIR_FILE_NAMES).map((file) => local(file.path));
729
+ const more = files.length - names.length;
730
+ const hits: string[] = [];
731
+ for (const file of ordered.filter((candidate) => candidate.score > 0).slice(0, FIND_DIR_HIT_LINES)) {
732
+ for (const line of hitLines(file).slice(0, 2)) {
733
+ if (hits.length < FIND_DIR_HIT_LINES) hits.push(`${local(file.path)} ${line.slice(0, 160)}`);
734
+ }
735
+ }
736
+ return [
737
+ `${files.length} file(s): ${names.join(", ")}${more > 0 ? `, … (+${more} more)` : ""}`,
738
+ ...(hits.length > 0 ? ["Keyword hits:", ...hits] : []),
739
+ ].join("\n");
740
+ }
741
+
742
+ export interface DescentResult {
743
+ /** Files still worth judging: those in kept directories, and those sitting directly in a descended one. */
744
+ kept: FindFile[];
745
+ prunedDirs: number;
746
+ prunedFiles: number;
747
+ /** Directories kept without an answer, and why. */
748
+ unjudgedDirs: number;
749
+ reasons: string[];
750
+ }
751
+
752
+ /**
753
+ * Narrow a large file set directory by directory: one noul per next-level directory ("could this
754
+ * hold the answer?"), all of a level in one parallel round trip. A directory below FIND_DIR_KEEP is
755
+ * dropped with every file under it; an unanswered one is kept (unknown is not rejected). Kept
756
+ * directories still larger than FIND_DIRECT_FILES are descended again, up to FIND_MAX_DEPTH
757
+ * levels; a lone subdirectory is entered without a question.
758
+ */
759
+ export async function narrowByDirectory(
760
+ files: readonly FindFile[],
761
+ root: string,
762
+ options: { request: string; maxBytes: number; requests: FindRequests; hitLines: (file: FindFile) => string[] },
763
+ ): Promise<DescentResult> {
764
+ const result: DescentResult = { kept: [], prunedDirs: 0, prunedFiles: 0, unjudgedDirs: 0, reasons: [] };
765
+ if (files.length <= FIND_DIRECT_FILES) return { ...result, kept: [...files] };
766
+ let pending: Array<{ dir: string; files: FindFile[]; depth: number }> = [{ dir: root, files: [...files], depth: 0 }];
767
+ let dirRequests = 0;
768
+ while (pending.length > 0) {
769
+ const children: Array<{ dir: string; files: FindFile[]; depth: number; free: boolean }> = [];
770
+ for (const group of pending) {
771
+ const byChild = new Map<string, FindFile[]>();
772
+ let direct = 0;
773
+ for (const file of group.files) {
774
+ const child = childDirectory(file.path, group.dir);
775
+ if (child === undefined) {
776
+ result.kept.push(file);
777
+ direct += 1;
778
+ } else {
779
+ byChild.set(child, [...(byChild.get(child) ?? []), file]);
780
+ }
781
+ }
782
+ const free = byChild.size === 1 && direct === 0;
783
+ for (const [dir, under] of byChild) children.push({ dir, files: under, depth: group.depth + 1, free });
784
+ }
785
+ const sendCap = Math.min(FIND_DIR_REQUESTS - dirRequests, options.requests.left - FIND_UNIT_REQUESTS - 1);
786
+ const questionable = children.filter((child) => !child.free && child.depth <= FIND_MAX_DEPTH);
787
+ const asked =
788
+ sendCap > 0
789
+ ? questionable.sort(
790
+ (a, b) => b.files.reduce((sum, f) => sum + f.score, 0) - a.files.reduce((sum, f) => sum + f.score, 0),
791
+ )
792
+ : [];
793
+ if (children.some((child) => !child.free && child.depth > FIND_MAX_DEPTH)) result.reasons.push("depth limit");
794
+ if (questionable.length > 0 && asked.length === 0) result.reasons.push("request cap");
795
+ const verdicts = new Map<string, number | undefined>();
796
+ if (asked.length > 0) {
797
+ const before = options.requests.sent;
798
+ const outcome = await judgeNouls(
799
+ options.requests,
800
+ { request: options.request, stateKey: "directories", criteria: DIRECTORY_CRITERIA },
801
+ asked.map((child) => ({
802
+ key: `${child.dir}${sep}`,
803
+ text: describeDirectory(child.dir, child.files, options.hitLines),
804
+ question: `Could the directory "${child.dir}${sep}" contain code that answers the request, judged by its description in state.directories?`,
805
+ })),
806
+ options.maxBytes,
807
+ { send: sendCap, retry: sendCap },
808
+ );
809
+ dirRequests += options.requests.sent - before;
810
+ asked.forEach((child, index) => verdicts.set(child.dir, outcome.scores[index]));
811
+ result.reasons.push(...outcome.reasons);
812
+ }
813
+ const next: typeof pending = [];
814
+ for (const child of children) {
815
+ const wasAsked = verdicts.has(child.dir);
816
+ const verdict = verdicts.get(child.dir);
817
+ if (verdict !== undefined && verdict < FIND_DIR_KEEP) {
818
+ result.prunedDirs += 1;
819
+ result.prunedFiles += child.files.length;
820
+ continue;
821
+ }
822
+ if (verdict === undefined && !child.free) result.unjudgedDirs += 1;
823
+ const canDescend = child.depth < FIND_MAX_DEPTH && (child.free || wasAsked);
824
+ if (child.files.length > FIND_DIRECT_FILES && canDescend) {
825
+ next.push({ dir: child.dir, files: child.files, depth: child.depth });
826
+ } else result.kept.push(...child.files);
827
+ }
828
+ pending = next;
829
+ }
830
+ result.reasons = [...new Set(result.reasons)];
831
+ return result;
832
+ }
833
+
834
+ // Source units -------------------------------------------------------------
835
+
836
+ const FILE_CRITERIA = {
837
+ true: "The excerpt shows this file implements, defines, configures, or directly answers the request.",
838
+ false: "The file only mentions related words, or is about something else.",
839
+ };
840
+
841
+ const UNIT_CRITERIA = {
842
+ true: "This unit implements, defines, configures, or directly answers what the request asks.",
843
+ false: "The unit only mentions related words, or does something else.",
844
+ };
845
+
846
+ /** A unit's candidate score: matched/keyword lines inside it weigh most, then distinct question words. */
847
+ export function pickUnits(
848
+ units: readonly SourceUnit[],
849
+ rows: readonly string[],
850
+ hits: readonly number[],
851
+ words: readonly string[],
852
+ ): SourceUnit[] {
853
+ const bodies = units.filter((unit) => unit.name !== "imports");
854
+ const scored = bodies.map((unit) => {
855
+ const body = rows.slice(unit.start - 1, unit.end).join("\n").toLowerCase();
856
+ const hitCount = hits.filter((line) => line >= unit.start && line <= unit.end).length;
857
+ return { unit, score: hitCount * 10 + words.filter((word) => body.includes(word)).length };
858
+ });
859
+ const picks = scored
860
+ .filter((entry) => entry.score > 0)
861
+ .sort((a, b) => b.score - a.score || a.unit.start - b.unit.start)
862
+ .slice(0, FIND_UNITS_PER_FILE)
863
+ .map((entry) => entry.unit);
864
+ // Nothing matched a word: let Jev judge the file's first declarations.
865
+ return (picks.length > 0 ? picks : bodies.slice(0, 3)).sort((a, b) => a.start - b.start);
866
+ }
867
+
868
+ /**
869
+ * The line numbers a unit is shown as (0 marks elided lines): the header line of a member, then the
870
+ * unit whole, or — when it is longer than `maxLines` — its first line and a window from just above
871
+ * its best keyword line (`focus`), so a long function shows where the asked-about behavior sits.
872
+ */
873
+ export function unitView(unit: SourceUnit, focus: number | undefined, maxLines = FIND_UNIT_MAX_LINES): number[] {
874
+ const view: number[] = unit.header !== undefined && unit.header < unit.start ? [unit.header, 0] : [];
875
+ const range = (from: number, to: number) => Array.from({ length: Math.max(0, to - from + 1) }, (_, i) => from + i);
876
+ if (unit.end - unit.start + 1 <= maxLines) return [...view, ...range(unit.start, unit.end)];
877
+ if (focus === undefined || focus < unit.start + maxLines - FIND_FOCUS_LEAD) {
878
+ return [...view, ...range(unit.start, unit.start + maxLines - 1), 0];
879
+ }
880
+ const from = Math.max(unit.start + 1, focus - FIND_FOCUS_LEAD);
881
+ const to = Math.min(unit.end, from + maxLines - 2);
882
+ return [...view, unit.start, ...(from > unit.start + 1 ? [0] : []), ...range(from, to), ...(to < unit.end ? [0] : [])];
883
+ }
884
+
885
+ /** The first `hits` line inside the unit, in `hits` order (best first). */
886
+ function focusOf(unit: SourceUnit, hits: readonly number[]): number | undefined {
887
+ return hits.find((line) => line >= unit.start && line <= unit.end);
888
+ }
889
+
890
+ /** A file's rows without carriage returns (and without the empty row after a final newline), so numbered lines print as they read. */
891
+ function fileRows(text: string): string[] {
892
+ const rows = text.split("\n").map((row) => row.replace(/\r$/, ""));
893
+ if (rows.length > 1 && rows.at(-1) === "") rows.pop();
894
+ return rows;
895
+ }
896
+
897
+ interface SourceBlock {
898
+ path: string;
899
+ unit: SourceUnit;
900
+ view: number[];
901
+ rows: readonly string[];
902
+ }
903
+
904
+ /**
905
+ * `Source block "path" lines a-b:` and the numbered verbatim lines of each block, in order, within
906
+ * `budgetBytes` of source; a block cut by the budget ends in `…`, and one that cannot show three
907
+ * lines is left out.
908
+ */
909
+ export function renderSourceBlocks(
910
+ blocks: readonly SourceBlock[],
911
+ budgetBytes = FIND_MAX_SOURCE_BYTES,
912
+ ): { lines: string[]; bytes: number; omitted: number } {
913
+ const lines: string[] = [];
914
+ let bytes = 0;
915
+ let omitted = 0;
916
+ for (const block of blocks) {
917
+ const numbered: string[] = [];
918
+ let used = 0;
919
+ let cut = false;
920
+ for (const line of block.view) {
921
+ const text = line === 0 ? "…" : `${line}: ${block.rows[line - 1] ?? ""}`;
922
+ const size = Buffer.byteLength(text, "utf-8") + 1;
923
+ if (bytes + used + size > budgetBytes) {
924
+ cut = true;
925
+ break;
926
+ }
927
+ numbered.push(text);
928
+ used += size;
929
+ }
930
+ if (numbered.filter((text) => text !== "…").length < Math.min(3, block.view.filter((line) => line !== 0).length)) {
931
+ omitted += 1;
932
+ continue;
933
+ }
934
+ if (cut && numbered.at(-1) !== "…") numbered.push("…");
935
+ lines.push("", `Source block "${block.path}" lines ${block.unit.start}-${block.unit.end}:`, ...numbered);
936
+ bytes += used;
937
+ }
938
+ return { lines, bytes, omitted };
939
+ }
940
+
941
+ /** Keyword lines for a directory description or a fallback window: rg's matches, or the file's best question-word lines. */
942
+ function keywordLines(text: string | undefined, words: readonly string[]): { lines: number[]; snippet: string[] } {
943
+ if (text === undefined || words.length === 0) return { lines: [], snippet: [] };
944
+ const found = excerpt(text, words, FIND_SCAN_BYTES);
945
+ return found.lines.length === 0 ? { lines: [], snippet: [] } : { lines: found.lines, snippet: found.snippet.split("\n") };
946
+ }
947
+
948
+ function fallbackSourceUnits(
949
+ path: string,
950
+ text: string,
951
+ ts: Awaited<ReturnType<typeof loadTypeScript>>,
952
+ rows: readonly string[],
953
+ hits: readonly number[],
954
+ words: readonly string[],
955
+ ): SourceUnit[] {
956
+ if (hits.length === 0) return [];
957
+ return pickUnits(splitSourceUnits(path, text, ts), rows, hits, words).filter((unit) =>
958
+ hits.some((line) => (line >= unit.start && line <= unit.end) || line === unit.header),
959
+ );
960
+ }
961
+
962
+ function formatJevDiagnostic(diagnostic: JevFailureDiagnostic): string {
963
+ if (diagnostic.kind === "timeout") return "Jev request timed out.";
964
+ if (diagnostic.kind === "malformed-response") return "Jev returned a malformed response.";
965
+ return `Jev provider request failed${diagnostic.httpStatus ? ` (HTTP ${diagnostic.httpStatus})` : ""}.`;
966
+ }
967
+
968
+ // The whole call -----------------------------------------------------------
969
+
970
+ export interface JevFindOptions extends FindGatherOptions {
971
+ limit?: number;
972
+ /** The route's request budget; no request is sent larger than this. */
973
+ maxBytes: number;
544
974
  model: string;
545
- elapsedMs: number;
975
+ /** Sends one request; absent when Jev is out of reach (`unreachable` says why). */
976
+ ask?: JevFindAsk;
977
+ unreachable?: JevAbstainReason;
978
+ /** The TypeScript loader; tests pass one resolving undefined to force chunking. */
979
+ loadTypeScript?: typeof loadTypeScript;
980
+ signal?: AbortSignal;
981
+ }
982
+
983
+ export interface JevFindResult {
984
+ text: string;
985
+ /** Shape only, like ask_jev: details persist in the session. */
986
+ details: Record<string, unknown>;
987
+ judged: number;
988
+ topRelevance?: number;
546
989
  failure?: string;
547
- }): string {
548
- const pointer = (candidate: FindCandidate) =>
549
- candidate.lines.length > 0 ? `${candidate.path}:${candidate.lines.slice(0, 4).join(",")}` : candidate.path;
550
- if (input.total === 0) return "jev_find: no candidate files. Loosen `pattern`, `glob`, or `path`.";
551
- const lines: string[] = [];
552
- if (input.judged === 0) {
553
- lines.push(
554
- `jev_find: Jev did not judge (${input.failure ?? "missing"}); ${input.total} candidate file(s), unranked by relevance:`,
555
- ...input.ranked.slice(0, input.limit).map((candidate) => `- ${pointer(candidate)}`),
990
+ elapsedMs: number;
991
+ }
992
+
993
+ /**
994
+ * The whole jev_find call: gather files with ripgrep, narrow large sets by directory, judge files,
995
+ * then judge the source units of the relevant ones and return them verbatim. Without a reachable
996
+ * Jev (or when no file could be judged) the ripgrep-ranked list still comes back, with keyword
997
+ * windows of the top files, so the call is never wasted. Throws only when ripgrep fails.
998
+ */
999
+ export async function runJevFind(options: JevFindOptions): Promise<JevFindResult> {
1000
+ const files = await gatherFindFiles(options, options.signal);
1001
+ if (files.length === 0) {
1002
+ return {
1003
+ text: "jev_find: no candidate files. Loosen `pattern`, `glob`, or `path`.",
1004
+ details: { total: 0, judged: 0 },
1005
+ judged: 0,
1006
+ elapsedMs: 0,
1007
+ };
1008
+ }
1009
+ const words = questionWords(options.question);
1010
+ const limit = Math.max(1, Math.floor(options.limit ?? FIND_DEFAULT_LIMIT));
1011
+ const texts = new Map<string, string | undefined>();
1012
+ const readText = (path: string): string | undefined => {
1013
+ if (!texts.has(path)) {
1014
+ const file = readBounded(resolve(options.cwd, path), FIND_SCAN_BYTES);
1015
+ texts.set(path, "error" in file ? undefined : file.text);
1016
+ }
1017
+ return texts.get(path);
1018
+ };
1019
+ const hitsOf = (file: FindFile): { lines: number[]; snippet: string[] } =>
1020
+ file.lines.length > 0 ? { lines: file.lines, snippet: file.hits } : keywordLines(readText(file.path), words);
1021
+
1022
+ /** The ripgrep-ranked list and source units of the top files: what comes back when Jev did not rank. */
1023
+ const fallback = async (
1024
+ reason: string,
1025
+ notes: string[],
1026
+ elapsedMs: number,
1027
+ requests?: FindRequests,
1028
+ ): Promise<JevFindResult> => {
1029
+ const shown = files.slice(0, limit);
1030
+ const sourceFiles = shown.slice(0, FIND_FALLBACK_SOURCE_FILES);
1031
+ const ts = sourceFiles.some((file) => isScriptPath(file.path)) ? await (options.loadTypeScript ?? loadTypeScript)() : undefined;
1032
+ const blocks: SourceBlock[] = [];
1033
+ const leads = new Map<string, string>();
1034
+ for (const file of sourceFiles) {
1035
+ const text = readText(file.path);
1036
+ if (text === undefined) continue;
1037
+ const rows = fileRows(text);
1038
+ const hits = hitsOf(file).lines;
1039
+ const units = fallbackSourceUnits(file.path, text, ts, rows, hits, words);
1040
+ const selected = units.length > 0 ? units : keywordWindows(rows.length, hits);
1041
+ leads.set(file.path, selected.map((unit) => `${unit.name} lines ${unit.start}-${unit.end}`).join(", "));
1042
+ for (const unit of selected) blocks.push({ path: file.path, unit, view: unitView(unit, focusOf(unit, hits)), rows });
1043
+ }
1044
+ const source = renderSourceBlocks(blocks);
1045
+ const diagnostic = requests?.diagnostics[0];
1046
+ const lines = [
1047
+ `jev_find: Jev did not judge (${reason}); ${files.length} candidate file(s), ranked by ripgrep:`,
1048
+ ...notes,
1049
+ ...(diagnostic ? [`Diagnostic: ${formatJevDiagnostic(diagnostic)}`] : []),
1050
+ ...shown.map((file) => `- ${file.path}${leads.has(file.path) ? ` — ${leads.get(file.path)}` : ""}`),
1051
+ ...(files.length > shown.length ? [`${files.length - shown.length} more candidate file(s) not listed.`] : []),
1052
+ ...source.lines,
1053
+ "",
1054
+ "End context.",
1055
+ ];
1056
+ return {
1057
+ text: lines.join("\n"),
1058
+ details: {
1059
+ total: files.length,
1060
+ judged: 0,
1061
+ requests: requests?.sent ?? 0,
1062
+ sourceBytes: source.bytes,
1063
+ elapsedMs,
1064
+ failure: reason,
1065
+ ...(diagnostic ? { diagnostic } : {}),
1066
+ },
1067
+ judged: 0,
1068
+ failure: reason,
1069
+ ...(diagnostic ? { diagnostic } : {}),
1070
+ elapsedMs,
1071
+ };
1072
+ };
1073
+
1074
+ if (!options.ask) return fallback(options.unreachable ?? "missing", [], 0);
1075
+
1076
+ const started = Date.now();
1077
+ const requests = new FindRequests(options.ask);
1078
+ const request = clip(options.question, FIND_REQUEST_BYTES).text;
1079
+ const notes: string[] = [];
1080
+
1081
+ // 1. Directories.
1082
+ const descent = await narrowByDirectory(files, options.root, {
1083
+ request,
1084
+ maxBytes: options.maxBytes,
1085
+ requests,
1086
+ hitLines: (file) => hitsOf(file).snippet,
1087
+ });
1088
+ if (descent.prunedDirs > 0) {
1089
+ notes.push(`Jev ruled out ${descent.prunedDirs} director${descent.prunedDirs === 1 ? "y" : "ies"} holding ${descent.prunedFiles} file(s).`);
1090
+ }
1091
+ if (descent.unjudgedDirs > 0) {
1092
+ notes.push(
1093
+ `${descent.unjudgedDirs} director${descent.unjudgedDirs === 1 ? "y was" : "ies were"} kept unjudged (${descent.reasons.join(", ") || "missing"}).`,
556
1094
  );
557
- } else {
558
- const strong = input.ranked.filter((candidate) => (candidate.relevance ?? 0) >= FIND_MIN_RELEVANCE);
559
- lines.push(`jev_find: judged ${input.judged} of ${input.total} candidate file(s) via ${input.model} in ${input.elapsedMs}ms`);
560
- const shown = strong.length > 0 ? strong.slice(0, input.limit) : input.ranked.slice(0, 3);
561
- if (strong.length === 0) lines.push("No file is a confident match; the closest were:");
562
- for (const candidate of shown) {
563
- lines.push(`- ${pointer(candidate)} (${candidate.relevance === undefined ? "not judged" : candidate.relevance.toFixed(2)})`);
1095
+ }
1096
+
1097
+ // 2. Files, best score first, within the judging cap and the requests left after the unit reserve.
1098
+ const fileSendCap = requests.left - FIND_UNIT_REQUESTS;
1099
+ const judgeCap = Math.min(FIND_MAX_JUDGED_FILES, Math.max(0, fileSendCap) * FIND_BATCH);
1100
+ const ordered = [...descent.kept].sort(byScore);
1101
+ const pastCap = Math.max(0, ordered.length - judgeCap);
1102
+ const snippetBytes = Math.max(256, Math.floor((options.maxBytes - FIND_QUESTION_RESERVE) / FIND_BATCH));
1103
+ const candidates: FindCandidate[] = [];
1104
+ for (const file of ordered.slice(0, judgeCap)) {
1105
+ if (file.lines.length > 0) {
1106
+ candidates.push({ path: file.path, lines: file.lines, snippet: clip(file.hits.join("\n"), snippetBytes).text });
1107
+ continue;
564
1108
  }
565
- if (strong.length > shown.length) lines.push(`${strong.length - shown.length} more relevant file(s) past the limit.`);
1109
+ const text = readText(file.path);
1110
+ if (text !== undefined) candidates.push({ path: file.path, ...excerpt(text, words, snippetBytes) });
566
1111
  }
567
- if (input.total > MAX_FIND_CANDIDATES) {
568
- lines.push(`${input.total - MAX_FIND_CANDIDATES} candidate(s) were not judged; narrow with \`pattern\`, \`glob\`, or \`path\`.`);
1112
+ const filesOutcome = await judgeNouls(
1113
+ requests,
1114
+ { request, stateKey: "files", criteria: FILE_CRITERIA },
1115
+ candidates.map((candidate) => ({
1116
+ key: candidate.path,
1117
+ text: candidate.snippet,
1118
+ question: `Is the file "${candidate.path}" relevant to the request, judged by its excerpt in state.files?`,
1119
+ })),
1120
+ options.maxBytes,
1121
+ // A retry may borrow one request from the unit reserve: without file verdicts there is nothing to excerpt.
1122
+ { send: fileSendCap, retry: requests.left - 1 },
1123
+ );
1124
+ const scored = candidates.map((candidate, index) => ({ ...candidate, relevance: filesOutcome.scores[index] }));
1125
+ const judged = scored.filter((candidate) => candidate.relevance !== undefined).length;
1126
+ const unjudgedFiles = candidates.length - judged;
1127
+ if (pastCap > 0) {
1128
+ notes.push(`${pastCap} file(s) past the ${judgeCap}-file judging cap were not judged; narrow with \`pattern\`, \`glob\`, or \`path\`.`);
1129
+ }
1130
+ if (judged === 0) {
1131
+ return fallback(filesOutcome.reasons[0] ?? "missing", notes, Date.now() - started, requests);
1132
+ }
1133
+ if (unjudgedFiles > 0) notes.push(`${unjudgedFiles} file(s) were not judged (${filesOutcome.reasons.join(", ") || "missing"}).`);
1134
+
1135
+ const ranked = scored
1136
+ .filter((candidate): candidate is FindCandidate & { relevance: number } => candidate.relevance !== undefined)
1137
+ .sort((a, b) => b.relevance - a.relevance);
1138
+ const relevant = ranked.filter((candidate) => candidate.relevance >= FIND_MIN_RELEVANCE);
1139
+ const shown = relevant.slice(0, limit);
1140
+
1141
+ // 3. Source units of the relevant files, in relevance order, within the requests left.
1142
+ const ts = shown.some((candidate) => isScriptPath(candidate.path))
1143
+ ? await (options.loadTypeScript ?? loadTypeScript)()
1144
+ : undefined;
1145
+ const perFile = shown.map((candidate) => {
1146
+ const text = readText(candidate.path) ?? "";
1147
+ const rows = fileRows(text);
1148
+ const hits = [...new Set([...candidate.lines, ...keywordLines(text, words).lines])];
1149
+ const picks = text === "" ? [] : pickUnits(splitSourceUnits(candidate.path, text, ts), rows, hits, words);
1150
+ return { candidate, rows, hits, picks };
1151
+ });
1152
+ const unitItems: JudgeItem[] = [];
1153
+ const owners: Array<{ file: number; unit: SourceUnit }> = [];
1154
+ const unitCapacity = requests.left * FIND_BATCH;
1155
+ perFile.forEach((entry, file) => {
1156
+ for (const unit of entry.picks) {
1157
+ if (unitItems.length >= unitCapacity) return;
1158
+ const key = `${entry.candidate.path} lines ${unit.start}-${unit.end} (${unit.name})`;
1159
+ unitItems.push({
1160
+ key,
1161
+ text: unitView(unit, focusOf(unit, entry.hits))
1162
+ .map((line) => (line === 0 ? "…" : (entry.rows[line - 1] ?? "")))
1163
+ .join("\n"),
1164
+ question: `Does the source unit "${key}" in state.units implement, define, configure, or directly answer the request?`,
1165
+ });
1166
+ owners.push({ file, unit });
1167
+ }
1168
+ });
1169
+ const unitsOutcome = await judgeNouls(
1170
+ requests,
1171
+ { request, stateKey: "units", criteria: UNIT_CRITERIA },
1172
+ unitItems,
1173
+ options.maxBytes,
1174
+ { send: requests.left, retry: requests.left },
1175
+ );
1176
+ const elapsedMs = Date.now() - started;
1177
+
1178
+ const blocks: SourceBlock[] = [];
1179
+ const listing: string[] = [];
1180
+ let windowed = 0;
1181
+ perFile.forEach((entry, file) => {
1182
+ const judgedUnits = owners
1183
+ .map((owner, index) => ({ ...owner, score: unitsOutcome.scores[index] }))
1184
+ .filter((owner) => owner.file === file);
1185
+ const answered = judgedUnits.some((owner) => owner.score !== undefined);
1186
+ const fallbackUnits = fallbackSourceUnits(
1187
+ entry.candidate.path,
1188
+ readText(entry.candidate.path) ?? "",
1189
+ ts,
1190
+ entry.rows,
1191
+ entry.hits,
1192
+ words,
1193
+ );
1194
+ const units =
1195
+ answered || entry.rows.length === 0 || entry.picks.length === 0
1196
+ ? judgedUnits.filter((owner) => (owner.score ?? 0) >= FIND_MIN_RELEVANCE).map((owner) => owner.unit)
1197
+ : fallbackUnits.length > 0
1198
+ ? fallbackUnits
1199
+ : keywordWindows(entry.rows.length, entry.hits);
1200
+ if (!answered && units.length > 0) windowed += 1;
1201
+ for (const unit of units) {
1202
+ blocks.push({ path: entry.candidate.path, unit, view: unitView(unit, focusOf(unit, entry.hits)), rows: entry.rows });
1203
+ }
1204
+ const leads = units.map((unit) => `${unit.name} lines ${unit.start}-${unit.end}`).join(", ");
1205
+ listing.push(`- ${entry.candidate.path} (${entry.candidate.relevance.toFixed(2)})${leads ? ` — reading leads: ${leads}` : ""}`);
1206
+ });
1207
+ if (windowed > 0) {
1208
+ notes.push(
1209
+ `Jev did not judge the source units of ${windowed} file(s) (${unitsOutcome.reasons.join(", ") || "missing"}); matching source units or keyword windows are shown instead.`,
1210
+ );
1211
+ }
1212
+ if (relevant.length > shown.length) {
1213
+ const rest = relevant.slice(shown.length);
1214
+ notes.push(
1215
+ `${rest.length} more relevant file(s) past the limit: ${rest
1216
+ .slice(0, 5)
1217
+ .map((candidate) => `${candidate.path} (${candidate.relevance.toFixed(2)})`)
1218
+ .join(", ")}${rest.length > 5 ? ", …" : ""}.`,
1219
+ );
1220
+ }
1221
+ const source = renderSourceBlocks(blocks);
1222
+ const diagnostic = requests.diagnostics[0];
1223
+ if (diagnostic) notes.push(`Diagnostic: ${formatJevDiagnostic(diagnostic)}`);
1224
+ if (source.omitted > 0) {
1225
+ notes.push(`${source.omitted} source block(s) left out to stay within ${FIND_MAX_SOURCE_BYTES / 1024} KB; read them from the leads.`);
569
1226
  }
570
- lines.push("Only pointers are returned; read the files you need.");
571
- return lines.join("\n");
1227
+
1228
+ const lines = [
1229
+ `jev_find: ${relevant.length} relevant file(s) (judged ${judged} of ${files.length} candidates via ${options.model} in ${elapsedMs}ms)`,
1230
+ ...notes,
1231
+ ...(shown.length > 0
1232
+ ? listing
1233
+ : [
1234
+ "No file is a confident match; the closest were:",
1235
+ ...ranked.slice(0, 3).map((candidate) => `- ${candidate.path} (${candidate.relevance.toFixed(2)})`),
1236
+ ]),
1237
+ ...source.lines,
1238
+ "",
1239
+ "End context.",
1240
+ ];
1241
+ const failure = filesOutcome.reasons.find((reason) => reason !== "request cap" && reason !== "too large");
1242
+ return {
1243
+ text: lines.join("\n"),
1244
+ details: {
1245
+ total: files.length,
1246
+ judged,
1247
+ relevant: relevant.length,
1248
+ requests: requests.sent,
1249
+ retried: requests.retried,
1250
+ sourceBytes: source.bytes,
1251
+ elapsedMs,
1252
+ ...(failure ? { failure } : {}),
1253
+ ...(diagnostic ? { diagnostic } : {}),
1254
+ },
1255
+ judged,
1256
+ topRelevance: ranked[0]?.relevance,
1257
+ ...(failure ? { failure } : {}),
1258
+ ...(diagnostic ? { diagnostic } : {}),
1259
+ elapsedMs,
1260
+ };
572
1261
  }
573
1262
 
574
1263
  const JevFindParams = Type.Object({
575
- question: Type.String({ description: "What you are looking for, in plain words (e.g. 'where is the retry backoff configured')." }),
1264
+ question: Type.String({
1265
+ description: "What you want to understand, in plain words (e.g. 'how is the retry backoff configured').",
1266
+ }),
576
1267
  pattern: Type.Optional(
577
- Type.String({ description: "ripgrep regex to narrow candidates to files that match; omit to judge every listed file." }),
1268
+ Type.String({ description: "ripgrep regex to narrow candidates to files that match; omit to consider every listed file." }),
578
1269
  ),
579
1270
  glob: Type.Optional(Type.String({ description: "File glob filter, e.g. '*.ts' or 'src/**/*.md'." })),
580
1271
  path: Type.Optional(Type.String({ description: "Directory to search, inside the working directory (default: '.')." })),
581
1272
  ignoreCase: Type.Optional(Type.Boolean({ description: "Case-insensitive pattern." })),
582
- limit: Type.Optional(Type.Number({ description: `Most results to return (default ${FIND_DEFAULT_LIMIT}).` })),
1273
+ limit: Type.Optional(
1274
+ Type.Number({ description: `Most relevant files to return, with their source (default ${FIND_DEFAULT_LIMIT}).` }),
1275
+ ),
583
1276
  });
584
1277
 
585
1278
  // ---------------------------------------------------------------------------
@@ -809,19 +1502,23 @@ export default function jevAskToolExtension(pi: ExtensionAPI): void {
809
1502
  name: JEV_FIND_TOOL,
810
1503
  label: "Jev Find",
811
1504
  description: [
812
- "Find which files answer a question without reading them: ripgrep gathers candidate files (those matching",
813
- "`pattern`, or every file under `path` filtered by `glob`), and Jev judges each file's excerpt against",
814
- "`question` in one parallel round trip. Returns ranked `path:lines (relevance)` pointers, never contents.",
815
- `Judges up to ${MAX_FIND_CANDIDATES} files per call; respects .gitignore; secret-named files are never sent.`,
1505
+ "Find where behavior lives and read it in one call. ripgrep gathers candidate files (those matching `pattern`,",
1506
+ "or every file under `path` filtered by `glob`); Jev narrows a large tree directory by directory, judges each",
1507
+ "remaining file's excerpt against `question`, then judges the functions, classes, and sections of the relevant",
1508
+ "files. Returns the relevant files with reading leads AND the accepted source verbatim with original line",
1509
+ `numbers (about ${FIND_MAX_SOURCE_BYTES / 1024} KB of source, at most ${FIND_MAX_REQUESTS} Jev requests per call).`,
1510
+ "Respects .gitignore; secret-named files are never sent.",
816
1511
  ].join(" "),
817
- promptSnippet: "Find the files that answer a question: ripgrep candidates ranked by Jev, pointers only",
1512
+ promptSnippet: "Find where behavior lives: Jev-judged relevant files plus their verbatim source with line numbers",
818
1513
  promptGuidelines: [
819
- "When you look for code by what it does rather than by an exact identifier you already know, call jev_find first — before grep, find, ls, or reading candidate files. One call ranks up to 48 files in about a second and returns file:line pointers; then read only the top ones. Pass `question` in plain words; add `pattern` when you know a likely identifier, `glob`/`path` to scope it.",
820
- "Use grep for an exact string or identifier you already know; use jev_find when you would otherwise grep several guesses or read files to see which one is relevant.",
1514
+ "Use jev_find for how/why/where-does-this-behavior-live questions — even when the question names a function or setting — before grep, find, ls, or reading candidate files. One call returns the relevant files and their relevant source verbatim with line numbers. Pass `question` in plain words; add `pattern` when you know a likely identifier, `glob`/`path` to scope it.",
1515
+ "Read the source blocks jev_find returns before searching again; `read` only the leads you still need (to edit, or past a clipped block).",
1516
+ "Use grep or read instead for an exact string, a known symbol's definition, or a known filename.",
1517
+ "When delegating repository discovery to a subagent, tell it to start with jev_find.",
821
1518
  ],
822
1519
  discovery: {
823
- summary: "Semantic file finder: ripgrep candidates ranked by Jev relevance",
824
- aliases: ["jevgrep", "find", "search", "locate"],
1520
+ summary: "Semantic code finder: Jev-judged relevant files and their verbatim source excerpts",
1521
+ aliases: ["jevgrep", "jg", "find", "search", "locate"],
825
1522
  category: "Decisions",
826
1523
  },
827
1524
  parameters: JevFindParams,
@@ -848,79 +1545,41 @@ export default function jevAskToolExtension(pi: ExtensionAPI): void {
848
1545
  const rg = await ensureTool("rg");
849
1546
  if (!rg) return text("jev_find needs ripgrep (rg), which is not available; use grep instead.");
850
1547
 
851
- const snippetBytes = Math.max(256, Math.floor((route.payloadBytes - FIND_QUESTION_RESERVE) / FIND_BATCH));
852
- let found: { candidates: FindCandidate[]; total: number };
1548
+ // Without Jev the ripgrep-ranked files and their keyword lines still come back: the call is never wasted.
1549
+ const connection = jevConnection(config, route);
1550
+ const unreachable = await jevUnavailable(ctx, connection);
1551
+ let result: JevFindResult;
853
1552
  try {
854
- found = await findCandidates(
855
- {
856
- rg,
857
- cwd: ctx.cwd,
858
- root: inside === "" ? "." : inside,
859
- question,
860
- pattern: params.pattern || undefined,
861
- glob: params.glob || undefined,
862
- ignoreCase: params.ignoreCase,
863
- },
864
- snippetBytes,
1553
+ result = await runJevFind({
1554
+ rg,
1555
+ cwd: ctx.cwd,
1556
+ root: inside === "" ? "." : inside,
1557
+ question,
1558
+ pattern: params.pattern || undefined,
1559
+ glob: params.glob || undefined,
1560
+ ignoreCase: params.ignoreCase,
1561
+ limit: params.limit,
1562
+ maxBytes: route.payloadBytes,
1563
+ model: config.model,
1564
+ ...(unreachable
1565
+ ? { unreachable }
1566
+ : { ask: (payload) => askJevAnswers(ctx, connection, { payload, maxBytes: route.payloadBytes }) }),
865
1567
  signal,
866
- );
1568
+ });
867
1569
  } catch (error) {
868
1570
  return text(`jev_find: ripgrep failed (${error instanceof Error ? error.message.split("\n")[0] : String(error)}).`);
869
1571
  }
870
- const limit = Math.max(1, Math.floor(params.limit ?? FIND_DEFAULT_LIMIT));
871
- const { candidates, total } = found;
872
- if (candidates.length === 0) {
873
- return text(renderFindResults({ ranked: [], total: 0, judged: 0, limit, model: config.model, elapsedMs: 0 }));
874
- }
875
-
876
- // Without Jev the candidates are still worth returning: the call is never wasted.
877
- const connection = jevConnection(config, route);
878
- const unreachable = await jevUnavailable(ctx, connection);
879
- const batches: FindCandidate[][] = [];
880
- for (let start = 0; start < candidates.length; start += FIND_BATCH) {
881
- batches.push(candidates.slice(start, start + FIND_BATCH));
882
- }
883
- const started = Date.now();
884
- const results = unreachable
885
- ? []
886
- : await Promise.all(
887
- batches.map((batch) =>
888
- askJevAnswers(ctx, connection, {
889
- payload: fitFindPayload(question, batch, route.payloadBytes),
890
- maxBytes: route.payloadBytes,
891
- }),
892
- ),
893
- );
894
- const elapsedMs = Date.now() - started;
895
-
896
- const scored: Array<FindCandidate & { relevance?: number }> = [];
897
- let judged = 0;
898
- batches.forEach((batch, b) => {
899
- const answers = results[b]?.answers ?? {};
900
- batch.forEach((candidate, i) => {
901
- const answer = answers[`f${i}`];
902
- const relevance = isRecord(answer) && typeof answer.noul === "number" ? answer.noul : undefined;
903
- if (relevance !== undefined) judged += 1;
904
- scored.push({ ...candidate, ...(relevance === undefined ? {} : { relevance }) });
1572
+ if (result.details.total !== 0) {
1573
+ emitJevTelemetry(pi.events, "decision", {
1574
+ route: "find",
1575
+ outcome: result.judged > 0 ? "jev" : "fallback",
1576
+ candidates: result.details.total,
1577
+ confidence: confidenceBucket(result.topRelevance),
1578
+ elapsedMs: result.elapsedMs,
1579
+ ...(result.judged > 0 ? {} : { reason: result.failure ?? "missing" }),
905
1580
  });
906
- });
907
- // Judged files by relevance; unjudged ones keep ripgrep's order behind them.
908
- const ranked = judged > 0 ? [...scored].sort((a, b) => (b.relevance ?? -1) - (a.relevance ?? -1)) : scored;
909
- const failure = unreachable ?? results.find((result) => result.failure)?.failure;
910
-
911
- emitJevTelemetry(pi.events, "decision", {
912
- route: "find",
913
- outcome: judged > 0 ? "jev" : "fallback",
914
- candidates: candidates.length,
915
- confidence: confidenceBucket(ranked[0]?.relevance),
916
- elapsedMs,
917
- ...(judged > 0 ? {} : { reason: failure ?? "missing" }),
918
- });
919
- return text(
920
- renderFindResults({ ranked, total, judged, limit, model: config.model, elapsedMs, failure }),
921
- // Shape only, like ask_jev: details persist in the session.
922
- { candidates: candidates.length, total, judged, elapsedMs, ...(failure ? { failure } : {}) },
923
- );
1581
+ }
1582
+ return text(result.text, result.details);
924
1583
  },
925
1584
  });
926
1585