@selesai/code 0.13.37 → 0.13.38
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +8 -0
- package/dist/extensions/jev/decisions.test.ts +18 -0
- package/dist/extensions/jev/decisions.ts +54 -6
- package/dist/extensions/jev-ask-tool.test.ts +371 -25
- package/dist/extensions/jev-ask-tool.ts +826 -167
- package/dist/extensions/jev-find-source.ts +190 -0
- package/dist/extensions/package.json +1 -1
- package/dist/extensions/pi-subagents/agents/researcher.md +4 -5
- package/dist/extensions/pi-subagents/docs/agents.md +1 -7
- package/dist/extensions/pi-subagents/test/unit/agent-frontmatter.test.ts +3 -4
- package/dist/extensions/pi-web-agent/src/extension.ts +11 -0
- package/docs/settings.md +8 -6
- package/package.json +1 -1
|
@@ -15,7 +15,7 @@
|
|
|
15
15
|
*/
|
|
16
16
|
import { execFile } from "node:child_process";
|
|
17
17
|
import { closeSync, openSync, readSync, realpathSync, statSync } from "node:fs";
|
|
18
|
-
import { basename, isAbsolute, normalize, relative, resolve } from "node:path";
|
|
18
|
+
import { basename, isAbsolute, normalize, relative, resolve, sep } from "node:path";
|
|
19
19
|
import { StringEnum } from "@earendil-works/pi-ai";
|
|
20
20
|
import {
|
|
21
21
|
createLocalBashOperations,
|
|
@@ -36,7 +36,10 @@ import {
|
|
|
36
36
|
UNTRUSTED_MATERIAL_FOCUS,
|
|
37
37
|
warnJevUnavailableOnce,
|
|
38
38
|
type JevAbstainReason,
|
|
39
|
+
type JevAnswers,
|
|
40
|
+
type JevFailureDiagnostic,
|
|
39
41
|
} from "./jev/decisions.ts";
|
|
42
|
+
import { isScriptPath, keywordWindows, loadTypeScript, type SourceUnit, splitSourceUnits } from "./jev-find-source.ts";
|
|
40
43
|
|
|
41
44
|
/** The tool's registered name. */
|
|
42
45
|
export const ASK_JEV_TOOL = "ask_jev";
|
|
@@ -342,29 +345,68 @@ export function renderAskAnswers(input: {
|
|
|
342
345
|
}
|
|
343
346
|
|
|
344
347
|
// ---------------------------------------------------------------------------
|
|
345
|
-
// jev_find: ripgrep gathers
|
|
348
|
+
// jev_find: ripgrep gathers files; Jev narrows directories, then files, then source units
|
|
346
349
|
// ---------------------------------------------------------------------------
|
|
347
350
|
|
|
348
351
|
/** The file finder's registered name. */
|
|
349
352
|
export const JEV_FIND_TOOL = "jev_find";
|
|
350
353
|
/** ponytail: 16 nouls per request (the JevPDF batch size); raise once Jev is measured on larger blocks. */
|
|
351
354
|
export const FIND_BATCH = 16;
|
|
352
|
-
/**
|
|
353
|
-
export const
|
|
355
|
+
/** ponytail: a set this small (three parallel batches) is judged file by file, without directory descent. */
|
|
356
|
+
export const FIND_DIRECT_FILES = 48;
|
|
357
|
+
/** ponytail: at most 128 files judged per call (eight batches), best word/path score first. */
|
|
358
|
+
export const FIND_MAX_JUDGED_FILES = 128;
|
|
359
|
+
/** ponytail: hard cap of 12 Jev requests per call, retries included; raise once latency and cost are measured. */
|
|
360
|
+
export const FIND_MAX_REQUESTS = 12;
|
|
361
|
+
/** ponytail: directory descent spends at most 4 of those requests (64 directory questions over all levels). */
|
|
362
|
+
export const FIND_DIR_REQUESTS = 4;
|
|
363
|
+
/** ponytail: 2 requests are held back for the source-unit stage (32 units). */
|
|
364
|
+
export const FIND_UNIT_REQUESTS = 2;
|
|
365
|
+
/** ponytail: about 16 KB of verbatim source per call; the rest stays reachable through the reading leads. */
|
|
366
|
+
export const FIND_MAX_SOURCE_BYTES = 16 * 1024;
|
|
367
|
+
/** ponytail: descend at most 4 directory levels below the search root. */
|
|
368
|
+
export const FIND_MAX_DEPTH = 4;
|
|
369
|
+
/** ponytail: keep a directory at 0.35 or above: pruning drops every file under it, so it errs toward keeping. */
|
|
370
|
+
export const FIND_DIR_KEEP = 0.35;
|
|
371
|
+
/** ponytail: at most 6 candidate units per relevant file, each shown as at most 60 lines. */
|
|
372
|
+
export const FIND_UNITS_PER_FILE = 6;
|
|
373
|
+
export const FIND_UNIT_MAX_LINES = 60;
|
|
374
|
+
const FIND_DIR_FILE_NAMES = 15;
|
|
375
|
+
const FIND_DIR_HIT_LINES = 3;
|
|
354
376
|
const FIND_MATCH_LINES = 6;
|
|
355
377
|
const FIND_DEFAULT_LIMIT = 8;
|
|
356
378
|
const FIND_MIN_RELEVANCE = 0.5;
|
|
357
|
-
/**
|
|
379
|
+
/** Files that still get keyword-window source when Jev could not rank them. */
|
|
380
|
+
const FIND_FALLBACK_SOURCE_FILES = 3;
|
|
381
|
+
/** The asker's question is clipped to this before it is repeated in every request. */
|
|
382
|
+
const FIND_REQUEST_BYTES = 2 * 1024;
|
|
383
|
+
/** Room each file batch keeps for the questions and JSON scaffolding when sizing snippets. */
|
|
358
384
|
const FIND_QUESTION_RESERVE = 6 * 1024;
|
|
385
|
+
/** Lines of a long unit shown above its best keyword line. */
|
|
386
|
+
const FIND_FOCUS_LEAD = 10;
|
|
387
|
+
|
|
388
|
+
/** One file ripgrep offered. */
|
|
389
|
+
export interface FindFile {
|
|
390
|
+
path: string;
|
|
391
|
+
/** Matched line numbers (pattern mode). */
|
|
392
|
+
lines: number[];
|
|
393
|
+
/** Matched lines as `L12: text` (pattern mode). */
|
|
394
|
+
hits: string[];
|
|
395
|
+
/** Question words in the path, plus matched lines (pattern mode) or question words in the content. */
|
|
396
|
+
score: number;
|
|
397
|
+
}
|
|
359
398
|
|
|
360
399
|
export interface FindCandidate {
|
|
361
400
|
path: string;
|
|
362
|
-
/** Matched line numbers
|
|
401
|
+
/** Matched line numbers, or the best keyword lines, best first. */
|
|
363
402
|
lines: number[];
|
|
364
|
-
/** What Jev reads: the matched lines, or the
|
|
403
|
+
/** What Jev reads: the matched lines, or the file's best keyword lines, or its head. */
|
|
365
404
|
snippet: string;
|
|
366
405
|
}
|
|
367
406
|
|
|
407
|
+
/** Sends one decisions request. The tool binds it to `askJevAnswers`; tests pass a fake. */
|
|
408
|
+
export type JevFindAsk = (payload: Record<string, unknown>) => Promise<JevAnswers>;
|
|
409
|
+
|
|
368
410
|
/** Run ripgrep with an argument list (no shell). Exit 1 is "no matches"; an overfull buffer keeps what arrived. */
|
|
369
411
|
function runRg(bin: string, args: string[], cwd: string, signal: AbortSignal | undefined): Promise<string> {
|
|
370
412
|
return new Promise((done, fail) => {
|
|
@@ -414,7 +456,7 @@ export function excerpt(text: string, words: readonly string[], maxBytes: number
|
|
|
414
456
|
if (best.length === 0) return { lines: [], snippet: clip(text, maxBytes).text };
|
|
415
457
|
const inOrder = [...best].sort((a, b) => a.index - b.index);
|
|
416
458
|
const snippet = inOrder.map((hit) => `L${hit.index + 1}: ${rows[hit.index].trim().slice(0, 240)}`).join("\n");
|
|
417
|
-
//
|
|
459
|
+
// Leads start at the best-matching lines.
|
|
418
460
|
return { lines: best.map((hit) => hit.index + 1), snippet: clip(snippet, maxBytes).text };
|
|
419
461
|
}
|
|
420
462
|
|
|
@@ -424,22 +466,32 @@ function pathOverlap(path: string, words: readonly string[]): number {
|
|
|
424
466
|
return parts.filter((part) => words.some((word) => word.includes(part) || part.includes(word))).length;
|
|
425
467
|
}
|
|
426
468
|
|
|
469
|
+
const byScore = (a: FindFile, b: FindFile) => b.score - a.score || a.path.localeCompare(b.path);
|
|
470
|
+
|
|
471
|
+
export interface FindGatherOptions {
|
|
472
|
+
rg: string;
|
|
473
|
+
cwd: string;
|
|
474
|
+
/** Search root relative to `cwd`; "." for the working directory. */
|
|
475
|
+
root: string;
|
|
476
|
+
question: string;
|
|
477
|
+
pattern?: string;
|
|
478
|
+
glob?: string;
|
|
479
|
+
ignoreCase?: boolean;
|
|
480
|
+
}
|
|
481
|
+
|
|
427
482
|
/**
|
|
428
|
-
*
|
|
429
|
-
*
|
|
430
|
-
*
|
|
431
|
-
* .gitignore, and never offers a file `askPath` would refuse.
|
|
483
|
+
* Every file under `root` worth judging, most promising first. With `pattern`, the files ripgrep
|
|
484
|
+
* matches with their matched lines; without it, every file ripgrep lists, scored by question words
|
|
485
|
+
* in the path and (for a set too large to judge file by file) in the content, one fixed-string
|
|
486
|
+
* ripgrep per word, in parallel. Respects .gitignore, and never offers a file `askPath` would refuse.
|
|
432
487
|
*/
|
|
433
|
-
export async function
|
|
434
|
-
options: { rg: string; cwd: string; root: string; question: string; pattern?: string; glob?: string; ignoreCase?: boolean },
|
|
435
|
-
snippetBytes: number,
|
|
436
|
-
signal: AbortSignal | undefined,
|
|
437
|
-
): Promise<{ candidates: FindCandidate[]; total: number }> {
|
|
488
|
+
export async function gatherFindFiles(options: FindGatherOptions, signal: AbortSignal | undefined): Promise<FindFile[]> {
|
|
438
489
|
// Always name the root: with no path and a piped stdin, rg searches stdin and hangs.
|
|
439
490
|
// Paths come back as `./src/x.ts`, so they are normalized below.
|
|
440
491
|
const root = ["--", options.root];
|
|
441
492
|
const filters = [...(options.glob ? ["--glob", options.glob] : []), ...(options.ignoreCase ? ["-i"] : [])];
|
|
442
493
|
const allowed = (path: string) => "full" in askPath(path, options.cwd);
|
|
494
|
+
const words = questionWords(options.question);
|
|
443
495
|
if (options.pattern) {
|
|
444
496
|
const out = await runRg(
|
|
445
497
|
options.rg,
|
|
@@ -459,17 +511,11 @@ export async function findCandidates(
|
|
|
459
511
|
entry.text.push(`L${match[1]}: ${match[2].trim()}`);
|
|
460
512
|
byFile.set(path, entry);
|
|
461
513
|
}
|
|
462
|
-
|
|
463
|
-
|
|
464
|
-
|
|
465
|
-
|
|
466
|
-
path,
|
|
467
|
-
lines: entry.lines,
|
|
468
|
-
snippet: clip(entry.text.join("\n"), snippetBytes).text,
|
|
469
|
-
})),
|
|
470
|
-
};
|
|
514
|
+
return [...byFile]
|
|
515
|
+
.filter(([path]) => allowed(path))
|
|
516
|
+
.map(([path, entry]) => ({ path, lines: entry.lines, hits: entry.text, score: entry.lines.length + pathOverlap(path, words) }))
|
|
517
|
+
.sort(byScore);
|
|
471
518
|
}
|
|
472
|
-
const words = questionWords(options.question);
|
|
473
519
|
const out = await runRg(options.rg, ["--files", "--color=never", ...filters, ...root], options.cwd, signal);
|
|
474
520
|
const listed = out
|
|
475
521
|
.split("\n")
|
|
@@ -477,9 +523,7 @@ export async function findCandidates(
|
|
|
477
523
|
.map((path) => normalize(path))
|
|
478
524
|
.filter(allowed);
|
|
479
525
|
const score = new Map(listed.map((path) => [path, pathOverlap(path, words)]));
|
|
480
|
-
if (listed.length >
|
|
481
|
-
// More files than Jev judges: a file scores once per distinct question word its content holds,
|
|
482
|
-
// one fixed-string ripgrep per word, in parallel.
|
|
526
|
+
if (listed.length > FIND_DIRECT_FILES && words.length > 0) {
|
|
483
527
|
const hits = await Promise.all(
|
|
484
528
|
words.map((word) =>
|
|
485
529
|
runRg(options.rg, ["-l", "-i", "-F", "--color=never", ...filters, "-e", word, ...root], options.cwd, signal),
|
|
@@ -493,93 +537,742 @@ export async function findCandidates(
|
|
|
493
537
|
}
|
|
494
538
|
}
|
|
495
539
|
}
|
|
496
|
-
|
|
497
|
-
|
|
498
|
-
|
|
499
|
-
|
|
500
|
-
|
|
501
|
-
|
|
502
|
-
|
|
503
|
-
|
|
540
|
+
return listed.map((path) => ({ path, lines: [], hits: [], score: score.get(path) ?? 0 })).sort(byScore);
|
|
541
|
+
}
|
|
542
|
+
|
|
543
|
+
// Judging ------------------------------------------------------------------
|
|
544
|
+
|
|
545
|
+
/** One noul: a keyed piece of state and the question asked about it. */
|
|
546
|
+
export interface JudgeItem {
|
|
547
|
+
key: string;
|
|
548
|
+
text: string;
|
|
549
|
+
question: string;
|
|
550
|
+
}
|
|
551
|
+
|
|
552
|
+
export interface JudgeStage {
|
|
553
|
+
request: string;
|
|
554
|
+
/** Where the items sit in `state` (`directories`, `files`, `units`). */
|
|
555
|
+
stateKey: string;
|
|
556
|
+
criteria: { true: string; false: string };
|
|
504
557
|
}
|
|
505
558
|
|
|
506
|
-
/** One
|
|
507
|
-
export function
|
|
559
|
+
/** One request: the items at `items` (indexes into the stage's list), one noul `q<i>` each. */
|
|
560
|
+
export function buildJudgePayload(stage: JudgeStage, items: readonly JudgeItem[]): Record<string, unknown> {
|
|
508
561
|
return buildAskPayload(
|
|
509
|
-
|
|
510
|
-
|
|
511
|
-
type: "noul" as const,
|
|
512
|
-
instructions: `Is the file "${candidate.path}" relevant to the request, judged by its excerpt in state.files?`,
|
|
513
|
-
criteria: {
|
|
514
|
-
true: "The excerpt shows this file implements, defines, configures, or directly answers the request.",
|
|
515
|
-
false: "The file only mentions related words, or is about something else.",
|
|
516
|
-
},
|
|
517
|
-
})),
|
|
518
|
-
{ request: question, files: Object.fromEntries(batch.map((candidate) => [candidate.path, candidate.snippet])) },
|
|
562
|
+
items.map((item, index) => ({ name: `q${index}`, type: "noul" as const, instructions: item.question, criteria: stage.criteria })),
|
|
563
|
+
{ request: stage.request, [stage.stateKey]: Object.fromEntries(items.map((item) => [item.key, item.text])) },
|
|
519
564
|
);
|
|
520
565
|
}
|
|
521
566
|
|
|
522
567
|
/**
|
|
523
|
-
* The
|
|
524
|
-
*
|
|
568
|
+
* The stage's items as requests that each serialize within `maxBytes`: FIND_BATCH items per
|
|
569
|
+
* request, their texts shrunk together until the request fits (code grows under JSON escaping, so a
|
|
570
|
+
* byte budget per text alone does not guarantee a fit). When even empty texts do not fit — the
|
|
571
|
+
* questions, paths, and scaffolding alone are too large — the batch is halved, and an item that
|
|
572
|
+
* cannot fit alone is dropped. No request this returns can come back as `overflow`.
|
|
573
|
+
*/
|
|
574
|
+
export function fitJudgePayloads(
|
|
575
|
+
stage: JudgeStage,
|
|
576
|
+
items: readonly JudgeItem[],
|
|
577
|
+
maxBytes: number,
|
|
578
|
+
): { payloads: Array<{ payload: Record<string, unknown>; items: number[] }>; dropped: number[] } {
|
|
579
|
+
const payloads: Array<{ payload: Record<string, unknown>; items: number[] }> = [];
|
|
580
|
+
const dropped: number[] = [];
|
|
581
|
+
const place = (indexes: number[]): void => {
|
|
582
|
+
let limit = Math.max(0, ...indexes.map((index) => Buffer.byteLength(items[index].text, "utf-8")));
|
|
583
|
+
for (;;) {
|
|
584
|
+
const payload = buildJudgePayload(
|
|
585
|
+
stage,
|
|
586
|
+
indexes.map((index) => {
|
|
587
|
+
const text = clip(items[index].text, limit);
|
|
588
|
+
return { ...items[index], text: text.clipped && limit > 0 ? `${text.text}${TRUNCATION_MARKER}` : text.text };
|
|
589
|
+
}),
|
|
590
|
+
);
|
|
591
|
+
if (serializeJevRequest(payload, maxBytes) !== undefined) {
|
|
592
|
+
payloads.push({ payload, items: indexes });
|
|
593
|
+
return;
|
|
594
|
+
}
|
|
595
|
+
if (limit > 0) {
|
|
596
|
+
limit = limit < 64 ? 0 : Math.floor(limit * 0.75);
|
|
597
|
+
continue;
|
|
598
|
+
}
|
|
599
|
+
if (indexes.length === 1) {
|
|
600
|
+
dropped.push(indexes[0]);
|
|
601
|
+
return;
|
|
602
|
+
}
|
|
603
|
+
const half = Math.ceil(indexes.length / 2);
|
|
604
|
+
place(indexes.slice(0, half));
|
|
605
|
+
place(indexes.slice(half));
|
|
606
|
+
return;
|
|
607
|
+
}
|
|
608
|
+
};
|
|
609
|
+
for (let start = 0; start < items.length; start += FIND_BATCH) {
|
|
610
|
+
place(Array.from({ length: Math.min(FIND_BATCH, items.length - start) }, (_, offset) => start + offset));
|
|
611
|
+
}
|
|
612
|
+
return { payloads, dropped };
|
|
613
|
+
}
|
|
614
|
+
|
|
615
|
+
/**
|
|
616
|
+
* The call's Jev requests: every stage sends its batches in parallel (one round trip), retries a
|
|
617
|
+
* `transport` failure once, and never exceeds FIND_MAX_REQUESTS in total, retries included.
|
|
525
618
|
*/
|
|
526
|
-
export
|
|
527
|
-
|
|
528
|
-
|
|
529
|
-
|
|
530
|
-
|
|
531
|
-
|
|
619
|
+
export class FindRequests {
|
|
620
|
+
sent = 0;
|
|
621
|
+
retried = 0;
|
|
622
|
+
readonly diagnostics: JevFailureDiagnostic[] = [];
|
|
623
|
+
private readonly ask: JevFindAsk;
|
|
624
|
+
readonly max: number;
|
|
625
|
+
constructor(ask: JevFindAsk, max = FIND_MAX_REQUESTS) {
|
|
626
|
+
this.ask = ask;
|
|
627
|
+
this.max = max;
|
|
628
|
+
}
|
|
629
|
+
|
|
630
|
+
get left(): number {
|
|
631
|
+
return this.max - this.sent;
|
|
632
|
+
}
|
|
633
|
+
|
|
634
|
+
/**
|
|
635
|
+
* Send up to `sendCap` payloads in parallel, then retry the transport failures within `retryCap`
|
|
636
|
+
* total requests for this stage. A payload past the caps comes back undefined (not sent).
|
|
637
|
+
*/
|
|
638
|
+
async sendAll(
|
|
639
|
+
payloads: readonly Record<string, unknown>[],
|
|
640
|
+
sendCap: number,
|
|
641
|
+
retryCap = sendCap,
|
|
642
|
+
): Promise<Array<JevAnswers | undefined>> {
|
|
643
|
+
const count = Math.max(0, Math.min(payloads.length, sendCap, this.left));
|
|
644
|
+
this.sent += count;
|
|
645
|
+
const results: Array<JevAnswers | undefined> = await Promise.all(
|
|
646
|
+
payloads.slice(0, count).map((payload) => this.ask(payload)),
|
|
532
647
|
);
|
|
533
|
-
|
|
534
|
-
|
|
648
|
+
const retry = results
|
|
649
|
+
.map((result, index) => ({ result, index }))
|
|
650
|
+
.filter(({ result }) => result?.failure === "transport")
|
|
651
|
+
.slice(0, Math.max(0, Math.min(retryCap - count, this.left)));
|
|
652
|
+
this.sent += retry.length;
|
|
653
|
+
this.retried += retry.length;
|
|
654
|
+
const again = await Promise.all(retry.map(({ index }) => this.ask(payloads[index])));
|
|
655
|
+
retry.forEach(({ index }, k) => {
|
|
656
|
+
results[index] = again[k];
|
|
657
|
+
});
|
|
658
|
+
for (const result of results) {
|
|
659
|
+
const diagnostic = result?.diagnostic;
|
|
660
|
+
if (diagnostic && !this.diagnostics.some((item) => item.kind === diagnostic.kind && item.httpStatus === diagnostic.httpStatus)) {
|
|
661
|
+
this.diagnostics.push(diagnostic);
|
|
662
|
+
}
|
|
663
|
+
}
|
|
664
|
+
return [...results, ...Array.from({ length: payloads.length - count }, () => undefined)];
|
|
535
665
|
}
|
|
536
666
|
}
|
|
537
667
|
|
|
538
|
-
|
|
539
|
-
|
|
540
|
-
|
|
541
|
-
|
|
542
|
-
|
|
543
|
-
|
|
668
|
+
interface JudgeOutcome {
|
|
669
|
+
/** Jev's noul per item; undefined when it was not answered. */
|
|
670
|
+
scores: Array<number | undefined>;
|
|
671
|
+
/** Why unanswered items went unanswered: a request failure, `request cap`, or `too large`. */
|
|
672
|
+
reasons: string[];
|
|
673
|
+
}
|
|
674
|
+
|
|
675
|
+
/** Ask one noul per item, batched and fitted, within the stage's request caps. */
|
|
676
|
+
async function judgeNouls(
|
|
677
|
+
requests: FindRequests,
|
|
678
|
+
stage: JudgeStage,
|
|
679
|
+
items: readonly JudgeItem[],
|
|
680
|
+
maxBytes: number,
|
|
681
|
+
caps: { send: number; retry: number },
|
|
682
|
+
): Promise<JudgeOutcome> {
|
|
683
|
+
const scores: Array<number | undefined> = items.map(() => undefined);
|
|
684
|
+
const reasons: string[] = [];
|
|
685
|
+
if (items.length === 0) return { scores, reasons };
|
|
686
|
+
const fitted = fitJudgePayloads(stage, items, maxBytes);
|
|
687
|
+
if (fitted.dropped.length > 0) reasons.push("too large");
|
|
688
|
+
const results = await requests.sendAll(
|
|
689
|
+
fitted.payloads.map((entry) => entry.payload),
|
|
690
|
+
caps.send,
|
|
691
|
+
caps.retry,
|
|
692
|
+
);
|
|
693
|
+
fitted.payloads.forEach((entry, p) => {
|
|
694
|
+
const result = results[p];
|
|
695
|
+
if (!result) {
|
|
696
|
+
reasons.push("request cap");
|
|
697
|
+
return;
|
|
698
|
+
}
|
|
699
|
+
if (result.failure) reasons.push(result.failure);
|
|
700
|
+
const answers = result.answers ?? {};
|
|
701
|
+
entry.items.forEach((index, k) => {
|
|
702
|
+
const answer = answers[`q${k}`];
|
|
703
|
+
if (isRecord(answer) && typeof answer.noul === "number") scores[index] = answer.noul;
|
|
704
|
+
});
|
|
705
|
+
});
|
|
706
|
+
return { scores, reasons: [...new Set(reasons)] };
|
|
707
|
+
}
|
|
708
|
+
|
|
709
|
+
// Directory descent --------------------------------------------------------
|
|
710
|
+
|
|
711
|
+
const DIRECTORY_CRITERIA = {
|
|
712
|
+
true: "Its name, file names, or keyword hits suggest it holds code that implements, defines, or configures what the request asks about.",
|
|
713
|
+
false: "It clearly holds something unrelated to the request.",
|
|
714
|
+
};
|
|
715
|
+
|
|
716
|
+
/** The directory one level below `dir` that holds `path`, or undefined when `path` sits directly in `dir`. */
|
|
717
|
+
export function childDirectory(path: string, dir: string): string | undefined {
|
|
718
|
+
const rest = dir === "." ? path : path.slice(dir.length + 1);
|
|
719
|
+
const cut = rest.indexOf(sep);
|
|
720
|
+
if (cut < 0) return undefined;
|
|
721
|
+
return dir === "." ? rest.slice(0, cut) : `${dir}${sep}${rest.slice(0, cut)}`;
|
|
722
|
+
}
|
|
723
|
+
|
|
724
|
+
/** What Jev reads of a directory: its size, its most promising file names, and a few keyword hits. */
|
|
725
|
+
function describeDirectory(dir: string, files: readonly FindFile[], hitLines: (file: FindFile) => string[]): string {
|
|
726
|
+
const ordered = [...files].sort(byScore);
|
|
727
|
+
const local = (path: string) => path.slice(dir.length + 1);
|
|
728
|
+
const names = ordered.slice(0, FIND_DIR_FILE_NAMES).map((file) => local(file.path));
|
|
729
|
+
const more = files.length - names.length;
|
|
730
|
+
const hits: string[] = [];
|
|
731
|
+
for (const file of ordered.filter((candidate) => candidate.score > 0).slice(0, FIND_DIR_HIT_LINES)) {
|
|
732
|
+
for (const line of hitLines(file).slice(0, 2)) {
|
|
733
|
+
if (hits.length < FIND_DIR_HIT_LINES) hits.push(`${local(file.path)} ${line.slice(0, 160)}`);
|
|
734
|
+
}
|
|
735
|
+
}
|
|
736
|
+
return [
|
|
737
|
+
`${files.length} file(s): ${names.join(", ")}${more > 0 ? `, … (+${more} more)` : ""}`,
|
|
738
|
+
...(hits.length > 0 ? ["Keyword hits:", ...hits] : []),
|
|
739
|
+
].join("\n");
|
|
740
|
+
}
|
|
741
|
+
|
|
742
|
+
export interface DescentResult {
|
|
743
|
+
/** Files still worth judging: those in kept directories, and those sitting directly in a descended one. */
|
|
744
|
+
kept: FindFile[];
|
|
745
|
+
prunedDirs: number;
|
|
746
|
+
prunedFiles: number;
|
|
747
|
+
/** Directories kept without an answer, and why. */
|
|
748
|
+
unjudgedDirs: number;
|
|
749
|
+
reasons: string[];
|
|
750
|
+
}
|
|
751
|
+
|
|
752
|
+
/**
|
|
753
|
+
* Narrow a large file set directory by directory: one noul per next-level directory ("could this
|
|
754
|
+
* hold the answer?"), all of a level in one parallel round trip. A directory below FIND_DIR_KEEP is
|
|
755
|
+
* dropped with every file under it; an unanswered one is kept (unknown is not rejected). Kept
|
|
756
|
+
* directories still larger than FIND_DIRECT_FILES are descended again, up to FIND_MAX_DEPTH
|
|
757
|
+
* levels; a lone subdirectory is entered without a question.
|
|
758
|
+
*/
|
|
759
|
+
export async function narrowByDirectory(
|
|
760
|
+
files: readonly FindFile[],
|
|
761
|
+
root: string,
|
|
762
|
+
options: { request: string; maxBytes: number; requests: FindRequests; hitLines: (file: FindFile) => string[] },
|
|
763
|
+
): Promise<DescentResult> {
|
|
764
|
+
const result: DescentResult = { kept: [], prunedDirs: 0, prunedFiles: 0, unjudgedDirs: 0, reasons: [] };
|
|
765
|
+
if (files.length <= FIND_DIRECT_FILES) return { ...result, kept: [...files] };
|
|
766
|
+
let pending: Array<{ dir: string; files: FindFile[]; depth: number }> = [{ dir: root, files: [...files], depth: 0 }];
|
|
767
|
+
let dirRequests = 0;
|
|
768
|
+
while (pending.length > 0) {
|
|
769
|
+
const children: Array<{ dir: string; files: FindFile[]; depth: number; free: boolean }> = [];
|
|
770
|
+
for (const group of pending) {
|
|
771
|
+
const byChild = new Map<string, FindFile[]>();
|
|
772
|
+
let direct = 0;
|
|
773
|
+
for (const file of group.files) {
|
|
774
|
+
const child = childDirectory(file.path, group.dir);
|
|
775
|
+
if (child === undefined) {
|
|
776
|
+
result.kept.push(file);
|
|
777
|
+
direct += 1;
|
|
778
|
+
} else {
|
|
779
|
+
byChild.set(child, [...(byChild.get(child) ?? []), file]);
|
|
780
|
+
}
|
|
781
|
+
}
|
|
782
|
+
const free = byChild.size === 1 && direct === 0;
|
|
783
|
+
for (const [dir, under] of byChild) children.push({ dir, files: under, depth: group.depth + 1, free });
|
|
784
|
+
}
|
|
785
|
+
const sendCap = Math.min(FIND_DIR_REQUESTS - dirRequests, options.requests.left - FIND_UNIT_REQUESTS - 1);
|
|
786
|
+
const questionable = children.filter((child) => !child.free && child.depth <= FIND_MAX_DEPTH);
|
|
787
|
+
const asked =
|
|
788
|
+
sendCap > 0
|
|
789
|
+
? questionable.sort(
|
|
790
|
+
(a, b) => b.files.reduce((sum, f) => sum + f.score, 0) - a.files.reduce((sum, f) => sum + f.score, 0),
|
|
791
|
+
)
|
|
792
|
+
: [];
|
|
793
|
+
if (children.some((child) => !child.free && child.depth > FIND_MAX_DEPTH)) result.reasons.push("depth limit");
|
|
794
|
+
if (questionable.length > 0 && asked.length === 0) result.reasons.push("request cap");
|
|
795
|
+
const verdicts = new Map<string, number | undefined>();
|
|
796
|
+
if (asked.length > 0) {
|
|
797
|
+
const before = options.requests.sent;
|
|
798
|
+
const outcome = await judgeNouls(
|
|
799
|
+
options.requests,
|
|
800
|
+
{ request: options.request, stateKey: "directories", criteria: DIRECTORY_CRITERIA },
|
|
801
|
+
asked.map((child) => ({
|
|
802
|
+
key: `${child.dir}${sep}`,
|
|
803
|
+
text: describeDirectory(child.dir, child.files, options.hitLines),
|
|
804
|
+
question: `Could the directory "${child.dir}${sep}" contain code that answers the request, judged by its description in state.directories?`,
|
|
805
|
+
})),
|
|
806
|
+
options.maxBytes,
|
|
807
|
+
{ send: sendCap, retry: sendCap },
|
|
808
|
+
);
|
|
809
|
+
dirRequests += options.requests.sent - before;
|
|
810
|
+
asked.forEach((child, index) => verdicts.set(child.dir, outcome.scores[index]));
|
|
811
|
+
result.reasons.push(...outcome.reasons);
|
|
812
|
+
}
|
|
813
|
+
const next: typeof pending = [];
|
|
814
|
+
for (const child of children) {
|
|
815
|
+
const wasAsked = verdicts.has(child.dir);
|
|
816
|
+
const verdict = verdicts.get(child.dir);
|
|
817
|
+
if (verdict !== undefined && verdict < FIND_DIR_KEEP) {
|
|
818
|
+
result.prunedDirs += 1;
|
|
819
|
+
result.prunedFiles += child.files.length;
|
|
820
|
+
continue;
|
|
821
|
+
}
|
|
822
|
+
if (verdict === undefined && !child.free) result.unjudgedDirs += 1;
|
|
823
|
+
const canDescend = child.depth < FIND_MAX_DEPTH && (child.free || wasAsked);
|
|
824
|
+
if (child.files.length > FIND_DIRECT_FILES && canDescend) {
|
|
825
|
+
next.push({ dir: child.dir, files: child.files, depth: child.depth });
|
|
826
|
+
} else result.kept.push(...child.files);
|
|
827
|
+
}
|
|
828
|
+
pending = next;
|
|
829
|
+
}
|
|
830
|
+
result.reasons = [...new Set(result.reasons)];
|
|
831
|
+
return result;
|
|
832
|
+
}
|
|
833
|
+
|
|
834
|
+
// Source units -------------------------------------------------------------
|
|
835
|
+
|
|
836
|
+
const FILE_CRITERIA = {
|
|
837
|
+
true: "The excerpt shows this file implements, defines, configures, or directly answers the request.",
|
|
838
|
+
false: "The file only mentions related words, or is about something else.",
|
|
839
|
+
};
|
|
840
|
+
|
|
841
|
+
const UNIT_CRITERIA = {
|
|
842
|
+
true: "This unit implements, defines, configures, or directly answers what the request asks.",
|
|
843
|
+
false: "The unit only mentions related words, or does something else.",
|
|
844
|
+
};
|
|
845
|
+
|
|
846
|
+
/** A unit's candidate score: matched/keyword lines inside it weigh most, then distinct question words. */
|
|
847
|
+
export function pickUnits(
|
|
848
|
+
units: readonly SourceUnit[],
|
|
849
|
+
rows: readonly string[],
|
|
850
|
+
hits: readonly number[],
|
|
851
|
+
words: readonly string[],
|
|
852
|
+
): SourceUnit[] {
|
|
853
|
+
const bodies = units.filter((unit) => unit.name !== "imports");
|
|
854
|
+
const scored = bodies.map((unit) => {
|
|
855
|
+
const body = rows.slice(unit.start - 1, unit.end).join("\n").toLowerCase();
|
|
856
|
+
const hitCount = hits.filter((line) => line >= unit.start && line <= unit.end).length;
|
|
857
|
+
return { unit, score: hitCount * 10 + words.filter((word) => body.includes(word)).length };
|
|
858
|
+
});
|
|
859
|
+
const picks = scored
|
|
860
|
+
.filter((entry) => entry.score > 0)
|
|
861
|
+
.sort((a, b) => b.score - a.score || a.unit.start - b.unit.start)
|
|
862
|
+
.slice(0, FIND_UNITS_PER_FILE)
|
|
863
|
+
.map((entry) => entry.unit);
|
|
864
|
+
// Nothing matched a word: let Jev judge the file's first declarations.
|
|
865
|
+
return (picks.length > 0 ? picks : bodies.slice(0, 3)).sort((a, b) => a.start - b.start);
|
|
866
|
+
}
|
|
867
|
+
|
|
868
|
+
/**
|
|
869
|
+
* The line numbers a unit is shown as (0 marks elided lines): the header line of a member, then the
|
|
870
|
+
* unit whole, or — when it is longer than `maxLines` — its first line and a window from just above
|
|
871
|
+
* its best keyword line (`focus`), so a long function shows where the asked-about behavior sits.
|
|
872
|
+
*/
|
|
873
|
+
export function unitView(unit: SourceUnit, focus: number | undefined, maxLines = FIND_UNIT_MAX_LINES): number[] {
|
|
874
|
+
const view: number[] = unit.header !== undefined && unit.header < unit.start ? [unit.header, 0] : [];
|
|
875
|
+
const range = (from: number, to: number) => Array.from({ length: Math.max(0, to - from + 1) }, (_, i) => from + i);
|
|
876
|
+
if (unit.end - unit.start + 1 <= maxLines) return [...view, ...range(unit.start, unit.end)];
|
|
877
|
+
if (focus === undefined || focus < unit.start + maxLines - FIND_FOCUS_LEAD) {
|
|
878
|
+
return [...view, ...range(unit.start, unit.start + maxLines - 1), 0];
|
|
879
|
+
}
|
|
880
|
+
const from = Math.max(unit.start + 1, focus - FIND_FOCUS_LEAD);
|
|
881
|
+
const to = Math.min(unit.end, from + maxLines - 2);
|
|
882
|
+
return [...view, unit.start, ...(from > unit.start + 1 ? [0] : []), ...range(from, to), ...(to < unit.end ? [0] : [])];
|
|
883
|
+
}
|
|
884
|
+
|
|
885
|
+
/** The first `hits` line inside the unit, in `hits` order (best first). */
|
|
886
|
+
function focusOf(unit: SourceUnit, hits: readonly number[]): number | undefined {
|
|
887
|
+
return hits.find((line) => line >= unit.start && line <= unit.end);
|
|
888
|
+
}
|
|
889
|
+
|
|
890
|
+
/** A file's rows without carriage returns (and without the empty row after a final newline), so numbered lines print as they read. */
|
|
891
|
+
function fileRows(text: string): string[] {
|
|
892
|
+
const rows = text.split("\n").map((row) => row.replace(/\r$/, ""));
|
|
893
|
+
if (rows.length > 1 && rows.at(-1) === "") rows.pop();
|
|
894
|
+
return rows;
|
|
895
|
+
}
|
|
896
|
+
|
|
897
|
+
interface SourceBlock {
|
|
898
|
+
path: string;
|
|
899
|
+
unit: SourceUnit;
|
|
900
|
+
view: number[];
|
|
901
|
+
rows: readonly string[];
|
|
902
|
+
}
|
|
903
|
+
|
|
904
|
+
/**
|
|
905
|
+
* `Source block "path" lines a-b:` and the numbered verbatim lines of each block, in order, within
|
|
906
|
+
* `budgetBytes` of source; a block cut by the budget ends in `…`, and one that cannot show three
|
|
907
|
+
* lines is left out.
|
|
908
|
+
*/
|
|
909
|
+
export function renderSourceBlocks(
|
|
910
|
+
blocks: readonly SourceBlock[],
|
|
911
|
+
budgetBytes = FIND_MAX_SOURCE_BYTES,
|
|
912
|
+
): { lines: string[]; bytes: number; omitted: number } {
|
|
913
|
+
const lines: string[] = [];
|
|
914
|
+
let bytes = 0;
|
|
915
|
+
let omitted = 0;
|
|
916
|
+
for (const block of blocks) {
|
|
917
|
+
const numbered: string[] = [];
|
|
918
|
+
let used = 0;
|
|
919
|
+
let cut = false;
|
|
920
|
+
for (const line of block.view) {
|
|
921
|
+
const text = line === 0 ? "…" : `${line}: ${block.rows[line - 1] ?? ""}`;
|
|
922
|
+
const size = Buffer.byteLength(text, "utf-8") + 1;
|
|
923
|
+
if (bytes + used + size > budgetBytes) {
|
|
924
|
+
cut = true;
|
|
925
|
+
break;
|
|
926
|
+
}
|
|
927
|
+
numbered.push(text);
|
|
928
|
+
used += size;
|
|
929
|
+
}
|
|
930
|
+
if (numbered.filter((text) => text !== "…").length < Math.min(3, block.view.filter((line) => line !== 0).length)) {
|
|
931
|
+
omitted += 1;
|
|
932
|
+
continue;
|
|
933
|
+
}
|
|
934
|
+
if (cut && numbered.at(-1) !== "…") numbered.push("…");
|
|
935
|
+
lines.push("", `Source block "${block.path}" lines ${block.unit.start}-${block.unit.end}:`, ...numbered);
|
|
936
|
+
bytes += used;
|
|
937
|
+
}
|
|
938
|
+
return { lines, bytes, omitted };
|
|
939
|
+
}
|
|
940
|
+
|
|
941
|
+
/** Keyword lines for a directory description or a fallback window: rg's matches, or the file's best question-word lines. */
|
|
942
|
+
function keywordLines(text: string | undefined, words: readonly string[]): { lines: number[]; snippet: string[] } {
|
|
943
|
+
if (text === undefined || words.length === 0) return { lines: [], snippet: [] };
|
|
944
|
+
const found = excerpt(text, words, FIND_SCAN_BYTES);
|
|
945
|
+
return found.lines.length === 0 ? { lines: [], snippet: [] } : { lines: found.lines, snippet: found.snippet.split("\n") };
|
|
946
|
+
}
|
|
947
|
+
|
|
948
|
+
function fallbackSourceUnits(
|
|
949
|
+
path: string,
|
|
950
|
+
text: string,
|
|
951
|
+
ts: Awaited<ReturnType<typeof loadTypeScript>>,
|
|
952
|
+
rows: readonly string[],
|
|
953
|
+
hits: readonly number[],
|
|
954
|
+
words: readonly string[],
|
|
955
|
+
): SourceUnit[] {
|
|
956
|
+
if (hits.length === 0) return [];
|
|
957
|
+
return pickUnits(splitSourceUnits(path, text, ts), rows, hits, words).filter((unit) =>
|
|
958
|
+
hits.some((line) => (line >= unit.start && line <= unit.end) || line === unit.header),
|
|
959
|
+
);
|
|
960
|
+
}
|
|
961
|
+
|
|
962
|
+
function formatJevDiagnostic(diagnostic: JevFailureDiagnostic): string {
|
|
963
|
+
if (diagnostic.kind === "timeout") return "Jev request timed out.";
|
|
964
|
+
if (diagnostic.kind === "malformed-response") return "Jev returned a malformed response.";
|
|
965
|
+
return `Jev provider request failed${diagnostic.httpStatus ? ` (HTTP ${diagnostic.httpStatus})` : ""}.`;
|
|
966
|
+
}
|
|
967
|
+
|
|
968
|
+
// The whole call -----------------------------------------------------------
|
|
969
|
+
|
|
970
|
+
export interface JevFindOptions extends FindGatherOptions {
|
|
971
|
+
limit?: number;
|
|
972
|
+
/** The route's request budget; no request is sent larger than this. */
|
|
973
|
+
maxBytes: number;
|
|
544
974
|
model: string;
|
|
545
|
-
|
|
975
|
+
/** Sends one request; absent when Jev is out of reach (`unreachable` says why). */
|
|
976
|
+
ask?: JevFindAsk;
|
|
977
|
+
unreachable?: JevAbstainReason;
|
|
978
|
+
/** The TypeScript loader; tests pass one resolving undefined to force chunking. */
|
|
979
|
+
loadTypeScript?: typeof loadTypeScript;
|
|
980
|
+
signal?: AbortSignal;
|
|
981
|
+
}
|
|
982
|
+
|
|
983
|
+
export interface JevFindResult {
|
|
984
|
+
text: string;
|
|
985
|
+
/** Shape only, like ask_jev: details persist in the session. */
|
|
986
|
+
details: Record<string, unknown>;
|
|
987
|
+
judged: number;
|
|
988
|
+
topRelevance?: number;
|
|
546
989
|
failure?: string;
|
|
547
|
-
|
|
548
|
-
|
|
549
|
-
|
|
550
|
-
|
|
551
|
-
|
|
552
|
-
|
|
553
|
-
|
|
554
|
-
|
|
555
|
-
|
|
990
|
+
elapsedMs: number;
|
|
991
|
+
}
|
|
992
|
+
|
|
993
|
+
/**
|
|
994
|
+
* The whole jev_find call: gather files with ripgrep, narrow large sets by directory, judge files,
|
|
995
|
+
* then judge the source units of the relevant ones and return them verbatim. Without a reachable
|
|
996
|
+
* Jev (or when no file could be judged) the ripgrep-ranked list still comes back, with keyword
|
|
997
|
+
* windows of the top files, so the call is never wasted. Throws only when ripgrep fails.
|
|
998
|
+
*/
|
|
999
|
+
export async function runJevFind(options: JevFindOptions): Promise<JevFindResult> {
|
|
1000
|
+
const files = await gatherFindFiles(options, options.signal);
|
|
1001
|
+
if (files.length === 0) {
|
|
1002
|
+
return {
|
|
1003
|
+
text: "jev_find: no candidate files. Loosen `pattern`, `glob`, or `path`.",
|
|
1004
|
+
details: { total: 0, judged: 0 },
|
|
1005
|
+
judged: 0,
|
|
1006
|
+
elapsedMs: 0,
|
|
1007
|
+
};
|
|
1008
|
+
}
|
|
1009
|
+
const words = questionWords(options.question);
|
|
1010
|
+
const limit = Math.max(1, Math.floor(options.limit ?? FIND_DEFAULT_LIMIT));
|
|
1011
|
+
const texts = new Map<string, string | undefined>();
|
|
1012
|
+
const readText = (path: string): string | undefined => {
|
|
1013
|
+
if (!texts.has(path)) {
|
|
1014
|
+
const file = readBounded(resolve(options.cwd, path), FIND_SCAN_BYTES);
|
|
1015
|
+
texts.set(path, "error" in file ? undefined : file.text);
|
|
1016
|
+
}
|
|
1017
|
+
return texts.get(path);
|
|
1018
|
+
};
|
|
1019
|
+
const hitsOf = (file: FindFile): { lines: number[]; snippet: string[] } =>
|
|
1020
|
+
file.lines.length > 0 ? { lines: file.lines, snippet: file.hits } : keywordLines(readText(file.path), words);
|
|
1021
|
+
|
|
1022
|
+
/** The ripgrep-ranked list and source units of the top files: what comes back when Jev did not rank. */
|
|
1023
|
+
const fallback = async (
|
|
1024
|
+
reason: string,
|
|
1025
|
+
notes: string[],
|
|
1026
|
+
elapsedMs: number,
|
|
1027
|
+
requests?: FindRequests,
|
|
1028
|
+
): Promise<JevFindResult> => {
|
|
1029
|
+
const shown = files.slice(0, limit);
|
|
1030
|
+
const sourceFiles = shown.slice(0, FIND_FALLBACK_SOURCE_FILES);
|
|
1031
|
+
const ts = sourceFiles.some((file) => isScriptPath(file.path)) ? await (options.loadTypeScript ?? loadTypeScript)() : undefined;
|
|
1032
|
+
const blocks: SourceBlock[] = [];
|
|
1033
|
+
const leads = new Map<string, string>();
|
|
1034
|
+
for (const file of sourceFiles) {
|
|
1035
|
+
const text = readText(file.path);
|
|
1036
|
+
if (text === undefined) continue;
|
|
1037
|
+
const rows = fileRows(text);
|
|
1038
|
+
const hits = hitsOf(file).lines;
|
|
1039
|
+
const units = fallbackSourceUnits(file.path, text, ts, rows, hits, words);
|
|
1040
|
+
const selected = units.length > 0 ? units : keywordWindows(rows.length, hits);
|
|
1041
|
+
leads.set(file.path, selected.map((unit) => `${unit.name} lines ${unit.start}-${unit.end}`).join(", "));
|
|
1042
|
+
for (const unit of selected) blocks.push({ path: file.path, unit, view: unitView(unit, focusOf(unit, hits)), rows });
|
|
1043
|
+
}
|
|
1044
|
+
const source = renderSourceBlocks(blocks);
|
|
1045
|
+
const diagnostic = requests?.diagnostics[0];
|
|
1046
|
+
const lines = [
|
|
1047
|
+
`jev_find: Jev did not judge (${reason}); ${files.length} candidate file(s), ranked by ripgrep:`,
|
|
1048
|
+
...notes,
|
|
1049
|
+
...(diagnostic ? [`Diagnostic: ${formatJevDiagnostic(diagnostic)}`] : []),
|
|
1050
|
+
...shown.map((file) => `- ${file.path}${leads.has(file.path) ? ` — ${leads.get(file.path)}` : ""}`),
|
|
1051
|
+
...(files.length > shown.length ? [`${files.length - shown.length} more candidate file(s) not listed.`] : []),
|
|
1052
|
+
...source.lines,
|
|
1053
|
+
"",
|
|
1054
|
+
"End context.",
|
|
1055
|
+
];
|
|
1056
|
+
return {
|
|
1057
|
+
text: lines.join("\n"),
|
|
1058
|
+
details: {
|
|
1059
|
+
total: files.length,
|
|
1060
|
+
judged: 0,
|
|
1061
|
+
requests: requests?.sent ?? 0,
|
|
1062
|
+
sourceBytes: source.bytes,
|
|
1063
|
+
elapsedMs,
|
|
1064
|
+
failure: reason,
|
|
1065
|
+
...(diagnostic ? { diagnostic } : {}),
|
|
1066
|
+
},
|
|
1067
|
+
judged: 0,
|
|
1068
|
+
failure: reason,
|
|
1069
|
+
...(diagnostic ? { diagnostic } : {}),
|
|
1070
|
+
elapsedMs,
|
|
1071
|
+
};
|
|
1072
|
+
};
|
|
1073
|
+
|
|
1074
|
+
if (!options.ask) return fallback(options.unreachable ?? "missing", [], 0);
|
|
1075
|
+
|
|
1076
|
+
const started = Date.now();
|
|
1077
|
+
const requests = new FindRequests(options.ask);
|
|
1078
|
+
const request = clip(options.question, FIND_REQUEST_BYTES).text;
|
|
1079
|
+
const notes: string[] = [];
|
|
1080
|
+
|
|
1081
|
+
// 1. Directories.
|
|
1082
|
+
const descent = await narrowByDirectory(files, options.root, {
|
|
1083
|
+
request,
|
|
1084
|
+
maxBytes: options.maxBytes,
|
|
1085
|
+
requests,
|
|
1086
|
+
hitLines: (file) => hitsOf(file).snippet,
|
|
1087
|
+
});
|
|
1088
|
+
if (descent.prunedDirs > 0) {
|
|
1089
|
+
notes.push(`Jev ruled out ${descent.prunedDirs} director${descent.prunedDirs === 1 ? "y" : "ies"} holding ${descent.prunedFiles} file(s).`);
|
|
1090
|
+
}
|
|
1091
|
+
if (descent.unjudgedDirs > 0) {
|
|
1092
|
+
notes.push(
|
|
1093
|
+
`${descent.unjudgedDirs} director${descent.unjudgedDirs === 1 ? "y was" : "ies were"} kept unjudged (${descent.reasons.join(", ") || "missing"}).`,
|
|
556
1094
|
);
|
|
557
|
-
}
|
|
558
|
-
|
|
559
|
-
|
|
560
|
-
|
|
561
|
-
|
|
562
|
-
|
|
563
|
-
|
|
1095
|
+
}
|
|
1096
|
+
|
|
1097
|
+
// 2. Files, best score first, within the judging cap and the requests left after the unit reserve.
|
|
1098
|
+
const fileSendCap = requests.left - FIND_UNIT_REQUESTS;
|
|
1099
|
+
const judgeCap = Math.min(FIND_MAX_JUDGED_FILES, Math.max(0, fileSendCap) * FIND_BATCH);
|
|
1100
|
+
const ordered = [...descent.kept].sort(byScore);
|
|
1101
|
+
const pastCap = Math.max(0, ordered.length - judgeCap);
|
|
1102
|
+
const snippetBytes = Math.max(256, Math.floor((options.maxBytes - FIND_QUESTION_RESERVE) / FIND_BATCH));
|
|
1103
|
+
const candidates: FindCandidate[] = [];
|
|
1104
|
+
for (const file of ordered.slice(0, judgeCap)) {
|
|
1105
|
+
if (file.lines.length > 0) {
|
|
1106
|
+
candidates.push({ path: file.path, lines: file.lines, snippet: clip(file.hits.join("\n"), snippetBytes).text });
|
|
1107
|
+
continue;
|
|
564
1108
|
}
|
|
565
|
-
|
|
1109
|
+
const text = readText(file.path);
|
|
1110
|
+
if (text !== undefined) candidates.push({ path: file.path, ...excerpt(text, words, snippetBytes) });
|
|
566
1111
|
}
|
|
567
|
-
|
|
568
|
-
|
|
1112
|
+
const filesOutcome = await judgeNouls(
|
|
1113
|
+
requests,
|
|
1114
|
+
{ request, stateKey: "files", criteria: FILE_CRITERIA },
|
|
1115
|
+
candidates.map((candidate) => ({
|
|
1116
|
+
key: candidate.path,
|
|
1117
|
+
text: candidate.snippet,
|
|
1118
|
+
question: `Is the file "${candidate.path}" relevant to the request, judged by its excerpt in state.files?`,
|
|
1119
|
+
})),
|
|
1120
|
+
options.maxBytes,
|
|
1121
|
+
// A retry may borrow one request from the unit reserve: without file verdicts there is nothing to excerpt.
|
|
1122
|
+
{ send: fileSendCap, retry: requests.left - 1 },
|
|
1123
|
+
);
|
|
1124
|
+
const scored = candidates.map((candidate, index) => ({ ...candidate, relevance: filesOutcome.scores[index] }));
|
|
1125
|
+
const judged = scored.filter((candidate) => candidate.relevance !== undefined).length;
|
|
1126
|
+
const unjudgedFiles = candidates.length - judged;
|
|
1127
|
+
if (pastCap > 0) {
|
|
1128
|
+
notes.push(`${pastCap} file(s) past the ${judgeCap}-file judging cap were not judged; narrow with \`pattern\`, \`glob\`, or \`path\`.`);
|
|
1129
|
+
}
|
|
1130
|
+
if (judged === 0) {
|
|
1131
|
+
return fallback(filesOutcome.reasons[0] ?? "missing", notes, Date.now() - started, requests);
|
|
1132
|
+
}
|
|
1133
|
+
if (unjudgedFiles > 0) notes.push(`${unjudgedFiles} file(s) were not judged (${filesOutcome.reasons.join(", ") || "missing"}).`);
|
|
1134
|
+
|
|
1135
|
+
const ranked = scored
|
|
1136
|
+
.filter((candidate): candidate is FindCandidate & { relevance: number } => candidate.relevance !== undefined)
|
|
1137
|
+
.sort((a, b) => b.relevance - a.relevance);
|
|
1138
|
+
const relevant = ranked.filter((candidate) => candidate.relevance >= FIND_MIN_RELEVANCE);
|
|
1139
|
+
const shown = relevant.slice(0, limit);
|
|
1140
|
+
|
|
1141
|
+
// 3. Source units of the relevant files, in relevance order, within the requests left.
|
|
1142
|
+
const ts = shown.some((candidate) => isScriptPath(candidate.path))
|
|
1143
|
+
? await (options.loadTypeScript ?? loadTypeScript)()
|
|
1144
|
+
: undefined;
|
|
1145
|
+
const perFile = shown.map((candidate) => {
|
|
1146
|
+
const text = readText(candidate.path) ?? "";
|
|
1147
|
+
const rows = fileRows(text);
|
|
1148
|
+
const hits = [...new Set([...candidate.lines, ...keywordLines(text, words).lines])];
|
|
1149
|
+
const picks = text === "" ? [] : pickUnits(splitSourceUnits(candidate.path, text, ts), rows, hits, words);
|
|
1150
|
+
return { candidate, rows, hits, picks };
|
|
1151
|
+
});
|
|
1152
|
+
const unitItems: JudgeItem[] = [];
|
|
1153
|
+
const owners: Array<{ file: number; unit: SourceUnit }> = [];
|
|
1154
|
+
const unitCapacity = requests.left * FIND_BATCH;
|
|
1155
|
+
perFile.forEach((entry, file) => {
|
|
1156
|
+
for (const unit of entry.picks) {
|
|
1157
|
+
if (unitItems.length >= unitCapacity) return;
|
|
1158
|
+
const key = `${entry.candidate.path} lines ${unit.start}-${unit.end} (${unit.name})`;
|
|
1159
|
+
unitItems.push({
|
|
1160
|
+
key,
|
|
1161
|
+
text: unitView(unit, focusOf(unit, entry.hits))
|
|
1162
|
+
.map((line) => (line === 0 ? "…" : (entry.rows[line - 1] ?? "")))
|
|
1163
|
+
.join("\n"),
|
|
1164
|
+
question: `Does the source unit "${key}" in state.units implement, define, configure, or directly answer the request?`,
|
|
1165
|
+
});
|
|
1166
|
+
owners.push({ file, unit });
|
|
1167
|
+
}
|
|
1168
|
+
});
|
|
1169
|
+
const unitsOutcome = await judgeNouls(
|
|
1170
|
+
requests,
|
|
1171
|
+
{ request, stateKey: "units", criteria: UNIT_CRITERIA },
|
|
1172
|
+
unitItems,
|
|
1173
|
+
options.maxBytes,
|
|
1174
|
+
{ send: requests.left, retry: requests.left },
|
|
1175
|
+
);
|
|
1176
|
+
const elapsedMs = Date.now() - started;
|
|
1177
|
+
|
|
1178
|
+
const blocks: SourceBlock[] = [];
|
|
1179
|
+
const listing: string[] = [];
|
|
1180
|
+
let windowed = 0;
|
|
1181
|
+
perFile.forEach((entry, file) => {
|
|
1182
|
+
const judgedUnits = owners
|
|
1183
|
+
.map((owner, index) => ({ ...owner, score: unitsOutcome.scores[index] }))
|
|
1184
|
+
.filter((owner) => owner.file === file);
|
|
1185
|
+
const answered = judgedUnits.some((owner) => owner.score !== undefined);
|
|
1186
|
+
const fallbackUnits = fallbackSourceUnits(
|
|
1187
|
+
entry.candidate.path,
|
|
1188
|
+
readText(entry.candidate.path) ?? "",
|
|
1189
|
+
ts,
|
|
1190
|
+
entry.rows,
|
|
1191
|
+
entry.hits,
|
|
1192
|
+
words,
|
|
1193
|
+
);
|
|
1194
|
+
const units =
|
|
1195
|
+
answered || entry.rows.length === 0 || entry.picks.length === 0
|
|
1196
|
+
? judgedUnits.filter((owner) => (owner.score ?? 0) >= FIND_MIN_RELEVANCE).map((owner) => owner.unit)
|
|
1197
|
+
: fallbackUnits.length > 0
|
|
1198
|
+
? fallbackUnits
|
|
1199
|
+
: keywordWindows(entry.rows.length, entry.hits);
|
|
1200
|
+
if (!answered && units.length > 0) windowed += 1;
|
|
1201
|
+
for (const unit of units) {
|
|
1202
|
+
blocks.push({ path: entry.candidate.path, unit, view: unitView(unit, focusOf(unit, entry.hits)), rows: entry.rows });
|
|
1203
|
+
}
|
|
1204
|
+
const leads = units.map((unit) => `${unit.name} lines ${unit.start}-${unit.end}`).join(", ");
|
|
1205
|
+
listing.push(`- ${entry.candidate.path} (${entry.candidate.relevance.toFixed(2)})${leads ? ` — reading leads: ${leads}` : ""}`);
|
|
1206
|
+
});
|
|
1207
|
+
if (windowed > 0) {
|
|
1208
|
+
notes.push(
|
|
1209
|
+
`Jev did not judge the source units of ${windowed} file(s) (${unitsOutcome.reasons.join(", ") || "missing"}); matching source units or keyword windows are shown instead.`,
|
|
1210
|
+
);
|
|
1211
|
+
}
|
|
1212
|
+
if (relevant.length > shown.length) {
|
|
1213
|
+
const rest = relevant.slice(shown.length);
|
|
1214
|
+
notes.push(
|
|
1215
|
+
`${rest.length} more relevant file(s) past the limit: ${rest
|
|
1216
|
+
.slice(0, 5)
|
|
1217
|
+
.map((candidate) => `${candidate.path} (${candidate.relevance.toFixed(2)})`)
|
|
1218
|
+
.join(", ")}${rest.length > 5 ? ", …" : ""}.`,
|
|
1219
|
+
);
|
|
1220
|
+
}
|
|
1221
|
+
const source = renderSourceBlocks(blocks);
|
|
1222
|
+
const diagnostic = requests.diagnostics[0];
|
|
1223
|
+
if (diagnostic) notes.push(`Diagnostic: ${formatJevDiagnostic(diagnostic)}`);
|
|
1224
|
+
if (source.omitted > 0) {
|
|
1225
|
+
notes.push(`${source.omitted} source block(s) left out to stay within ${FIND_MAX_SOURCE_BYTES / 1024} KB; read them from the leads.`);
|
|
569
1226
|
}
|
|
570
|
-
|
|
571
|
-
|
|
1227
|
+
|
|
1228
|
+
const lines = [
|
|
1229
|
+
`jev_find: ${relevant.length} relevant file(s) (judged ${judged} of ${files.length} candidates via ${options.model} in ${elapsedMs}ms)`,
|
|
1230
|
+
...notes,
|
|
1231
|
+
...(shown.length > 0
|
|
1232
|
+
? listing
|
|
1233
|
+
: [
|
|
1234
|
+
"No file is a confident match; the closest were:",
|
|
1235
|
+
...ranked.slice(0, 3).map((candidate) => `- ${candidate.path} (${candidate.relevance.toFixed(2)})`),
|
|
1236
|
+
]),
|
|
1237
|
+
...source.lines,
|
|
1238
|
+
"",
|
|
1239
|
+
"End context.",
|
|
1240
|
+
];
|
|
1241
|
+
const failure = filesOutcome.reasons.find((reason) => reason !== "request cap" && reason !== "too large");
|
|
1242
|
+
return {
|
|
1243
|
+
text: lines.join("\n"),
|
|
1244
|
+
details: {
|
|
1245
|
+
total: files.length,
|
|
1246
|
+
judged,
|
|
1247
|
+
relevant: relevant.length,
|
|
1248
|
+
requests: requests.sent,
|
|
1249
|
+
retried: requests.retried,
|
|
1250
|
+
sourceBytes: source.bytes,
|
|
1251
|
+
elapsedMs,
|
|
1252
|
+
...(failure ? { failure } : {}),
|
|
1253
|
+
...(diagnostic ? { diagnostic } : {}),
|
|
1254
|
+
},
|
|
1255
|
+
judged,
|
|
1256
|
+
topRelevance: ranked[0]?.relevance,
|
|
1257
|
+
...(failure ? { failure } : {}),
|
|
1258
|
+
...(diagnostic ? { diagnostic } : {}),
|
|
1259
|
+
elapsedMs,
|
|
1260
|
+
};
|
|
572
1261
|
}
|
|
573
1262
|
|
|
574
1263
|
const JevFindParams = Type.Object({
|
|
575
|
-
question: Type.String({
|
|
1264
|
+
question: Type.String({
|
|
1265
|
+
description: "What you want to understand, in plain words (e.g. 'how is the retry backoff configured').",
|
|
1266
|
+
}),
|
|
576
1267
|
pattern: Type.Optional(
|
|
577
|
-
Type.String({ description: "ripgrep regex to narrow candidates to files that match; omit to
|
|
1268
|
+
Type.String({ description: "ripgrep regex to narrow candidates to files that match; omit to consider every listed file." }),
|
|
578
1269
|
),
|
|
579
1270
|
glob: Type.Optional(Type.String({ description: "File glob filter, e.g. '*.ts' or 'src/**/*.md'." })),
|
|
580
1271
|
path: Type.Optional(Type.String({ description: "Directory to search, inside the working directory (default: '.')." })),
|
|
581
1272
|
ignoreCase: Type.Optional(Type.Boolean({ description: "Case-insensitive pattern." })),
|
|
582
|
-
limit: Type.Optional(
|
|
1273
|
+
limit: Type.Optional(
|
|
1274
|
+
Type.Number({ description: `Most relevant files to return, with their source (default ${FIND_DEFAULT_LIMIT}).` }),
|
|
1275
|
+
),
|
|
583
1276
|
});
|
|
584
1277
|
|
|
585
1278
|
// ---------------------------------------------------------------------------
|
|
@@ -809,19 +1502,23 @@ export default function jevAskToolExtension(pi: ExtensionAPI): void {
|
|
|
809
1502
|
name: JEV_FIND_TOOL,
|
|
810
1503
|
label: "Jev Find",
|
|
811
1504
|
description: [
|
|
812
|
-
"Find
|
|
813
|
-
"
|
|
814
|
-
"`question
|
|
815
|
-
|
|
1505
|
+
"Find where behavior lives and read it in one call. ripgrep gathers candidate files (those matching `pattern`,",
|
|
1506
|
+
"or every file under `path` filtered by `glob`); Jev narrows a large tree directory by directory, judges each",
|
|
1507
|
+
"remaining file's excerpt against `question`, then judges the functions, classes, and sections of the relevant",
|
|
1508
|
+
"files. Returns the relevant files with reading leads AND the accepted source verbatim with original line",
|
|
1509
|
+
`numbers (about ${FIND_MAX_SOURCE_BYTES / 1024} KB of source, at most ${FIND_MAX_REQUESTS} Jev requests per call).`,
|
|
1510
|
+
"Respects .gitignore; secret-named files are never sent.",
|
|
816
1511
|
].join(" "),
|
|
817
|
-
promptSnippet: "Find
|
|
1512
|
+
promptSnippet: "Find where behavior lives: Jev-judged relevant files plus their verbatim source with line numbers",
|
|
818
1513
|
promptGuidelines: [
|
|
819
|
-
"
|
|
820
|
-
"
|
|
1514
|
+
"Use jev_find for how/why/where-does-this-behavior-live questions — even when the question names a function or setting — before grep, find, ls, or reading candidate files. One call returns the relevant files and their relevant source verbatim with line numbers. Pass `question` in plain words; add `pattern` when you know a likely identifier, `glob`/`path` to scope it.",
|
|
1515
|
+
"Read the source blocks jev_find returns before searching again; `read` only the leads you still need (to edit, or past a clipped block).",
|
|
1516
|
+
"Use grep or read instead for an exact string, a known symbol's definition, or a known filename.",
|
|
1517
|
+
"When delegating repository discovery to a subagent, tell it to start with jev_find.",
|
|
821
1518
|
],
|
|
822
1519
|
discovery: {
|
|
823
|
-
summary: "Semantic
|
|
824
|
-
aliases: ["jevgrep", "find", "search", "locate"],
|
|
1520
|
+
summary: "Semantic code finder: Jev-judged relevant files and their verbatim source excerpts",
|
|
1521
|
+
aliases: ["jevgrep", "jg", "find", "search", "locate"],
|
|
825
1522
|
category: "Decisions",
|
|
826
1523
|
},
|
|
827
1524
|
parameters: JevFindParams,
|
|
@@ -848,79 +1545,41 @@ export default function jevAskToolExtension(pi: ExtensionAPI): void {
|
|
|
848
1545
|
const rg = await ensureTool("rg");
|
|
849
1546
|
if (!rg) return text("jev_find needs ripgrep (rg), which is not available; use grep instead.");
|
|
850
1547
|
|
|
851
|
-
|
|
852
|
-
|
|
1548
|
+
// Without Jev the ripgrep-ranked files and their keyword lines still come back: the call is never wasted.
|
|
1549
|
+
const connection = jevConnection(config, route);
|
|
1550
|
+
const unreachable = await jevUnavailable(ctx, connection);
|
|
1551
|
+
let result: JevFindResult;
|
|
853
1552
|
try {
|
|
854
|
-
|
|
855
|
-
|
|
856
|
-
|
|
857
|
-
|
|
858
|
-
|
|
859
|
-
|
|
860
|
-
|
|
861
|
-
|
|
862
|
-
|
|
863
|
-
|
|
864
|
-
|
|
1553
|
+
result = await runJevFind({
|
|
1554
|
+
rg,
|
|
1555
|
+
cwd: ctx.cwd,
|
|
1556
|
+
root: inside === "" ? "." : inside,
|
|
1557
|
+
question,
|
|
1558
|
+
pattern: params.pattern || undefined,
|
|
1559
|
+
glob: params.glob || undefined,
|
|
1560
|
+
ignoreCase: params.ignoreCase,
|
|
1561
|
+
limit: params.limit,
|
|
1562
|
+
maxBytes: route.payloadBytes,
|
|
1563
|
+
model: config.model,
|
|
1564
|
+
...(unreachable
|
|
1565
|
+
? { unreachable }
|
|
1566
|
+
: { ask: (payload) => askJevAnswers(ctx, connection, { payload, maxBytes: route.payloadBytes }) }),
|
|
865
1567
|
signal,
|
|
866
|
-
);
|
|
1568
|
+
});
|
|
867
1569
|
} catch (error) {
|
|
868
1570
|
return text(`jev_find: ripgrep failed (${error instanceof Error ? error.message.split("\n")[0] : String(error)}).`);
|
|
869
1571
|
}
|
|
870
|
-
|
|
871
|
-
|
|
872
|
-
|
|
873
|
-
|
|
874
|
-
|
|
875
|
-
|
|
876
|
-
|
|
877
|
-
|
|
878
|
-
const unreachable = await jevUnavailable(ctx, connection);
|
|
879
|
-
const batches: FindCandidate[][] = [];
|
|
880
|
-
for (let start = 0; start < candidates.length; start += FIND_BATCH) {
|
|
881
|
-
batches.push(candidates.slice(start, start + FIND_BATCH));
|
|
882
|
-
}
|
|
883
|
-
const started = Date.now();
|
|
884
|
-
const results = unreachable
|
|
885
|
-
? []
|
|
886
|
-
: await Promise.all(
|
|
887
|
-
batches.map((batch) =>
|
|
888
|
-
askJevAnswers(ctx, connection, {
|
|
889
|
-
payload: fitFindPayload(question, batch, route.payloadBytes),
|
|
890
|
-
maxBytes: route.payloadBytes,
|
|
891
|
-
}),
|
|
892
|
-
),
|
|
893
|
-
);
|
|
894
|
-
const elapsedMs = Date.now() - started;
|
|
895
|
-
|
|
896
|
-
const scored: Array<FindCandidate & { relevance?: number }> = [];
|
|
897
|
-
let judged = 0;
|
|
898
|
-
batches.forEach((batch, b) => {
|
|
899
|
-
const answers = results[b]?.answers ?? {};
|
|
900
|
-
batch.forEach((candidate, i) => {
|
|
901
|
-
const answer = answers[`f${i}`];
|
|
902
|
-
const relevance = isRecord(answer) && typeof answer.noul === "number" ? answer.noul : undefined;
|
|
903
|
-
if (relevance !== undefined) judged += 1;
|
|
904
|
-
scored.push({ ...candidate, ...(relevance === undefined ? {} : { relevance }) });
|
|
1572
|
+
if (result.details.total !== 0) {
|
|
1573
|
+
emitJevTelemetry(pi.events, "decision", {
|
|
1574
|
+
route: "find",
|
|
1575
|
+
outcome: result.judged > 0 ? "jev" : "fallback",
|
|
1576
|
+
candidates: result.details.total,
|
|
1577
|
+
confidence: confidenceBucket(result.topRelevance),
|
|
1578
|
+
elapsedMs: result.elapsedMs,
|
|
1579
|
+
...(result.judged > 0 ? {} : { reason: result.failure ?? "missing" }),
|
|
905
1580
|
});
|
|
906
|
-
}
|
|
907
|
-
|
|
908
|
-
const ranked = judged > 0 ? [...scored].sort((a, b) => (b.relevance ?? -1) - (a.relevance ?? -1)) : scored;
|
|
909
|
-
const failure = unreachable ?? results.find((result) => result.failure)?.failure;
|
|
910
|
-
|
|
911
|
-
emitJevTelemetry(pi.events, "decision", {
|
|
912
|
-
route: "find",
|
|
913
|
-
outcome: judged > 0 ? "jev" : "fallback",
|
|
914
|
-
candidates: candidates.length,
|
|
915
|
-
confidence: confidenceBucket(ranked[0]?.relevance),
|
|
916
|
-
elapsedMs,
|
|
917
|
-
...(judged > 0 ? {} : { reason: failure ?? "missing" }),
|
|
918
|
-
});
|
|
919
|
-
return text(
|
|
920
|
-
renderFindResults({ ranked, total, judged, limit, model: config.model, elapsedMs, failure }),
|
|
921
|
-
// Shape only, like ask_jev: details persist in the session.
|
|
922
|
-
{ candidates: candidates.length, total, judged, elapsedMs, ...(failure ? { failure } : {}) },
|
|
923
|
-
);
|
|
1581
|
+
}
|
|
1582
|
+
return text(result.text, result.details);
|
|
924
1583
|
},
|
|
925
1584
|
});
|
|
926
1585
|
|