@a-t-h-i/bot-lobby 0.6.6 → 0.6.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (58) hide show
  1. package/README.md +183 -11
  2. package/package.json +1 -1
  3. package/prompts/master.md +21 -0
  4. package/prompts/panel.md +4 -0
  5. package/prompts/planner.md +5 -0
  6. package/prompts/pr-review.md +62 -0
  7. package/prompts/splitter.md +58 -0
  8. package/src/ask/dialog.ts +23 -7
  9. package/src/ask/state.ts +18 -3
  10. package/src/ask/tool.ts +9 -2
  11. package/src/ask/view.ts +8 -4
  12. package/src/classifier/instance.ts +6 -0
  13. package/src/classifier/knowledge.ts +158 -0
  14. package/src/classifier/review.ts +146 -0
  15. package/src/execution/agent-runner.ts +23 -1
  16. package/src/execution/fallback.ts +75 -0
  17. package/src/execution/workspace.ts +185 -0
  18. package/src/knowledge/edit.ts +109 -0
  19. package/src/knowledge/notes.ts +143 -0
  20. package/src/knowledge/selector.ts +1 -1
  21. package/src/knowledge/store.ts +16 -2
  22. package/src/lobby/ask.ts +42 -21
  23. package/src/lobby/feed.ts +7 -0
  24. package/src/lobby/issues.ts +1 -1
  25. package/src/lobby/knowledge.ts +179 -0
  26. package/src/lobby/layout.ts +99 -1
  27. package/src/lobby/mini.ts +159 -0
  28. package/src/lobby/planner.ts +209 -10
  29. package/src/lobby/pr-review.ts +360 -0
  30. package/src/lobby/pulls.ts +250 -0
  31. package/src/lobby/quickfix.ts +17 -3
  32. package/src/lobby/runtime.ts +142 -17
  33. package/src/lobby/split.ts +230 -0
  34. package/src/lobby/tabs/git.ts +162 -0
  35. package/src/lobby/tabs/knowledge.ts +135 -0
  36. package/src/lobby/tabs/plan.ts +3 -1
  37. package/src/lobby/tabs/tasks.ts +11 -2
  38. package/src/lobby/view.ts +450 -35
  39. package/src/master/master.ts +40 -14
  40. package/src/master/research.ts +1 -0
  41. package/src/pi/commands.ts +12 -7
  42. package/src/pi/events.ts +2 -0
  43. package/src/pi/master-fallback.ts +61 -0
  44. package/src/pi/model-support.ts +27 -3
  45. package/src/pi/plan-checklist.ts +42 -0
  46. package/src/pi/route.ts +2 -1
  47. package/src/pi/settings-ui.ts +104 -11
  48. package/src/pi/start-flags.ts +14 -0
  49. package/src/pi/start-task.ts +56 -5
  50. package/src/pi/tools.ts +4 -2
  51. package/src/schemas/configuration.ts +74 -11
  52. package/src/schemas/findings.ts +2 -0
  53. package/src/schemas/task.ts +15 -0
  54. package/src/state/backlog.ts +27 -4
  55. package/src/state/metrics.ts +2 -2
  56. package/src/state/persistence.ts +5 -5
  57. package/src/text.ts +38 -0
  58. package/src/workflow/workflow.ts +23 -2
package/src/ask/tool.ts CHANGED
@@ -65,23 +65,30 @@ export function invalidQuestions(questions: readonly AskQuestion[]): string | un
65
65
  return undefined;
66
66
  }
67
67
 
68
+ /** What the model does with questions the user left unanswered: wait, never fill them in. */
69
+ const STILL_OPEN = "Do not assume answers, do not pick the recommended options, and do not go on with work that depends on them. Say in one short line that the questions are waiting, end your turn, and put them to the user again when they next write.";
70
+ const NOT_ANSWERED = `The user left the questions without answering. ${STILL_OPEN}`;
71
+
68
72
  /** What the model reads back: each question with its answer, or that it was skipped. */
69
73
  export function answerSummary(questions: readonly AskQuestion[], result: AskResult): string {
70
74
  if (result.cancelled && result.answers.length === 0) {
71
75
  // A relay that could not ask anyone (auto mode) says why.
72
76
  if (result.globalNote) return result.globalNote;
73
- return "The user put the questions away without answering. Do not ask the same again right away: go on with your best judgement and say what you assumed, or ask something narrower.";
77
+ return NOT_ANSWERED;
74
78
  }
75
79
  const lines = questions.map((question, index) => {
76
80
  const answer = result.answers.find((entry) => entry.questionIndex === index);
77
81
  const head = `${index + 1}. [${question.header}] ${question.question.replace(/\s+/g, " ").trim()}`;
78
82
  if (!answer) return `${head}\n → (not answered)`;
79
83
  const text = answer.kind === "multi" ? (answer.selected ?? []).join(", ") : answer.answer ?? "";
80
- return `${head}\n → ${text}${answer.kind === "custom" ? " (in the user's own words)" : ""}${answer.notes ? `\n note: ${answer.notes}` : ""}`;
84
+ // A several-line answer keeps its lines under the arrow.
85
+ const lines = (value: string) => value.replace(/\n/g, "\n ");
86
+ return `${head}\n → ${lines(text)}${answer.kind === "custom" ? " (in the user's own words)" : ""}${answer.notes ? `\n note: ${lines(answer.notes)}` : ""}`;
81
87
  });
82
88
  return [
83
89
  result.cancelled ? "The user answered some questions, then put the rest away:" : "The user answered:",
84
90
  ...lines,
91
+ ...(result.cancelled ? ["", `The questions marked "(not answered)" are still open. ${STILL_OPEN}`] : []),
85
92
  ...(result.globalNote ? ["", `Note: ${result.globalNote}`] : []),
86
93
  ].join("\n");
87
94
  }
package/src/ask/view.ts CHANGED
@@ -65,10 +65,13 @@ function optionLines(state: AskState, question: AskQuestion, width: number, them
65
65
  const lead = `${pointer} ${paint(theme, own ? "accent" : "dim", "✎")} `;
66
66
  const room = Math.max(1, width - textWidth(lead));
67
67
  if (state.editing) {
68
- const draft = wrap(`${state.draft}▏`, room);
68
+ // Each line of the answer wraps on its own, so Shift+Enter shows as a new row.
69
+ const draft = `${state.draft}▏`.split("\n").flatMap((line) => (line ? wrap(line, room) : [""]));
69
70
  lines.push(`${lead}${paint(theme, "accent", draft[0] ?? "")}`, ...draft.slice(1).map((line) => `${" ".repeat(textWidth(lead))}${paint(theme, "accent", line)}`));
70
71
  } else {
71
- lines.push(`${lead}${own ? paint(theme, "accent", `“${own}”`) : paint(theme, ownFocused ? "text" : "dim", OWN_ANSWER)}`);
72
+ // A several-line answer shows on one row, its line breaks marked.
73
+ const shown = own?.replace(/\s*\n\s*/g, " ⏎ ");
74
+ lines.push(`${lead}${shown ? paint(theme, "accent", `“${shown}”`) : paint(theme, ownFocused ? "text" : "dim", OWN_ANSWER)}`);
72
75
  }
73
76
  return lines;
74
77
  }
@@ -112,14 +115,15 @@ function previewBox(preview: Preview, width: number, rows: number, theme: LobbyT
112
115
  }
113
116
 
114
117
  function hints(state: AskState, question: AskQuestion, width: number, theme?: LobbyTheme): string[] {
118
+ if (state.leaving) return wrap(paint(theme, "warning", "Leave without answering? The oracle will not guess for you. enter leaves · any other key keeps answering"), width);
115
119
  const parts = state.editing
116
- ? ["enter keep it", "esc back to the options"]
120
+ ? ["enter keep it", "shift+enter new line", "esc back to the options"]
117
121
  : [
118
122
  "↑↓ move",
119
123
  question.multiSelect ? "space pick · enter next" : "enter choose",
120
124
  `1-${question.options.length} pick`,
121
125
  ...(state.questions.length > 1 ? ["←→ questions"] : []),
122
- "esc put away",
126
+ "esc leave",
123
127
  ];
124
128
  return wrap(paint(theme, "dim", parts.join(" · ")), width);
125
129
  }
@@ -11,6 +11,7 @@ import { isSubagentProcess } from "../pi/quiet.ts";
11
11
  import { Classifier } from "./classifier.ts";
12
12
  import { registerJevProvider, type KeySource, type KeyStatus } from "./hosts.ts";
13
13
  import { fileHinter, type FileHinter, type FileScope } from "./files.ts";
14
+ import { knowledgePicker, type KnowledgePicker } from "./knowledge.ts";
14
15
  import { lobbyFeed } from "../lobby/feed.ts";
15
16
  import { triageLine, triageWithContext } from "./triage.ts";
16
17
  import type { TaskTriage } from "../schemas/task.ts";
@@ -45,6 +46,11 @@ export function hintsFor(scope: FileScope): FileHinter {
45
46
  return fileHinter(classifier(), scope, (text) => lobbyFeed.log("CLASSIFIER", text, "info"));
46
47
  }
47
48
 
49
+ /** Relevant knowledge for agents, with what Jev kept of each long file logged to the lobby's activity feed. */
50
+ export function knowledgeFor(): KnowledgePicker {
51
+ return knowledgePicker(classifier(), (text) => lobbyFeed.log("CLASSIFIER", text, "info"));
52
+ }
53
+
48
54
  /** Triage a request for a task in a tree, logged to the lobby's activity feed; undefined when triage is off or fails. */
49
55
  export async function triageFor(scope: FileScope, request: string, signal?: AbortSignal): Promise<TaskTriage | undefined> {
50
56
  const triage = await triageWithContext(classifier(), scope, request, signal);
@@ -0,0 +1,158 @@
1
+ /**
2
+ * Relevant knowledge. An agent's knowledge, standards and decisions are put in
3
+ * its prompt, not looked up, so a gate ("does this step need the knowledge
4
+ * base?") would have to guess before the agent knows what it needs, and a wrong
5
+ * "no" is silent. What Jev does safely instead is the ranking the keyword
6
+ * selector does badly: when a file is longer than the prompt has room for, Jev
7
+ * judges each of its sections against the step (one yes/no per section) and the
8
+ * ones that bear on it stay. A file that fits goes in whole and costs nothing.
9
+ *
10
+ * Whatever is left out is said so, with the file's path, so an agent that does
11
+ * need it reads the rest itself. Any failure, timeout or switch-off keeps the
12
+ * keyword selection, which is what agents got before.
13
+ */
14
+ import { basename } from "node:path";
15
+ import { score, selectRelevant, splitSections, tokenize, type KnowledgeInput, type KnowledgeSelection } from "../knowledge/selector.ts";
16
+ import type { Classifier } from "./classifier.ts";
17
+ import { noul, yesOf, type SystemOneRequest } from "./client.ts";
18
+ import { clip } from "./limits.ts";
19
+
20
+ /** How many characters of each knowledge file a prompt has room for. */
21
+ export const KNOWLEDGE_BUDGET_CHARS = 4000;
22
+ /** A section is judged on its first characters; the whole section is kept when it stays. */
23
+ export const SECTION_EXCERPT_CHARS = 700;
24
+ /** Sections judged per file; a longer file is narrowed by keywords first. */
25
+ export const MAX_SECTIONS = 60;
26
+ /** The step waits this long for Jev, then goes with keywords. */
27
+ export const KNOWLEDGE_BUDGET_MS = 1500;
28
+
29
+ /** The three files a prompt carries: `knowledge`, `standards` and `decisions`. */
30
+ export type KnowledgeSlice = keyof KnowledgeSelection;
31
+ export type KnowledgePaths = Partial<Record<KnowledgeSlice, string>>;
32
+
33
+ export interface SectionCall {
34
+ request: SystemOneRequest;
35
+ /** The section each question key stands for, by index into the file's sections. */
36
+ keys: Map<string, number>;
37
+ }
38
+
39
+ /** One request that asks about each candidate section of a file. */
40
+ export function sectionRequest(task: string, sections: readonly string[], candidates: readonly number[]): SectionCall {
41
+ const entries: Record<string, string> = {};
42
+ const questions: SystemOneRequest["questions"] = {};
43
+ const keys = new Map<string, number>();
44
+ candidates.forEach((index, position) => {
45
+ const key = `s${position + 1}`;
46
+ entries[key] = clip(sections[index]!, SECTION_EXCERPT_CHARS);
47
+ questions[key] = noul(`Does \`sections.${key}\` hold a fact, rule or decision that an agent doing \`task\` should have in front of it? Yes when it bears on the code, files, conventions or choices the task touches; no when it is about something else or only shares words with it.`);
48
+ keys.set(key, index);
49
+ });
50
+ return { request: { state: { task: clip(task, 3000), sections: entries }, questions }, keys };
51
+ }
52
+
53
+ /** The sections to judge: all of them, or the `MAX_SECTIONS` that share the most words with the task. */
54
+ export function candidateSections(task: string, sections: readonly string[]): number[] {
55
+ const all = sections.map((_, index) => index);
56
+ if (sections.length <= MAX_SECTIONS) return all;
57
+ const tokens = tokenize(task);
58
+ const scored = all.map((index) => ({ index, hits: score(sections[index]!, tokens) }));
59
+ scored.sort((a, b) => b.hits - a.hits || b.index - a.index);
60
+ return scored.slice(0, MAX_SECTIONS).map((entry) => entry.index).sort((a, b) => a - b);
61
+ }
62
+
63
+ /** What was kept of a file: its text and how many of its sections that was. */
64
+ export interface Kept {
65
+ text: string;
66
+ kept: number;
67
+ total: number;
68
+ }
69
+
70
+ /**
71
+ * The most relevant sections that fit `budget`, in the order of the file. A
72
+ * section too long to fit on its own is cut to fit when nothing else is kept.
73
+ */
74
+ export function keepSections(sections: readonly string[], relevance: ReadonlyMap<number, number>, floor: number, budget: number): Kept {
75
+ const wanted = [...relevance].filter(([, value]) => value >= floor).sort((a, b) => b[1] - a[1] || a[0] - b[0]);
76
+ const chosen: number[] = [];
77
+ const texts = new Map<number, string>();
78
+ let used = 0;
79
+ for (const [index] of wanted) {
80
+ const text = sections[index]!;
81
+ if (used + text.length <= budget) {
82
+ chosen.push(index);
83
+ texts.set(index, text);
84
+ used += text.length + 2;
85
+ } else if (chosen.length === 0) {
86
+ chosen.push(index);
87
+ texts.set(index, clip(text, budget));
88
+ used = budget;
89
+ }
90
+ }
91
+ chosen.sort((a, b) => a - b);
92
+ return { text: chosen.map((index) => texts.get(index)!).join("\n\n").trim(), kept: chosen.length, total: sections.length };
93
+ }
94
+
95
+ /** The line that tells an agent what it was not given, and where all of it is. */
96
+ export function leftOutNote(kept: Kept, name: string, path?: string): string {
97
+ const where = path ? ` The whole file is ${path}.` : "";
98
+ return kept.kept === 0
99
+ ? `_No section of ${name} bears on this step (${kept.total} sections).${where}_`
100
+ : `_Kept ${kept.kept} of ${kept.total} sections of ${name} for this step.${where}_`;
101
+ }
102
+
103
+ /** Picks what an agent's prompt carries of its knowledge files. */
104
+ export interface KnowledgePicker {
105
+ select(task: string, files: KnowledgeInput, options?: { signal?: AbortSignal; paths?: KnowledgePaths }): Promise<KnowledgeSelection>;
106
+ }
107
+
108
+ const SLICES: readonly KnowledgeSlice[] = ["knowledge", "standards", "decisions"];
109
+
110
+ /**
111
+ * Jev's picks for one file over the budget, or undefined when it cannot judge
112
+ * (off, failed, out of time, no key): the caller then uses keywords.
113
+ */
114
+ async function rank(classifier: Classifier, task: string, content: string, budget: number, options: { signal?: AbortSignal }): Promise<{ kept: Kept; sections: string[]; ms: number } | undefined> {
115
+ const sections = splitSections(content.trim());
116
+ if (sections.length < 2) return undefined;
117
+ const call = sectionRequest(task, sections, candidateSections(task, sections));
118
+ const result = await classifier.ask("knowledge", call.request, { ...(options.signal ? { signal: options.signal } : {}), timeoutMs: Math.min(classifier.config.timeoutMs, KNOWLEDGE_BUDGET_MS) });
119
+ if (!result) return undefined;
120
+ const relevance = new Map<number, number>();
121
+ for (const [key, index] of call.keys) {
122
+ const value = yesOf(result.answers, key);
123
+ if (value !== undefined) relevance.set(index, value);
124
+ }
125
+ // An answer to none of the questions says nothing: keywords decide.
126
+ if (relevance.size === 0) return undefined;
127
+ return { kept: keepSections(sections, relevance, classifier.config.thresholds.knowledgeRelevantAt, budget), sections, ms: result.ms };
128
+ }
129
+
130
+ /**
131
+ * A picker on Jev. Files that fit the prompt are untouched; a longer one keeps
132
+ * the sections Jev judges to bear on the step, with a note of what was left
133
+ * out. Standards are the exception to an empty result: they are rules for all
134
+ * work, so when Jev sees none that applies the keyword selection stands.
135
+ */
136
+ export function knowledgePicker(classifier: Classifier, log?: (text: string) => void, budget = KNOWLEDGE_BUDGET_CHARS): KnowledgePicker {
137
+ return {
138
+ async select(task, files, options = {}) {
139
+ const plain = (kind: KnowledgeSlice) => selectRelevant(task, files[kind] ?? "", budget);
140
+ const picked = await Promise.all(SLICES.map(async (kind): Promise<string> => {
141
+ const content = files[kind] ?? "";
142
+ if (content.trim().length <= budget || !task.trim() || !classifier.enabled("knowledge")) return plain(kind);
143
+ try {
144
+ const ranked = await rank(classifier, task, content, budget, options.signal ? { signal: options.signal } : {});
145
+ if (!ranked) return plain(kind);
146
+ const path = options.paths?.[kind];
147
+ const name = path ? basename(path) : kind;
148
+ if (ranked.kept.kept === 0 && kind === "standards") return plain(kind);
149
+ log?.(`knowledge: kept ${ranked.kept.kept} of ${ranked.kept.total} sections of ${name} · ${ranked.ms} ms`);
150
+ return [ranked.kept.text, leftOutNote(ranked.kept, name, path)].filter(Boolean).join("\n\n");
151
+ } catch {
152
+ return plain(kind);
153
+ }
154
+ }));
155
+ return { knowledge: picked[0]!, standards: picked[1]!, decisions: picked[2]! };
156
+ },
157
+ };
158
+ }
@@ -0,0 +1,146 @@
1
+ /**
2
+ * Jev's quick read of a pull request, for the Git tab: how big the change is,
3
+ * how likely it is risky, to break callers, to touch security-sensitive code
4
+ * or to change behaviour without tests, and what kind of change it is — in
5
+ * one call, in a moment, without writing a word. It is a triage, not a review:
6
+ * it says whether a full review by an agent is worth its tokens.
7
+ */
8
+ import type { Classifier } from "./classifier.ts";
9
+ import { choice, choiceOf, noul, score, scoreOf, yesOf, type SystemOneRequest } from "./client.ts";
10
+ import { clip } from "./limits.ts";
11
+ import { TRIAGE_SIZES } from "./triage.ts";
12
+ import type { TriageSize } from "../schemas/task.ts";
13
+
14
+ const SIZE_LEVELS = [
15
+ "trivial: a one-line or mechanical change (a rename, a typo, a version bump)",
16
+ "small: a few files in one area, following an existing pattern",
17
+ "medium: several files or two areas, with some design decisions",
18
+ "large: a cross-cutting change, a new subsystem, a migration or an architecture change",
19
+ ];
20
+
21
+ const KINDS: Record<string, string> = {
22
+ feature: "New capability or behaviour",
23
+ bugfix: "Existing behaviour was wrong and is fixed",
24
+ refactor: "Restructures code without changing behaviour",
25
+ tests: "Adds or fixes tests only",
26
+ docs: "Documentation only",
27
+ dependency: "Upgrades or changes dependencies",
28
+ chore: "Configuration, build, tooling or cleanup",
29
+ };
30
+
31
+ /** The pull request as Jev reads it. */
32
+ export interface PullInput {
33
+ title: string;
34
+ body: string;
35
+ /** Changed files with their line counts, e.g. `src/a.ts (+12 −3)`. */
36
+ files: readonly string[];
37
+ diff: string;
38
+ }
39
+
40
+ /** What Jev made of a pull request. */
41
+ export interface PullRead {
42
+ size: TriageSize;
43
+ sizeConfidence: number;
44
+ /** Probabilities of yes, 0 to 1. */
45
+ risky: number;
46
+ breaking: number;
47
+ security: number;
48
+ testsMissing: number;
49
+ kind?: string;
50
+ kindProbability?: number;
51
+ model: string;
52
+ ms: number;
53
+ }
54
+
55
+ /** Characters of diff sent: the rest is what the agent's review is for. */
56
+ export const READ_DIFF_CHARS = 80_000;
57
+
58
+ export function pullRequest(input: PullInput): SystemOneRequest {
59
+ return {
60
+ state: {
61
+ title: clip(input.title, 300),
62
+ description: clip(input.body, 3000),
63
+ changed_files: clip(input.files.join("\n"), 4000),
64
+ diff: clip(input.diff, READ_DIFF_CHARS),
65
+ },
66
+ questions: {
67
+ size: score("How big is the change in `diff`, judged by what it would take to review and verify?", SIZE_LEVELS),
68
+ risky: noul(
69
+ "Could merging `diff` break existing behaviour or cause harm: data loss, corrupted state, wrong results, concurrency bugs, performance cliffs, or an outage?",
70
+ "Yes: there is a plausible way this breaks something that works today.",
71
+ "No: it is low-risk, well contained, or purely additive.",
72
+ ),
73
+ breaking: noul(
74
+ "Does `diff` change a public API, a config or file format, a database schema or a command-line interface in a way that breaks existing callers or users?",
75
+ "Yes: something outside this change has to adapt.",
76
+ "No: existing callers and data keep working as they are.",
77
+ ),
78
+ security: noul(
79
+ "Does `diff` touch authentication, authorisation, secrets, input validation, cryptography, or other security-sensitive code?",
80
+ "Yes: a mistake here could be a vulnerability.",
81
+ "No: nothing in it is security-relevant.",
82
+ ),
83
+ tests_missing: noul(
84
+ "Does `diff` change behaviour without adding or updating tests that cover the change?",
85
+ "Yes: the behaviour changes and no test in the diff covers it.",
86
+ "No: it adds or updates tests for what it changes, or it changes no behaviour.",
87
+ ),
88
+ kind: choice("What kind of change is `diff`?", KINDS),
89
+ },
90
+ };
91
+ }
92
+
93
+ /** Jev's read of a pull request, or undefined when the classifier is off or fails. */
94
+ export async function readPull(classifier: Classifier, input: PullInput, signal?: AbortSignal): Promise<PullRead | undefined> {
95
+ if (!classifier.enabled("review")) return undefined;
96
+ const result = await classifier.ask("review", pullRequest(input), signal ? { signal } : {});
97
+ if (!result) return undefined;
98
+ const size = scoreOf(result.answers, "size");
99
+ if (!size) return undefined;
100
+ const kind = choiceOf(result.answers, "kind");
101
+ return {
102
+ size: TRIAGE_SIZES[Math.max(0, Math.min(TRIAGE_SIZES.length - 1, size.level))]!,
103
+ sizeConfidence: size.confidence,
104
+ risky: yesOf(result.answers, "risky") ?? 0,
105
+ breaking: yesOf(result.answers, "breaking") ?? 0,
106
+ security: yesOf(result.answers, "security") ?? 0,
107
+ testsMissing: yesOf(result.answers, "tests_missing") ?? 0,
108
+ ...(kind ? { kind: kind.choice, kindProbability: kind.probability } : {}),
109
+ model: result.model,
110
+ ms: result.ms,
111
+ };
112
+ }
113
+
114
+ const CONCERN = 0.5;
115
+
116
+ /** The concerns worth a look, most likely first, as words. */
117
+ export function concerns(read: PullRead): string[] {
118
+ const found: Array<[number, string]> = [
119
+ [read.risky, "risky"],
120
+ [read.security, "touches security"],
121
+ [read.breaking, "may break callers"],
122
+ [read.testsMissing, "no tests for the change"],
123
+ ];
124
+ return found.filter(([probability]) => probability >= CONCERN).sort((a, b) => b[0] - a[0]).map(([, words]) => words);
125
+ }
126
+
127
+ /** Whether a full review by an agent is worth its tokens: a concern, or a change too big to skim. */
128
+ export function worthReview(read: PullRead): boolean {
129
+ return concerns(read).length > 0 || read.size === "large";
130
+ }
131
+
132
+ /** One line: `medium · bugfix · risky 0.71, no tests for the change 0.64 → worth a full review`. */
133
+ export function readLine(read: PullRead): string {
134
+ const kind = read.kind ? ` · ${read.kind}` : "";
135
+ const flagged = concerns(read);
136
+ const figures = [
137
+ ["risky", read.risky],
138
+ ["security", read.security],
139
+ ["breaking", read.breaking],
140
+ ["tests missing", read.testsMissing],
141
+ ] as const;
142
+ const detail = flagged.length > 0
143
+ ? figures.filter(([, probability]) => probability >= CONCERN).sort((a, b) => b[1] - a[1]).map(([name, probability]) => `${name} ${probability.toFixed(2)}`).join(", ")
144
+ : "nothing stands out";
145
+ return `${read.size}${kind} · ${detail} → ${worthReview(read) ? "worth a full review" : "looks routine"}`;
146
+ }
@@ -7,6 +7,7 @@ import { shortDuration, truncate } from "../text.ts";
7
7
  import { EditLog } from "../state/changes.ts";
8
8
  import { formatMinutes, REPORT_GRACE_MS } from "../state/budget.ts";
9
9
  import { runPiAgent, spawnPiProcess, type PiStreamEvent, type ProcessRunner, type RelayAsk } from "./pi-runner.ts";
10
+ import { isUnavailable, looksUnavailable, markUnavailable, usableFallback } from "./fallback.ts";
10
11
  import { ASK_ENV } from "../ask/relay.ts";
11
12
  import { ASK_TOOL } from "../ask/types.ts";
12
13
 
@@ -40,6 +41,10 @@ export interface AgentRequest {
40
41
  context: AgentContext;
41
42
  model?: string;
42
43
  thinking?: string;
44
+ /** Where the run goes when its model runs out of usage or is unavailable. */
45
+ fallback?: { model: string; thinking: string };
46
+ /** Set by the runner once it has switched: the model that ran out. */
47
+ fellBackFrom?: string;
43
48
  timeoutMs: number;
44
49
  cwd: string;
45
50
  signal?: AbortSignal;
@@ -135,13 +140,24 @@ function retryable(run: AgentRun): boolean {
135
140
  */
136
141
  export async function runAgent(request: AgentRequest, run: ProcessRunner = spawnPiProcess): Promise<AgentRun> {
137
142
  const startedAt = new Date().toISOString();
138
- const attempts = Math.max(1, (request.retries ?? 0) + 1);
143
+ let attempts = Math.max(1, (request.retries ?? 0) + 1);
139
144
  // A failed attempt may have edited files before the retry: the run owns every edit.
140
145
  const edits = new EditLog(request.cwd);
141
146
  let last: AgentRun | undefined;
147
+ // A model already known to be out of usage goes straight to the fallback.
148
+ const other = usableFallback(request.fallback, request.model);
149
+ if (other && isUnavailable(request.model)) request = switchToFallback(request, other, request.model ?? "the session model");
142
150
  for (let attempt = 1; attempt <= attempts; attempt++) {
143
151
  last = await runAgentOnce(request, run, attempt, startedAt, edits);
144
152
  request.onAttemptEnd?.(last);
153
+ // Out of usage (or the model unavailable): the same model would fail again, so the fallback takes the retry.
154
+ const fallback = usableFallback(request.fallback, request.model);
155
+ if (fallback && !request.fellBackFrom && last.status === "failed" && looksUnavailable(last.error)) {
156
+ if (request.model) markUnavailable(request.model);
157
+ request = switchToFallback(request, fallback, request.model ?? "the session model");
158
+ attempts += 1;
159
+ continue;
160
+ }
145
161
  if (!retryable(last)) break;
146
162
  // Under a budget a retry only uses what is left of the allotment.
147
163
  if (request.time && request.time.endsAt - Date.now() < MIN_ATTEMPT_MS) break;
@@ -150,6 +166,11 @@ export async function runAgent(request: AgentRequest, run: ProcessRunner = spawn
150
166
  return edited.length > 0 ? { ...last!, edited } : last!;
151
167
  }
152
168
 
169
+ /** The request moved to its fallback model and thinking level. */
170
+ function switchToFallback(request: AgentRequest, fallback: { model: string; thinking?: string }, from: string): AgentRequest {
171
+ return { ...request, model: fallback.model, thinking: fallback.thinking ?? request.thinking, fellBackFrom: from };
172
+ }
173
+
153
174
  /** Abort every in-flight subagent (session shutdown, user cancel). */
154
175
  export function cancelAllRuns(): void {
155
176
  for (const controller of activeControllers) controller.abort();
@@ -168,6 +189,7 @@ function baseRun(request: AgentRequest, runId: string, startedAt: string, attemp
168
189
  attempts,
169
190
  startedAt,
170
191
  ...(request.thinking ? { thinking: request.thinking } : {}),
192
+ ...(request.fellBackFrom ? { fellBackFrom: request.fellBackFrom } : {}),
171
193
  ...(request.routedFrom ? { routedFrom: request.routedFrom } : {}),
172
194
  ...(request.route ? { route: request.route } : {}),
173
195
  ...(request.time ? { allotMs: request.time.allotMs, endsAt: request.time.endsAt } : {}),
@@ -0,0 +1,75 @@
1
+ /**
2
+ * Fallback models. A subscription or key that runs out mid-task (a usage
3
+ * limit, a rate limit, no credit, the provider down or refusing the key)
4
+ * fails every run on that model the same way, so the run goes to the agent's
5
+ * configured fallback model instead, and the exhausted model is skipped for a
6
+ * while so the next agents do not each spend a failed run finding out.
7
+ */
8
+
9
+ /** What a provider says when a model cannot be used right now. Deliberately narrow: an ordinary failure is not a reason to switch model. */
10
+ const UNAVAILABLE = [
11
+ /usage limit/i, /rate.?limit/i, /\bquota\b/i, /limit (?:reached|exceeded|will reset)/i, /exceeded (?:your|the|current)/i,
12
+ /out of (?:usage|credits?|extra usage|tokens)/i, /(?:insufficient|no) (?:credits?|funds|balance|quota)/i, /credit balance/i,
13
+ /billing/i, /payment required/i, /\b402\b/, /\b429\b/, /too many requests/i, /resource.?exhausted/i,
14
+ /overloaded/i, /\b529\b/, /(?:at|over) capacity/i, /resets? (?:at|in|on)/i, /subscription/i,
15
+ /unauthori[sz]ed/i, /\b401\b/, /invalid (?:api )?key/i, /no api key/i, /authentication (?:failed|error)/i,
16
+ /model (?:is )?(?:not found|not available|unavailable)/i, /unknown model/i, /does not have access/i,
17
+ ];
18
+
19
+ /** True when a failed run's error says its model is out of usage or unavailable, so another model may succeed. */
20
+ export function looksUnavailable(error: string | undefined): boolean {
21
+ return Boolean(error && UNAVAILABLE.some((pattern) => pattern.test(error)));
22
+ }
23
+
24
+ /** How long a model that ran out is skipped before it is tried again. */
25
+ export const COOLDOWN_MS = 20 * 60 * 1000;
26
+
27
+ const exhausted = new Map<string, number>();
28
+
29
+ /** Remember that `model` cannot be used until the cooldown passes. */
30
+ export function markUnavailable(model: string, now = Date.now()): void {
31
+ exhausted.set(model, now + COOLDOWN_MS);
32
+ }
33
+
34
+ /** Whether `model` ran out recently enough to skip. */
35
+ export function isUnavailable(model: string | undefined, now = Date.now()): boolean {
36
+ if (!model) return false;
37
+ const until = exhausted.get(model);
38
+ if (until === undefined) return false;
39
+ if (until <= now) {
40
+ exhausted.delete(model);
41
+ return false;
42
+ }
43
+ return true;
44
+ }
45
+
46
+ /** Forget every exhausted model (tests). */
47
+ export function resetUnavailable(): void {
48
+ exhausted.clear();
49
+ }
50
+
51
+ /** A fallback worth switching to: a model other than the one that failed. */
52
+ export function usableFallback<F extends { model: string }>(fallback: F | undefined, current: string | undefined): F | undefined {
53
+ return fallback && fallback.model !== current ? fallback : undefined;
54
+ }
55
+
56
+ /**
57
+ * Run `attempt` on `model`; when it fails because that model is out of usage
58
+ * or unavailable, run it again on the fallback. A model already known to be
59
+ * out goes straight to the fallback. `switched` says which model took over.
60
+ */
61
+ export async function withFallback<T extends { status: string; error?: string }>(
62
+ model: string | undefined,
63
+ thinking: string,
64
+ fallback: { model: string; thinking: string } | undefined,
65
+ attempt: (model: string | undefined, thinking: string) => Promise<T>,
66
+ ): Promise<{ result: T; switchedFrom?: string }> {
67
+ const other = usableFallback(fallback, model);
68
+ if (other && isUnavailable(model)) return { result: await attempt(other.model, other.thinking), switchedFrom: model ?? "the session model" };
69
+ const result = await attempt(model, thinking);
70
+ if (other && result.status === "failed" && looksUnavailable(result.error)) {
71
+ if (model) markUnavailable(model);
72
+ return { result: await attempt(other.model, other.thinking), switchedFrom: model ?? "the session model" };
73
+ }
74
+ return { result };
75
+ }