@a-t-h-i/bot-lobby 0.6.1 → 0.6.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +177 -834
- package/package.json +1 -1
- package/prompts/master.md +17 -0
- package/prompts/panel.md +3 -1
- package/prompts/planner.md +14 -0
- package/prompts/quickfix.md +3 -1
- package/prompts/scout.md +3 -0
- package/prompts/worker.md +3 -0
- package/src/classifier/answers.ts +110 -0
- package/src/classifier/classifier.ts +171 -0
- package/src/classifier/client.ts +239 -0
- package/src/classifier/effort.ts +143 -0
- package/src/classifier/files.ts +427 -0
- package/src/classifier/hosts.ts +157 -0
- package/src/classifier/instance.ts +98 -0
- package/src/classifier/limits.ts +37 -0
- package/src/classifier/seats.ts +102 -0
- package/src/classifier/tools.ts +58 -0
- package/src/classifier/triage.ts +203 -0
- package/src/execution/agent-runner.ts +5 -0
- package/src/index.ts +6 -0
- package/src/lobby/planner.ts +305 -27
- package/src/lobby/quickfix.ts +120 -8
- package/src/lobby/runtime.ts +12 -1
- package/src/lobby/tabs/metrics.ts +28 -1
- package/src/lobby/tabs/plan.ts +27 -4
- package/src/lobby/tabs/quickfix.ts +5 -1
- package/src/lobby/view.ts +28 -4
- package/src/master/master.ts +95 -29
- package/src/pi/commands.ts +4 -2
- package/src/pi/events.ts +11 -2
- package/src/pi/run-summary.ts +3 -1
- package/src/pi/settings-ui.ts +138 -2
- package/src/pi/start-task.ts +4 -0
- package/src/pi/tools.ts +7 -2
- package/src/schemas/configuration.ts +125 -1
- package/src/schemas/findings.ts +4 -0
- package/src/schemas/task.ts +22 -0
- package/src/state/metrics.ts +74 -1
- package/src/workflow/workflow.ts +49 -1
|
@@ -108,6 +108,59 @@ export interface LobbyConfig {
|
|
|
108
108
|
keys: Record<string, string>;
|
|
109
109
|
/** Clicks and the wheel work in the lobby (click a draft line to comment on it); shift+drag still selects text. */
|
|
110
110
|
mouse: boolean;
|
|
111
|
+
/**
|
|
112
|
+
* Planning rounds before the oracle finalizes the plan on its own: the last
|
|
113
|
+
* round skips the seats and asks nothing; later replies only revise. 0 = unlimited.
|
|
114
|
+
*/
|
|
115
|
+
maxPlanningRounds: number;
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
/** Decisions the classifier can make, each switched on or off on its own. */
|
|
119
|
+
export const CLASSIFIER_FEATURES = ["seats", "answers", "files", "triage", "effort"] as const;
|
|
120
|
+
export type ClassifierFeature = (typeof CLASSIFIER_FEATURES)[number];
|
|
121
|
+
|
|
122
|
+
/** Hosts that serve Jev behind the same System One API; `auto` takes OpenCode's free Jev when pi holds an OpenCode key, else TypeSafe. */
|
|
123
|
+
export const JEV_HOSTS = ["auto", "opencode", "typesafe", "openrouter", "vercel"] as const;
|
|
124
|
+
export type JevHostName = (typeof JEV_HOSTS)[number];
|
|
125
|
+
|
|
126
|
+
/** Probability and confidence cut-offs, 0 to 1; edited in the config file only. */
|
|
127
|
+
export interface ClassifierThresholds {
|
|
128
|
+
/** A planning seat runs when the idea or the latest answers touch its domain at least this likely. */
|
|
129
|
+
seatAt: number;
|
|
130
|
+
/** A seat that was READY comes back only at this probability. */
|
|
131
|
+
reseatReadyAt: number;
|
|
132
|
+
/** An obvious question is answered for you at this probability, when the pick is the recommended option… */
|
|
133
|
+
autoAnswerAt: number;
|
|
134
|
+
/** …and leads the runner-up by at least this much. */
|
|
135
|
+
autoAnswerMargin: number;
|
|
136
|
+
/** A file is a likely file at this relevance. */
|
|
137
|
+
fileRelevantAt: number;
|
|
138
|
+
/** A step scored simple at this confidence runs one thinking level lower. */
|
|
139
|
+
simpleAt: number;
|
|
140
|
+
/** A step scored trivial at this confidence runs on the cheaper model. */
|
|
141
|
+
trivialAt: number;
|
|
142
|
+
/** A quick fix scored large at this confidence is held instead of started. */
|
|
143
|
+
quickFixLargeAt: number;
|
|
144
|
+
}
|
|
145
|
+
|
|
146
|
+
/** The Jev classifier: a fast model for obvious decisions, so large models spend fewer tokens on them. */
|
|
147
|
+
export interface ClassifierConfig {
|
|
148
|
+
enabled: boolean;
|
|
149
|
+
/** Where Jev is called; the key comes from pi's own key store for that host. */
|
|
150
|
+
provider: JevHostName;
|
|
151
|
+
/** A pinned Jev model; empty uses the host's default. */
|
|
152
|
+
model: string;
|
|
153
|
+
/** A base URL override (a proxy); empty uses the host's. */
|
|
154
|
+
baseUrl: string;
|
|
155
|
+
/** Time limit per classifier call. */
|
|
156
|
+
timeoutMs: number;
|
|
157
|
+
features: Record<ClassifierFeature, boolean>;
|
|
158
|
+
thresholds: ClassifierThresholds;
|
|
159
|
+
fileHints: { topK: number; maxCandidates: number; budgetMs: number };
|
|
160
|
+
/** The cheaper model trivial steps run on; `inherit` keeps the configured model and only lowers thinking. */
|
|
161
|
+
effort: { cheapModel: ModelRef };
|
|
162
|
+
/** Path globs never sent to the classifier (file hints). */
|
|
163
|
+
exclude: string[];
|
|
111
164
|
}
|
|
112
165
|
|
|
113
166
|
export interface BotLobbyConfig {
|
|
@@ -122,6 +175,7 @@ export interface BotLobbyConfig {
|
|
|
122
175
|
workflow: WorkflowConfig;
|
|
123
176
|
knowledge: KnowledgeConfig;
|
|
124
177
|
lobby: LobbyConfig;
|
|
178
|
+
classifier: ClassifierConfig;
|
|
125
179
|
}
|
|
126
180
|
|
|
127
181
|
export const DEFAULT_CONFIG: BotLobbyConfig = {
|
|
@@ -162,9 +216,34 @@ export const DEFAULT_CONFIG: BotLobbyConfig = {
|
|
|
162
216
|
panels: { animations: false, conversation: true, activity: true, thinking: true },
|
|
163
217
|
keys: {},
|
|
164
218
|
mouse: true,
|
|
219
|
+
maxPlanningRounds: 5,
|
|
220
|
+
},
|
|
221
|
+
classifier: {
|
|
222
|
+
enabled: false,
|
|
223
|
+
provider: "auto",
|
|
224
|
+
model: "",
|
|
225
|
+
baseUrl: "",
|
|
226
|
+
timeoutMs: 4000,
|
|
227
|
+
features: { seats: true, answers: true, files: true, triage: true, effort: true },
|
|
228
|
+
thresholds: {
|
|
229
|
+
seatAt: 0.35,
|
|
230
|
+
reseatReadyAt: 0.6,
|
|
231
|
+
autoAnswerAt: 0.9,
|
|
232
|
+
autoAnswerMargin: 0.5,
|
|
233
|
+
fileRelevantAt: 0.5,
|
|
234
|
+
simpleAt: 0.7,
|
|
235
|
+
trivialAt: 0.8,
|
|
236
|
+
quickFixLargeAt: 0.8,
|
|
237
|
+
},
|
|
238
|
+
fileHints: { topK: 8, maxCandidates: 480, budgetMs: 1500 },
|
|
239
|
+
effort: { cheapModel: INHERIT_MODEL },
|
|
240
|
+
exclude: [],
|
|
165
241
|
},
|
|
166
242
|
};
|
|
167
243
|
|
|
244
|
+
/** Choices the settings menu cycles through for the planning round limit; 0 = unlimited. */
|
|
245
|
+
export const PLANNING_ROUND_CHOICES = [2, 3, 5, 8, 0] as const;
|
|
246
|
+
|
|
168
247
|
function positive(value: unknown): number | undefined {
|
|
169
248
|
return typeof value === "number" && Number.isFinite(value) && value > 0 ? value : undefined;
|
|
170
249
|
}
|
|
@@ -191,7 +270,7 @@ function flag(value: unknown, fallback: boolean): boolean {
|
|
|
191
270
|
}
|
|
192
271
|
|
|
193
272
|
function normalizeLobby(value: unknown): LobbyConfig {
|
|
194
|
-
const source = value as { autoOpen?: unknown; planningPanel?: unknown; autoAsk?: unknown; issues?: unknown; panels?: unknown; keys?: unknown; mouse?: unknown } | undefined;
|
|
273
|
+
const source = value as { autoOpen?: unknown; planningPanel?: unknown; autoAsk?: unknown; issues?: unknown; panels?: unknown; keys?: unknown; mouse?: unknown; maxPlanningRounds?: unknown } | undefined;
|
|
195
274
|
const defaults = DEFAULT_CONFIG.lobby;
|
|
196
275
|
const panel = Array.isArray(source?.planningPanel)
|
|
197
276
|
? [...new Set(source.planningPanel.filter((entry): entry is PanelMember => typeof entry === "string" && isPanelMember(entry)))]
|
|
@@ -206,6 +285,50 @@ function normalizeLobby(value: unknown): LobbyConfig {
|
|
|
206
285
|
panels: Object.fromEntries(LOBBY_PANELS.map((name) => [name, flag(panels[name], defaults.panels[name])])) as Record<LobbyPanel, boolean>,
|
|
207
286
|
keys: Object.fromEntries(Object.entries(keys).filter((entry): entry is [string, string] => typeof entry[1] === "string" && entry[1].trim().length > 0)),
|
|
208
287
|
mouse: flag(source?.mouse, defaults.mouse),
|
|
288
|
+
maxPlanningRounds: roundLimit(source?.maxPlanningRounds, defaults.maxPlanningRounds),
|
|
289
|
+
};
|
|
290
|
+
}
|
|
291
|
+
|
|
292
|
+
/** A whole number of rounds, 0 for unlimited; anything else keeps the default. */
|
|
293
|
+
function roundLimit(value: unknown, fallback: number): number {
|
|
294
|
+
return typeof value === "number" && Number.isInteger(value) && value >= 0 ? value : fallback;
|
|
295
|
+
}
|
|
296
|
+
|
|
297
|
+
function probability(value: unknown, fallback: number): number {
|
|
298
|
+
return typeof value === "number" && Number.isFinite(value) && value >= 0 && value <= 1 ? value : fallback;
|
|
299
|
+
}
|
|
300
|
+
|
|
301
|
+
function count(value: unknown, fallback: number): number {
|
|
302
|
+
return typeof value === "number" && Number.isInteger(value) && value > 0 ? value : fallback;
|
|
303
|
+
}
|
|
304
|
+
|
|
305
|
+
function text(value: unknown, fallback: string): string {
|
|
306
|
+
return typeof value === "string" ? value.trim() : fallback;
|
|
307
|
+
}
|
|
308
|
+
|
|
309
|
+
function normalizeClassifier(value: unknown): ClassifierConfig {
|
|
310
|
+
const source = (value && typeof value === "object" ? value : {}) as Record<string, unknown>;
|
|
311
|
+
const defaults = DEFAULT_CONFIG.classifier;
|
|
312
|
+
const features = (source.features ?? {}) as Partial<Record<ClassifierFeature, unknown>>;
|
|
313
|
+
const thresholds = (source.thresholds ?? {}) as Partial<Record<keyof ClassifierThresholds, unknown>>;
|
|
314
|
+
const hints = (source.fileHints ?? {}) as Record<string, unknown>;
|
|
315
|
+
const effort = (source.effort ?? {}) as Record<string, unknown>;
|
|
316
|
+
const provider = typeof source.provider === "string" && (JEV_HOSTS as readonly string[]).includes(source.provider) ? source.provider as JevHostName : defaults.provider;
|
|
317
|
+
return {
|
|
318
|
+
enabled: flag(source.enabled, defaults.enabled),
|
|
319
|
+
provider,
|
|
320
|
+
model: text(source.model, defaults.model),
|
|
321
|
+
baseUrl: text(source.baseUrl, defaults.baseUrl),
|
|
322
|
+
timeoutMs: count(source.timeoutMs, defaults.timeoutMs),
|
|
323
|
+
features: Object.fromEntries(CLASSIFIER_FEATURES.map((name) => [name, flag(features[name], defaults.features[name])])) as Record<ClassifierFeature, boolean>,
|
|
324
|
+
thresholds: Object.fromEntries(Object.entries(defaults.thresholds).map(([name, fallback]) => [name, probability(thresholds[name as keyof ClassifierThresholds], fallback)])) as unknown as ClassifierThresholds,
|
|
325
|
+
fileHints: {
|
|
326
|
+
topK: count(hints.topK, defaults.fileHints.topK),
|
|
327
|
+
maxCandidates: count(hints.maxCandidates, defaults.fileHints.maxCandidates),
|
|
328
|
+
budgetMs: count(hints.budgetMs, defaults.fileHints.budgetMs),
|
|
329
|
+
},
|
|
330
|
+
effort: { cheapModel: typeof effort.cheapModel === "string" && effort.cheapModel.trim() ? effort.cheapModel.trim() : defaults.effort.cheapModel },
|
|
331
|
+
exclude: Array.isArray(source.exclude) ? source.exclude.filter((entry): entry is string => typeof entry === "string" && entry.trim().length > 0) : [...defaults.exclude],
|
|
209
332
|
};
|
|
210
333
|
}
|
|
211
334
|
|
|
@@ -229,6 +352,7 @@ export function resolveConfig(partial: unknown): BotLobbyConfig {
|
|
|
229
352
|
workflow,
|
|
230
353
|
knowledge,
|
|
231
354
|
lobby: normalizeLobby(src.lobby),
|
|
355
|
+
classifier: normalizeClassifier(src.classifier),
|
|
232
356
|
};
|
|
233
357
|
}
|
|
234
358
|
|
package/src/schemas/findings.ts
CHANGED
|
@@ -133,4 +133,8 @@ export interface AgentRun {
|
|
|
133
133
|
wrappedUp?: boolean;
|
|
134
134
|
/** File this worker is queued for at the file desk. */
|
|
135
135
|
waitingFor?: string;
|
|
136
|
+
/** The classifier routed this run down: the configured profile it came from (`p/big · medium`). */
|
|
137
|
+
routedFrom?: string;
|
|
138
|
+
/** Why and where it was routed (`trivial 0.88: p/big · medium → p/cheap · low`). */
|
|
139
|
+
route?: string;
|
|
136
140
|
}
|
package/src/schemas/task.ts
CHANGED
|
@@ -89,11 +89,31 @@ export interface RunLogEntry {
|
|
|
89
89
|
stalled?: boolean;
|
|
90
90
|
wrappedUp?: boolean;
|
|
91
91
|
error?: string;
|
|
92
|
+
/** The classifier routed this run down from this configured profile. */
|
|
93
|
+
routedFrom?: string;
|
|
92
94
|
}
|
|
93
95
|
|
|
94
96
|
/** Upper bound on persisted run-log entries. */
|
|
95
97
|
export const MAX_RUN_LOG = 64;
|
|
96
98
|
|
|
99
|
+
export type TriageSize = "trivial" | "small" | "medium" | "large";
|
|
100
|
+
|
|
101
|
+
/** What the classifier made of the request when the task started: hints for the Master, never rules. */
|
|
102
|
+
export interface TaskTriage {
|
|
103
|
+
size: TriageSize;
|
|
104
|
+
sizeConfidence: number;
|
|
105
|
+
/** How likely each domain is touched, 0 to 1. */
|
|
106
|
+
domains: Partial<Record<Domain, number>>;
|
|
107
|
+
/** How likely building it needs outside facts. */
|
|
108
|
+
research: number;
|
|
109
|
+
/** How likely it is ambiguous as written. */
|
|
110
|
+
ambiguous: number;
|
|
111
|
+
kind?: string;
|
|
112
|
+
kindProbability?: number;
|
|
113
|
+
likelyFiles?: string[];
|
|
114
|
+
at: string;
|
|
115
|
+
}
|
|
116
|
+
|
|
97
117
|
export interface Task {
|
|
98
118
|
id: string;
|
|
99
119
|
title: string;
|
|
@@ -123,6 +143,8 @@ export interface Task {
|
|
|
123
143
|
approvedPlan?: string;
|
|
124
144
|
/** When the task was archived from the lobby (it then lives under archive/tasks, out of every list). */
|
|
125
145
|
archivedAt?: string;
|
|
146
|
+
/** The classifier's read of the request, when it was on as the task started. */
|
|
147
|
+
triage?: TaskTriage;
|
|
126
148
|
}
|
|
127
149
|
|
|
128
150
|
export function createTask(
|
package/src/state/metrics.ts
CHANGED
|
@@ -11,7 +11,7 @@ import type { AgentRun } from "../schemas/findings.ts";
|
|
|
11
11
|
import type { RunLogEntry, Task } from "../schemas/task.ts";
|
|
12
12
|
import { dataRoot } from "./project.ts";
|
|
13
13
|
|
|
14
|
-
export const METRIC_KINDS = ["master", "scout", "worker", "reviewer", "researcher", "quickfix", "planner", "panel"] as const;
|
|
14
|
+
export const METRIC_KINDS = ["master", "scout", "worker", "reviewer", "researcher", "quickfix", "planner", "panel", "classifier"] as const;
|
|
15
15
|
export type MetricKind = (typeof METRIC_KINDS)[number];
|
|
16
16
|
|
|
17
17
|
export type MetricStatus = "success" | "failed" | "cancelled" | "timeout";
|
|
@@ -33,6 +33,67 @@ export interface MetricRecord {
|
|
|
33
33
|
cost?: number;
|
|
34
34
|
taskId?: string;
|
|
35
35
|
stalled?: boolean;
|
|
36
|
+
/** Classifier calls: what the call decided (seats, answers, files, triage, effort, test). */
|
|
37
|
+
purpose?: string;
|
|
38
|
+
/** A run the classifier routed down: the configured profile it came from. */
|
|
39
|
+
routedFrom?: string;
|
|
40
|
+
/** Classifier calls: what the decision spared (seat runs skipped, questions answered, quick fixes held). */
|
|
41
|
+
saved?: number;
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
/** What the classifier did and spared, for the Metrics tab. */
|
|
45
|
+
export interface ClassifierSummary {
|
|
46
|
+
calls: number;
|
|
47
|
+
ok: number;
|
|
48
|
+
p50Ms: number;
|
|
49
|
+
p90Ms: number;
|
|
50
|
+
/** Tokens Jev read across every call. */
|
|
51
|
+
input: number;
|
|
52
|
+
/** Calls per decision (seats, answers, files, triage, effort, test). */
|
|
53
|
+
byPurpose: Record<string, number>;
|
|
54
|
+
seatRunsSkipped: number;
|
|
55
|
+
questionsAnswered: number;
|
|
56
|
+
quickFixesHeld: number;
|
|
57
|
+
/** Agent runs the classifier routed down, and how many of them succeeded. */
|
|
58
|
+
routed: number;
|
|
59
|
+
routedOk: number;
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
function quantile(sorted: readonly number[], q: number): number {
|
|
63
|
+
if (sorted.length === 0) return 0;
|
|
64
|
+
return sorted[Math.min(sorted.length - 1, Math.floor(q * sorted.length))]!;
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
/** Classifier calls and the agent runs it routed, summarised; undefined when it never ran. */
|
|
68
|
+
export function summarizeClassifier(calls: readonly MetricRecord[], runs: readonly MetricRecord[]): ClassifierSummary | undefined {
|
|
69
|
+
const routed = runs.filter((record) => record.routedFrom);
|
|
70
|
+
if (calls.length === 0 && routed.length === 0) return undefined;
|
|
71
|
+
const times = calls.map((record) => record.durationMs).sort((a, b) => a - b);
|
|
72
|
+
const byPurpose: Record<string, number> = {};
|
|
73
|
+
let seatRunsSkipped = 0;
|
|
74
|
+
let questionsAnswered = 0;
|
|
75
|
+
let quickFixesHeld = 0;
|
|
76
|
+
for (const record of calls) {
|
|
77
|
+
const purpose = record.purpose ?? "other";
|
|
78
|
+
byPurpose[purpose] = (byPurpose[purpose] ?? 0) + 1;
|
|
79
|
+
const saved = record.saved ?? 0;
|
|
80
|
+
if (purpose === "seats") seatRunsSkipped += saved;
|
|
81
|
+
else if (purpose === "answers") questionsAnswered += saved;
|
|
82
|
+
else if (purpose === "triage") quickFixesHeld += saved;
|
|
83
|
+
}
|
|
84
|
+
return {
|
|
85
|
+
calls: calls.length,
|
|
86
|
+
ok: calls.filter((record) => record.status === "success").length,
|
|
87
|
+
p50Ms: quantile(times, 0.5),
|
|
88
|
+
p90Ms: quantile(times, 0.9),
|
|
89
|
+
input: calls.reduce((total, record) => total + (record.input ?? 0), 0),
|
|
90
|
+
byPurpose,
|
|
91
|
+
seatRunsSkipped,
|
|
92
|
+
questionsAnswered,
|
|
93
|
+
quickFixesHeld,
|
|
94
|
+
routed: routed.length,
|
|
95
|
+
routedOk: routed.filter((record) => record.status === "success").length,
|
|
96
|
+
};
|
|
36
97
|
}
|
|
37
98
|
|
|
38
99
|
/** Newest records kept in memory for aggregation. */
|
|
@@ -58,7 +119,17 @@ function isRecord(value: unknown): value is MetricRecord {
|
|
|
58
119
|
return Boolean(record && typeof record.id === "string" && typeof record.kind === "string" && typeof record.durationMs === "number");
|
|
59
120
|
}
|
|
60
121
|
|
|
122
|
+
/** Agent runs for the Metrics tab; classifier calls, a few hundred milliseconds each, are read apart. */
|
|
61
123
|
export function readMetrics(root: string, configDir: string, limit = MAX_READ): MetricRecord[] {
|
|
124
|
+
return readRecords(root, configDir, limit).filter((record) => record.kind !== "classifier");
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
/** Classifier calls only. */
|
|
128
|
+
export function readClassifierMetrics(root: string, configDir: string, limit = MAX_READ): MetricRecord[] {
|
|
129
|
+
return readRecords(root, configDir, limit).filter((record) => record.kind === "classifier");
|
|
130
|
+
}
|
|
131
|
+
|
|
132
|
+
function readRecords(root: string, configDir: string, limit: number): MetricRecord[] {
|
|
62
133
|
const path = metricsPath(root, configDir);
|
|
63
134
|
if (!existsSync(path)) return [];
|
|
64
135
|
let text: string;
|
|
@@ -110,6 +181,7 @@ export function metricFromRun(run: AgentRun): MetricRecord {
|
|
|
110
181
|
...(run.usage ? { input: run.usage.input, output: run.usage.output, cost: run.usage.cost } : {}),
|
|
111
182
|
taskId: run.taskId,
|
|
112
183
|
...(run.stalled ? { stalled: true } : {}),
|
|
184
|
+
...(run.routedFrom ? { routedFrom: run.routedFrom } : {}),
|
|
113
185
|
};
|
|
114
186
|
}
|
|
115
187
|
|
|
@@ -130,6 +202,7 @@ export function metricFromLog(entry: RunLogEntry, taskId: string): MetricRecord
|
|
|
130
202
|
...(entry.input !== undefined ? { input: entry.input, output: entry.output ?? 0, cost: entry.cost ?? 0 } : {}),
|
|
131
203
|
taskId,
|
|
132
204
|
...(entry.stalled ? { stalled: true } : {}),
|
|
205
|
+
...(entry.routedFrom ? { routedFrom: entry.routedFrom } : {}),
|
|
133
206
|
};
|
|
134
207
|
}
|
|
135
208
|
|
package/src/workflow/workflow.ts
CHANGED
|
@@ -47,6 +47,10 @@ import { assertNoPendingApprovals, pendingApprovals, requestApproval, resolveApp
|
|
|
47
47
|
import { pingApproval } from "../pi/notify.ts";
|
|
48
48
|
import { describeRun, runLogEntry } from "../pi/run-summary.ts";
|
|
49
49
|
import { nextStates } from "./transitions.ts";
|
|
50
|
+
import type { FileHinter } from "../classifier/files.ts";
|
|
51
|
+
import type { Classifier } from "../classifier/classifier.ts";
|
|
52
|
+
import type { EffortRouter } from "../classifier/effort.ts";
|
|
53
|
+
import { answerClarify } from "../classifier/triage.ts";
|
|
50
54
|
import { appendMetrics, metricFromRun } from "../state/metrics.ts";
|
|
51
55
|
import { markCommentsAddressed, pendingComments, readPlanComments } from "../state/comments.ts";
|
|
52
56
|
|
|
@@ -116,6 +120,14 @@ export interface WorkflowDeps {
|
|
|
116
120
|
choose: (title: string, options: string[]) => Promise<string | undefined>;
|
|
117
121
|
notify: (message: string, level?: "info" | "warning" | "error") => void;
|
|
118
122
|
runProcess?: ProcessRunner;
|
|
123
|
+
/** Likely files for scouts and workers, while the classifier's file hints are on. */
|
|
124
|
+
hints?: FileHinter;
|
|
125
|
+
/** Answers a clarify question whose recommended option is clearly right, when it is on. */
|
|
126
|
+
classifier?: Classifier;
|
|
127
|
+
/** Re-reads a request after an amendment (the task's triage), when the classifier is on. */
|
|
128
|
+
triage?: (request: string, signal?: AbortSignal) => Promise<Task["triage"]>;
|
|
129
|
+
/** Lowers thinking or the model for scouts and workers on steps the classifier judges simple or trivial. */
|
|
130
|
+
effort?: EffortRouter;
|
|
119
131
|
}
|
|
120
132
|
|
|
121
133
|
export interface WorkflowResult {
|
|
@@ -253,6 +265,11 @@ async function handleClarify(task: Task, params: OrchestrateParams, deps: Workfl
|
|
|
253
265
|
const question = params.question?.trim();
|
|
254
266
|
if (!question) throw new Error("clarify requires a question");
|
|
255
267
|
transition(task, "clarifying");
|
|
268
|
+
const settled = await clarifyByClassifier(task, question, params.options ?? [], deps);
|
|
269
|
+
if (settled) {
|
|
270
|
+
recordDecision(task, `Answered by the classifier (${settled.probability.toFixed(2)}): ${truncate(question, 300)} → ${settled.answer}`);
|
|
271
|
+
return `Answered by the classifier with your recommended option (${settled.probability.toFixed(2)}): ${settled.answer}. It is recorded as a decision; mention it in your proposal so the user can amend it, and continue.`;
|
|
272
|
+
}
|
|
256
273
|
const unattended = unattendedReason(task, isAutoMode(deps.root, deps.configDir, task.id));
|
|
257
274
|
if (unattended) {
|
|
258
275
|
recordDecision(task, `Not asked (${unattended}): ${truncate(question, 300)}`);
|
|
@@ -265,6 +282,20 @@ async function handleClarify(task: Task, params: OrchestrateParams, deps: Workfl
|
|
|
265
282
|
return `User answered: ${answer}`;
|
|
266
283
|
}
|
|
267
284
|
|
|
285
|
+
/** The classifier's answer to a clarify question with options, when the request already settles it. */
|
|
286
|
+
async function clarifyByClassifier(task: Task, question: string, options: readonly string[], deps: WorkflowDeps): Promise<{ answer: string; probability: number } | undefined> {
|
|
287
|
+
if (!deps.classifier || options.length < 2) return undefined;
|
|
288
|
+
const notes = [
|
|
289
|
+
...task.amendments.map((text) => `Amendment: ${text}`),
|
|
290
|
+
...task.decisions.slice(-12).map((decision) => `Decision: ${decision.text}`),
|
|
291
|
+
].join("\n");
|
|
292
|
+
try {
|
|
293
|
+
return await answerClarify(deps.classifier, question, options, { request: taskRequest(task), notes, ...(task.proposal ? { proposal: task.proposal } : {}) }, deps.signal);
|
|
294
|
+
} catch {
|
|
295
|
+
return undefined;
|
|
296
|
+
}
|
|
297
|
+
}
|
|
298
|
+
|
|
268
299
|
async function handleScout(task: Task, params: OrchestrateParams, deps: WorkflowDeps): Promise<string> {
|
|
269
300
|
requireState(task, ["created", "clarifying", "scouting", "synthesizing"]);
|
|
270
301
|
const domains = deriveDomains(params.domains);
|
|
@@ -284,6 +315,8 @@ async function handleScout(task: Task, params: OrchestrateParams, deps: Workflow
|
|
|
284
315
|
profile: deps.profile,
|
|
285
316
|
signal: deps.signal,
|
|
286
317
|
onUpdate: deps.onUpdate,
|
|
318
|
+
...(deps.hints ? { hints: deps.hints } : {}),
|
|
319
|
+
...(deps.effort ? { effort: deps.effort } : {}),
|
|
287
320
|
},
|
|
288
321
|
deps.runProcess ?? spawnPiProcess,
|
|
289
322
|
);
|
|
@@ -421,7 +454,20 @@ async function handlePropose(task: Task, params: OrchestrateParams, deps: Workfl
|
|
|
421
454
|
const lower = choice.toLowerCase();
|
|
422
455
|
if (lower.startsWith("approve")) return applyApprovalChoice(task, "approve");
|
|
423
456
|
if (lower.startsWith("decline")) return applyApprovalChoice(task, "decline");
|
|
424
|
-
|
|
457
|
+
const result = applyApprovalChoice(task, "amend", await deps.ask("What should change?"));
|
|
458
|
+
await retriage(task, deps);
|
|
459
|
+
return result;
|
|
460
|
+
}
|
|
461
|
+
|
|
462
|
+
/** After an amendment, the classifier reads the request again, so the Master's hints follow the change. */
|
|
463
|
+
async function retriage(task: Task, deps: WorkflowDeps): Promise<void> {
|
|
464
|
+
if (!deps.triage || !task.triage) return;
|
|
465
|
+
try {
|
|
466
|
+
const next = await deps.triage([taskRequest(task), ...task.amendments.map((text) => `Amendment: ${text}`)].join("\n\n"), deps.signal);
|
|
467
|
+
if (next) task.triage = next;
|
|
468
|
+
} catch {
|
|
469
|
+
// Hints only: the old triage stays.
|
|
470
|
+
}
|
|
425
471
|
}
|
|
426
472
|
|
|
427
473
|
/** States in which `plan` replaces an approved plan instead of recording the first one. */
|
|
@@ -570,6 +616,8 @@ function workerRequest(deps: WorkflowDeps, task: Task, domain: Domain, instructi
|
|
|
570
616
|
profile: deps.profile,
|
|
571
617
|
signal: deps.signal,
|
|
572
618
|
onUpdate: deps.onUpdate,
|
|
619
|
+
...(deps.hints ? { hints: deps.hints } : {}),
|
|
620
|
+
...(deps.effort ? { effort: deps.effort } : {}),
|
|
573
621
|
};
|
|
574
622
|
}
|
|
575
623
|
|