flint-agent 1.14.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.example +108 -0
- package/CHANGELOG.md +55 -0
- package/FEATURES.md +298 -0
- package/LICENSE +21 -0
- package/README.md +435 -0
- package/bin/flint.js +47 -0
- package/config/classifier-prompt.md +218 -0
- package/config/models-curated.json +4 -0
- package/config/providers.json +74 -0
- package/package.json +92 -0
- package/patches/ink+6.8.0.patch +78 -0
- package/profiles/desktop.md +65 -0
- package/profiles/generic.md +20 -0
- package/profiles/marketer.md +20 -0
- package/profiles/profiles.json +34 -0
- package/profiles/ux-reviewer.md +25 -0
- package/src/agent/agent.js +1743 -0
- package/src/agent/auto.js +346 -0
- package/src/agent/backoff.js +143 -0
- package/src/agent/compression.js +310 -0
- package/src/agent/content-resolver.js +180 -0
- package/src/agent/flow-controller.js +309 -0
- package/src/agent/intent-manifest.js +231 -0
- package/src/agent/intent-timeout.js +46 -0
- package/src/agent/intent.js +633 -0
- package/src/agent/knowledge.js +114 -0
- package/src/agent/learning.js +180 -0
- package/src/agent/modes.js +187 -0
- package/src/agent/outcome-ask.js +91 -0
- package/src/agent/project-context.js +76 -0
- package/src/agent/prompt-budget.js +117 -0
- package/src/agent/reflection-extractor.js +140 -0
- package/src/agent/steering.js +86 -0
- package/src/agent/supervisor.js +430 -0
- package/src/agent/swap.js +443 -0
- package/src/agent/system-prompt.js +446 -0
- package/src/agent/time-stamp.js +48 -0
- package/src/agent/tool-guard.js +201 -0
- package/src/agent/toolcall-text.js +162 -0
- package/src/agent/usage.js +297 -0
- package/src/agent/vision.js +94 -0
- package/src/agent/watchdog.js +139 -0
- package/src/agent/workspace-changes.js +177 -0
- package/src/api/address.js +14 -0
- package/src/api/client.js +280 -0
- package/src/api/server.js +535 -0
- package/src/api/stream-pipe.js +113 -0
- package/src/app-state.js +39 -0
- package/src/bootstrap.js +501 -0
- package/src/bus/drain-loop.js +497 -0
- package/src/bus/index.js +270 -0
- package/src/bus/plugins.js +65 -0
- package/src/child-idle.js +14 -0
- package/src/cli.js +118 -0
- package/src/commands/commands.js +1297 -0
- package/src/commands/registry.js +132 -0
- package/src/components/App.js +491 -0
- package/src/components/CarefulMenu.js +145 -0
- package/src/components/HistoryWriter.js +86 -0
- package/src/components/LineInput.js +69 -0
- package/src/components/LiveZone.js +294 -0
- package/src/components/OverlayMenu.js +179 -0
- package/src/components/SystemPanel.js +156 -0
- package/src/components/Table.js +54 -0
- package/src/config.js +249 -0
- package/src/free-models.js +230 -0
- package/src/index.js +1111 -0
- package/src/input-handler.js +13 -0
- package/src/input-text.js +123 -0
- package/src/launcher.js +129 -0
- package/src/logging/api-log.js +95 -0
- package/src/logging/chat-log-follower.js +113 -0
- package/src/logging/chat-log.js +15 -0
- package/src/logging/log-collector.js +182 -0
- package/src/logging/logger.js +112 -0
- package/src/logging/tool-log.js +20 -0
- package/src/mcp-client.js +314 -0
- package/src/memory/conversation-digest.js +113 -0
- package/src/memory/extract-facts.js +98 -0
- package/src/memory/facts.js +181 -0
- package/src/memory/inbox.js +63 -0
- package/src/memory/markdown.js +38 -0
- package/src/memory/patterns.js +185 -0
- package/src/memory/project.js +66 -0
- package/src/memory/reflections.js +74 -0
- package/src/memory/retrieval.js +84 -0
- package/src/memory/rules.js +105 -0
- package/src/memory/session-facts.js +125 -0
- package/src/memory/skills.js +191 -0
- package/src/memory/sqlite-store.js +653 -0
- package/src/memory/store.js +208 -0
- package/src/memory/tools.js +196 -0
- package/src/memory/user-model.js +86 -0
- package/src/message-handler.js +775 -0
- package/src/model-check.js +218 -0
- package/src/plugins/loader.js +120 -0
- package/src/plugins/manager.js +88 -0
- package/src/production-env.js +22 -0
- package/src/profiles.js +42 -0
- package/src/providers/adapters/anthropic.js +270 -0
- package/src/providers/adapters/openai.js +120 -0
- package/src/providers/keys-dpapi.js +41 -0
- package/src/providers/keys-fallback.js +31 -0
- package/src/providers/keys.js +132 -0
- package/src/providers/models.js +154 -0
- package/src/providers/registry.js +56 -0
- package/src/providers/state.js +56 -0
- package/src/registry.js +96 -0
- package/src/restart.js +29 -0
- package/src/sandbox/backend.js +130 -0
- package/src/security/api-auth.js +132 -0
- package/src/security/audit.js +98 -0
- package/src/security/child-policy.js +41 -0
- package/src/security/command-guard.js +173 -0
- package/src/security/content-fence.js +250 -0
- package/src/security/content-validator.js +132 -0
- package/src/security/index.js +143 -0
- package/src/security/network-guard.js +126 -0
- package/src/security/pairing.js +180 -0
- package/src/security/path-guard.js +140 -0
- package/src/security/persona-guard.js +67 -0
- package/src/security/policies.js +452 -0
- package/src/security/safety-constants.js +34 -0
- package/src/security/watchdog.js +107 -0
- package/src/sessions.js +130 -0
- package/src/spend.js +97 -0
- package/src/startup-watchdog.js +59 -0
- package/src/stdio/args.js +71 -0
- package/src/stdio/guard.js +59 -0
- package/src/stdio/protocol.js +167 -0
- package/src/stdio/run.js +106 -0
- package/src/stdio/session.js +180 -0
- package/src/store/agent-slice.js +306 -0
- package/src/store/dataset-slice.js +73 -0
- package/src/store/index.js +22 -0
- package/src/store/process-slice.js +135 -0
- package/src/store/session-slice.js +191 -0
- package/src/store/ui-slice.js +119 -0
- package/src/tasks/db.js +184 -0
- package/src/tasks/queries.js +589 -0
- package/src/tools/agent-tools.js +473 -0
- package/src/tools/checkpoint.js +152 -0
- package/src/tools/command-approvals.js +180 -0
- package/src/tools/dataset.js +50 -0
- package/src/tools/filesystem.js +682 -0
- package/src/tools/inbox-tools.js +48 -0
- package/src/tools/mesh.js +135 -0
- package/src/tools/own-env.js +136 -0
- package/src/tools/permissions.js +681 -0
- package/src/tools/plugin-tools.js +123 -0
- package/src/tools/process-tools.js +595 -0
- package/src/tools/registry.js +307 -0
- package/src/tools/swap-tools.js +72 -0
- package/src/tools/system.js +662 -0
- package/src/tools/tasks.js +532 -0
- package/src/tools/tool-search.js +171 -0
- package/src/ui/header.js +140 -0
- package/src/ui/input-cursor.js +23 -0
- package/src/ui/last-line.js +25 -0
- package/src/ui/line-edit.js +135 -0
- package/src/ui/output.js +399 -0
- package/src/ui/paste-tokens.js +131 -0
- package/src/ui/prompt-attention.js +134 -0
- package/src/ui/render-options.js +13 -0
- package/src/ui/replay.js +94 -0
- package/src/ui/splash.js +49 -0
- package/src/ui/status-level.js +36 -0
- package/src/ui/tool-ledger.js +203 -0
- package/src/ui/window-title.js +150 -0
- package/src/update.js +205 -0
- package/system.md +63 -0
|
@@ -0,0 +1,346 @@
|
|
|
1
|
+
// Autonomous mode for Flint
|
|
2
|
+
// Wraps processMessage in a plan-driven loop
|
|
3
|
+
// Tasks persist in SQLite — can resume across sessions
|
|
4
|
+
|
|
5
|
+
import chalk from "chalk";
|
|
6
|
+
import { getActiveGoal, getTasksByGoal, getTaskStats } from "../tasks/queries.js";
|
|
7
|
+
import { syncPlanToStore } from "../tasks/queries.js";
|
|
8
|
+
import path from "node:path";
|
|
9
|
+
import { setDeniedPaths, clearDeniedPaths } from "../tools/filesystem.js";
|
|
10
|
+
import { beginRun, endRun } from "./usage.js";
|
|
11
|
+
import { config } from "../config.js";
|
|
12
|
+
import {
|
|
13
|
+
FATAL_PROVIDER_REASONS,
|
|
14
|
+
MAX_CONSECUTIVE_RATE_LIMITS,
|
|
15
|
+
RATE_LIMIT_WAIT_MS,
|
|
16
|
+
STOP_REASON_TEXT,
|
|
17
|
+
} from "./flow-controller.js";
|
|
18
|
+
|
|
19
|
+
const AUTO_SYSTEM_INJECTION = `
|
|
20
|
+
You are in AUTONOMOUS MODE. You are working on a task independently.
|
|
21
|
+
|
|
22
|
+
Rules:
|
|
23
|
+
1. You MUST create a plan (create_plan tool) as your FIRST action if no plan exists
|
|
24
|
+
2. Work through the plan step by step — focus on ONE task at a time
|
|
25
|
+
3. After completing each step, call update_task with status "done"
|
|
26
|
+
4. Verify your work after each step (read files you wrote, run tests, check output)
|
|
27
|
+
5. If stuck on a step after 2 attempts, REVISE the plan:
|
|
28
|
+
- If the step can be done differently, use add_task to create an alternative step, then skip the stuck one
|
|
29
|
+
- If later steps depend on the stuck step, reorder or skip them too
|
|
30
|
+
- Use add_task_note to explain why the original step failed and what alternative you chose
|
|
31
|
+
- Only mark a step "skipped" as last resort after trying an alternative
|
|
32
|
+
6. When all steps are done, provide a final summary of what was accomplished
|
|
33
|
+
7. Do NOT ask the user questions — make reasonable decisions yourself
|
|
34
|
+
8. If a step requires information you don't have, skip it and note why in the result
|
|
35
|
+
9. Use link_task_file to link files you create or modify to the relevant task
|
|
36
|
+
10. Use add_task_note for important observations during work
|
|
37
|
+
`.trim();
|
|
38
|
+
|
|
39
|
+
const CONTINUE_PROMPT = "Continue with the next pending task in the plan. Check the plan status (list_tasks) and work on the next 'pending' task.";
|
|
40
|
+
|
|
41
|
+
const PLAN_REMINDER = "You MUST create a plan first using the create_plan tool. Break the task into concrete steps, then execute them one by one.";
|
|
42
|
+
|
|
43
|
+
const BUDGET_WARNING_70 = "[BUDGET WARNING: You have used 70% of your iteration budget. Start wrapping up — finish the current task, skip non-essential remaining tasks, prepare a summary.]";
|
|
44
|
+
|
|
45
|
+
const BUDGET_WARNING_90 = "[BUDGET CRITICAL: 90% of iterations used. STOP after this step. Provide a final summary of what was accomplished and what remains.]";
|
|
46
|
+
|
|
47
|
+
const FINISH_PROMPT = "All tasks in the plan are complete (or skipped). Provide a final summary: what was accomplished, what was skipped and why, and any next steps the user should know about.";
|
|
48
|
+
|
|
49
|
+
const RESUME_PROMPT = `You are RESUMING autonomous work on an existing plan from a previous session.
|
|
50
|
+
The plan and task statuses are loaded from the database.
|
|
51
|
+
Review the current plan status (list_tasks) and continue with the next pending task.`;
|
|
52
|
+
|
|
53
|
+
// What the provider's refusals mean lives in flow-controller.js, shared with
|
|
54
|
+
// the bus loop so the two autonomous doors cannot disagree about which
|
|
55
|
+
// reasons are fatal. A 429 is not one of them: it says not now, rather than not
|
|
56
|
+
// ever, and a run left going overnight has to survive a per-minute cap.
|
|
57
|
+
|
|
58
|
+
/**
|
|
59
|
+
* Run autonomous mode
|
|
60
|
+
*
|
|
61
|
+
* @param {string} task - The user's task description (or null to resume existing)
|
|
62
|
+
* @param {object} options
|
|
63
|
+
* @param {Function} options.processMessage - The main processMessage function
|
|
64
|
+
* @param {Function} options.getStore - Returns the store
|
|
65
|
+
* @param {Function} options.printSystem - Print system messages
|
|
66
|
+
* @param {Function} options.printWarning - Print warnings
|
|
67
|
+
* @param {number} options.maxIterations - Max auto iterations (default 50)
|
|
68
|
+
* @param {number} options.maxCost - Max cost in dollars (default 0.50)
|
|
69
|
+
* @param {Function} options.isAborted - Check if aborted
|
|
70
|
+
* @param {boolean} options.resume - True if resuming existing plan
|
|
71
|
+
* @param {Function} options.wait - Sleep for n ms; injected so a test can read
|
|
72
|
+
* the decision (how long the run chose to wait out a 429) without serving it
|
|
73
|
+
* @returns {object} { completed, tasksTotal, tasksDone, tasksSkipped, totalCost, iterations, stopReason, stopDetail }
|
|
74
|
+
*/
|
|
75
|
+
export async function runAutoMode(task, options) {
|
|
76
|
+
const {
|
|
77
|
+
processMessage,
|
|
78
|
+
getStore,
|
|
79
|
+
printSystem,
|
|
80
|
+
printWarning,
|
|
81
|
+
maxIterations = 50,
|
|
82
|
+
maxCost = 0.50,
|
|
83
|
+
isAborted,
|
|
84
|
+
resume = false,
|
|
85
|
+
wait = (ms) => new Promise((resolve) => setTimeout(resolve, ms)),
|
|
86
|
+
} = options;
|
|
87
|
+
|
|
88
|
+
const store = getStore();
|
|
89
|
+
|
|
90
|
+
// Sandbox: block access to Flint's own source + permissions during auto mode
|
|
91
|
+
const permFile = path.resolve(config.projectRoot, ".permissions.json");
|
|
92
|
+
setDeniedPaths([config.projectRoot, permFile]);
|
|
93
|
+
printSystem(`[AUTO] Sandbox: ${config.projectRoot} is protected`);
|
|
94
|
+
|
|
95
|
+
// Sync SQLite → store at start
|
|
96
|
+
syncPlanToStore(store);
|
|
97
|
+
|
|
98
|
+
let iteration = 0;
|
|
99
|
+
// This run's ceiling, declared to the door as well. The door refuses a call
|
|
100
|
+
// the moment the run's budget is gone, which closes the gap this loop cannot
|
|
101
|
+
// see: between two of its checks a single turn can make dozens of calls.
|
|
102
|
+
beginRun(maxCost);
|
|
103
|
+
// What this run has spent, read from the one place that publishes it: the
|
|
104
|
+
// session total the message handler keeps, which is the drained ledger and
|
|
105
|
+
// nothing else. Auto mode used to add up `stats.cost` from each turn it
|
|
106
|
+
// happened to look at, which is the main loop's share only — the classifier,
|
|
107
|
+
// the extractor and the three final messages whose result is discarded all
|
|
108
|
+
// spent outside that sum. A run asked for a $1 ceiling went to $2.80.
|
|
109
|
+
//
|
|
110
|
+
// Deliberately the published total and not the ledger's own counter: this
|
|
111
|
+
// loop runs BETWEEN turns, where the two agree, and reading what the
|
|
112
|
+
// operator is shown means the number in the banner and the number the run
|
|
113
|
+
// stops on can never be two different numbers.
|
|
114
|
+
//
|
|
115
|
+
// A delta, not the absolute total: `maxCost` is this run's budget, and the
|
|
116
|
+
// session may already have spent money before /auto was typed.
|
|
117
|
+
const costAtStart = store.getState().sessionCost || 0;
|
|
118
|
+
const spentSoFar = () => (store.getState().sessionCost || 0) - costAtStart;
|
|
119
|
+
let totalCost = 0;
|
|
120
|
+
let planCreated = false;
|
|
121
|
+
let planReminderSent = false;
|
|
122
|
+
let finished = false;
|
|
123
|
+
// Why the provider ended the run, if it did. Null means the run ended on its
|
|
124
|
+
// own terms: plan finished, budget spent, operator aborted.
|
|
125
|
+
let stopReason = null;
|
|
126
|
+
let stopDetail = null;
|
|
127
|
+
let rateLimitedInARow = 0;
|
|
128
|
+
|
|
129
|
+
/**
|
|
130
|
+
* One turn, and the only place this function talks to the agent.
|
|
131
|
+
*
|
|
132
|
+
* There used to be four such places, and two separate benchmark runs
|
|
133
|
+
* patched three of them and left the one inside the loop, which is the one a
|
|
134
|
+
* long run spends all its time in. One door cannot be half-closed.
|
|
135
|
+
*/
|
|
136
|
+
async function turn(message) {
|
|
137
|
+
for (;;) {
|
|
138
|
+
const r = await processMessage(message, null);
|
|
139
|
+
const reason = r?.stop_reason;
|
|
140
|
+
|
|
141
|
+
if (reason === "rate-limit") {
|
|
142
|
+
rateLimitedInARow++;
|
|
143
|
+
if (rateLimitedInARow > MAX_CONSECUTIVE_RATE_LIMITS || isAborted?.()) {
|
|
144
|
+
stopReason = "rate-limit";
|
|
145
|
+
stopDetail = r?.text || "";
|
|
146
|
+
return r;
|
|
147
|
+
}
|
|
148
|
+
// The provider's own number when it sent one, a fixed pause when it did
|
|
149
|
+
// not. Retrying the same turn, so it is not counted as an iteration:
|
|
150
|
+
// nothing was done and nothing was spent.
|
|
151
|
+
const waitMs = r?.retryAfter != null ? r.retryAfter * 1000 : RATE_LIMIT_WAIT_MS;
|
|
152
|
+
printWarning(
|
|
153
|
+
`[AUTO] Rate limited by the provider. Waiting ${Math.round(waitMs / 1000)}s, ` +
|
|
154
|
+
`then retrying (${rateLimitedInARow} of ${MAX_CONSECUTIVE_RATE_LIMITS}).`,
|
|
155
|
+
);
|
|
156
|
+
await wait(waitMs);
|
|
157
|
+
continue;
|
|
158
|
+
}
|
|
159
|
+
|
|
160
|
+
rateLimitedInARow = 0;
|
|
161
|
+
if (FATAL_PROVIDER_REASONS.has(reason)) {
|
|
162
|
+
stopReason = reason;
|
|
163
|
+
stopDetail = r?.text || "";
|
|
164
|
+
}
|
|
165
|
+
return r;
|
|
166
|
+
}
|
|
167
|
+
}
|
|
168
|
+
|
|
169
|
+
// Check for existing active goal
|
|
170
|
+
const existingGoal = getActiveGoal();
|
|
171
|
+
|
|
172
|
+
if (resume && existingGoal) {
|
|
173
|
+
const stats = getTaskStats(existingGoal.id);
|
|
174
|
+
printSystem(`[AUTO] Resuming: ${existingGoal.title} (${stats.done}/${stats.total} done, ${stats.pending} pending)`);
|
|
175
|
+
printSystem(`[AUTO] Budget: ${maxIterations} iterations, $${maxCost.toFixed(2)} max cost`);
|
|
176
|
+
planCreated = true;
|
|
177
|
+
|
|
178
|
+
// First message: resume context
|
|
179
|
+
await turn(`${AUTO_SYSTEM_INJECTION}\n\n${RESUME_PROMPT}\n\nOriginal task: ${existingGoal.title}`);
|
|
180
|
+
iteration++;
|
|
181
|
+
totalCost = spentSoFar();
|
|
182
|
+
updateAutoStatus(store, iteration, maxIterations, totalCost);
|
|
183
|
+
} else {
|
|
184
|
+
printSystem(`[AUTO] Starting: ${task}`);
|
|
185
|
+
printSystem(`[AUTO] Budget: ${maxIterations} iterations, $${maxCost.toFixed(2)} max cost`);
|
|
186
|
+
|
|
187
|
+
// First message: the task itself with auto mode instructions
|
|
188
|
+
await turn(`${AUTO_SYSTEM_INJECTION}\n\nTask: ${task}`);
|
|
189
|
+
iteration++;
|
|
190
|
+
totalCost = spentSoFar();
|
|
191
|
+
updateAutoStatus(store, iteration, maxIterations, totalCost);
|
|
192
|
+
}
|
|
193
|
+
|
|
194
|
+
try {
|
|
195
|
+
// Main auto loop. `stopReason` ends it the moment the provider says the
|
|
196
|
+
// next turn is pointless, including when it said so on the opening turn
|
|
197
|
+
// above, before the loop was ever entered.
|
|
198
|
+
while (!finished && !stopReason) {
|
|
199
|
+
if (isAborted?.()) {
|
|
200
|
+
printWarning("[AUTO] Aborted by user");
|
|
201
|
+
break;
|
|
202
|
+
}
|
|
203
|
+
|
|
204
|
+
if (iteration >= maxIterations) {
|
|
205
|
+
printWarning(`[AUTO] Iteration limit reached (${maxIterations})`);
|
|
206
|
+
await processMessage(
|
|
207
|
+
"[BUDGET EXHAUSTED] You have reached the maximum number of iterations. Provide a final summary of what was accomplished and what remains.",
|
|
208
|
+
null,
|
|
209
|
+
);
|
|
210
|
+
break;
|
|
211
|
+
}
|
|
212
|
+
|
|
213
|
+
if (totalCost >= maxCost) {
|
|
214
|
+
// No closing turn here, unlike the iteration limit above. That turn was
|
|
215
|
+
// a model call, and the ceiling it would be paid over is the one just
|
|
216
|
+
// reached: buying a summary to be told the budget is gone is the same
|
|
217
|
+
// overshoot this rule prevents, and the budget door now refuses it
|
|
218
|
+
// anyway. The run's own summary below says what was done. Iterations
|
|
219
|
+
// are not money, so that limit still gets its wrap-up turn.
|
|
220
|
+
printWarning(`[AUTO] Cost limit reached ($${totalCost.toFixed(4)} >= $${maxCost.toFixed(2)}). No closing turn: it would be spent over the ceiling.`);
|
|
221
|
+
break;
|
|
222
|
+
}
|
|
223
|
+
|
|
224
|
+
// Sync from SQLite
|
|
225
|
+
const plan = syncPlanToStore(store);
|
|
226
|
+
|
|
227
|
+
if (!plan) {
|
|
228
|
+
if (!planReminderSent) {
|
|
229
|
+
planReminderSent = true;
|
|
230
|
+
await turn(PLAN_REMINDER);
|
|
231
|
+
iteration++;
|
|
232
|
+
totalCost = spentSoFar();
|
|
233
|
+
updateAutoStatus(store, iteration, maxIterations, totalCost);
|
|
234
|
+
if (stopReason) break;
|
|
235
|
+
continue;
|
|
236
|
+
} else {
|
|
237
|
+
printWarning("[AUTO] Agent did not create a plan. Stopping.");
|
|
238
|
+
break;
|
|
239
|
+
}
|
|
240
|
+
}
|
|
241
|
+
|
|
242
|
+
if (!planCreated) {
|
|
243
|
+
planCreated = true;
|
|
244
|
+
printSystem(`[AUTO] Plan created: ${plan.goal} (${plan.tasks.length} tasks)`);
|
|
245
|
+
}
|
|
246
|
+
|
|
247
|
+
// Check plan progress from SQLite
|
|
248
|
+
const pending = plan.tasks.filter((t) => t.status === "pending");
|
|
249
|
+
const inProgress = plan.tasks.filter((t) => t.status === "in_progress");
|
|
250
|
+
const done = plan.tasks.filter((t) => t.status === "done");
|
|
251
|
+
const skipped = plan.tasks.filter((t) => t.status === "skipped");
|
|
252
|
+
|
|
253
|
+
if (pending.length === 0 && inProgress.length === 0) {
|
|
254
|
+
printSystem(`[AUTO] All tasks completed (${done.length} done, ${skipped.length} skipped)`);
|
|
255
|
+
await processMessage(FINISH_PROMPT, null);
|
|
256
|
+
iteration++;
|
|
257
|
+
finished = true;
|
|
258
|
+
break;
|
|
259
|
+
}
|
|
260
|
+
|
|
261
|
+
// Budget warning
|
|
262
|
+
let budgetWarning = "";
|
|
263
|
+
const progress = iteration / maxIterations;
|
|
264
|
+
if (progress >= 0.9) {
|
|
265
|
+
budgetWarning = "\n\n" + BUDGET_WARNING_90;
|
|
266
|
+
} else if (progress >= 0.7) {
|
|
267
|
+
budgetWarning = "\n\n" + BUDGET_WARNING_70;
|
|
268
|
+
}
|
|
269
|
+
|
|
270
|
+
const continueMsg = CONTINUE_PROMPT + budgetWarning;
|
|
271
|
+
await turn(continueMsg);
|
|
272
|
+
iteration++;
|
|
273
|
+
totalCost = spentSoFar();
|
|
274
|
+
|
|
275
|
+
updateAutoStatus(store, iteration, maxIterations, totalCost);
|
|
276
|
+
}
|
|
277
|
+
} catch (err) {
|
|
278
|
+
clearDeniedPaths();
|
|
279
|
+
if (err.name === "AbortError") {
|
|
280
|
+
printWarning("[AUTO] Aborted");
|
|
281
|
+
} else {
|
|
282
|
+
printWarning(`[AUTO] Error: ${err.message}`);
|
|
283
|
+
}
|
|
284
|
+
}
|
|
285
|
+
|
|
286
|
+
// Final stats from SQLite
|
|
287
|
+
const finalPlan = syncPlanToStore(store);
|
|
288
|
+
// Read once more: the three messages that end a run, the iteration limit, the cost
|
|
289
|
+
// limit and the final summary, throw their result away, so their cost never reached
|
|
290
|
+
// the old counter and the figure the operator was shown was short by a turn
|
|
291
|
+
// or two of every finished run.
|
|
292
|
+
totalCost = spentSoFar();
|
|
293
|
+
// The run window closes here, and with it its ceiling. Anything the operator
|
|
294
|
+
// does afterwards is bounded by the per-action and session ceilings, not by
|
|
295
|
+
// the budget of a run that has ended.
|
|
296
|
+
endRun();
|
|
297
|
+
const stats = {
|
|
298
|
+
completed: finished,
|
|
299
|
+
iterations: iteration,
|
|
300
|
+
totalCost,
|
|
301
|
+
tasksTotal: finalPlan ? finalPlan.tasks.length : 0,
|
|
302
|
+
tasksDone: finalPlan ? finalPlan.tasks.filter((t) => t.status === "done").length : 0,
|
|
303
|
+
tasksSkipped: finalPlan ? finalPlan.tasks.filter((t) => t.status === "skipped").length : 0,
|
|
304
|
+
// Null unless the provider ended the run. Callers that only counted
|
|
305
|
+
// iterations reported "50 iterations, not done" for a run that actually
|
|
306
|
+
// stopped on turn five with an expired key.
|
|
307
|
+
stopReason,
|
|
308
|
+
stopDetail,
|
|
309
|
+
};
|
|
310
|
+
|
|
311
|
+
// Remove sandbox
|
|
312
|
+
clearDeniedPaths();
|
|
313
|
+
|
|
314
|
+
if (stopReason) {
|
|
315
|
+
printWarning(
|
|
316
|
+
`[AUTO] Stopped on iteration ${iteration} of ${maxIterations}: ` +
|
|
317
|
+
`${STOP_REASON_TEXT[stopReason] || stopReason} (${stopReason}). ` +
|
|
318
|
+
`${stats.tasksDone}/${stats.tasksTotal} tasks done, $${totalCost.toFixed(4)} spent.`,
|
|
319
|
+
);
|
|
320
|
+
if (stopDetail) printSystem(`[AUTO] Provider said: ${stopDetail}`);
|
|
321
|
+
} else {
|
|
322
|
+
printSystem(
|
|
323
|
+
`[AUTO] Finished: ${stats.tasksDone}/${stats.tasksTotal} tasks done, ` +
|
|
324
|
+
`${stats.tasksSkipped} skipped, ${iteration} iterations, $${totalCost.toFixed(4)}`,
|
|
325
|
+
);
|
|
326
|
+
}
|
|
327
|
+
|
|
328
|
+
store.setState({ autoMode: null });
|
|
329
|
+
return stats;
|
|
330
|
+
}
|
|
331
|
+
|
|
332
|
+
function updateAutoStatus(store, iteration, maxIterations, cost) {
|
|
333
|
+
const plan = store.getState().plan;
|
|
334
|
+
const tasksDone = plan ? plan.tasks.filter((t) => t.status === "done").length : 0;
|
|
335
|
+
const tasksTotal = plan ? plan.tasks.length : 0;
|
|
336
|
+
|
|
337
|
+
store.setState({
|
|
338
|
+
autoMode: {
|
|
339
|
+
iteration,
|
|
340
|
+
maxIterations,
|
|
341
|
+
cost,
|
|
342
|
+
tasksDone,
|
|
343
|
+
tasksTotal,
|
|
344
|
+
},
|
|
345
|
+
});
|
|
346
|
+
}
|
|
@@ -0,0 +1,143 @@
|
|
|
1
|
+
// Pauses that grow, for the two failures that are not the model's fault.
|
|
2
|
+
//
|
|
3
|
+
// 1. The model answers with nothing, several times running. Every one of those
|
|
4
|
+
// is billed (17 in one session). Today Flint gives up on the third and
|
|
5
|
+
// asks the operator to type "continue", which is the operator doing the
|
|
6
|
+
// waiting that the agent is built to do. A silent model is usually a
|
|
7
|
+
// provider that is briefly overloaded, and it comes back on its own.
|
|
8
|
+
//
|
|
9
|
+
// 2. The provider refuses with a temporary error: a 429, or the 400 that an
|
|
10
|
+
// overloaded OpenRouter backend sends with "rate-limited upstream" in the
|
|
11
|
+
// body. Both say "not now", and both used to end the turn instantly.
|
|
12
|
+
//
|
|
13
|
+
// So: both get the same treatment — wait, and the pause grows. 30s, 1m, 2m.
|
|
14
|
+
// The operator sees a countdown rather than a frozen screen, because the
|
|
15
|
+
// difference between "waiting, and here is when" and "hung" is the only thing
|
|
16
|
+
// that lets somebody decide not to press Esc.
|
|
17
|
+
|
|
18
|
+
/** Empty answers in a row before the turn stops and says so. */
|
|
19
|
+
export const EMPTY_RETRY_LIMIT = 3;
|
|
20
|
+
|
|
21
|
+
/** Temporary provider refusals in a row before the turn stops and says so. */
|
|
22
|
+
export function tempErrorRetryLimit() {
|
|
23
|
+
const v = parseInt(process.env.AGENT_TEMP_ERROR_RETRIES || "", 10);
|
|
24
|
+
return Number.isFinite(v) && v > 0 ? v : 3;
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
/** First pause. Every later pause is this times two, capped at 2 minutes. */
|
|
28
|
+
export function backoffBaseMs() {
|
|
29
|
+
const v = parseInt(process.env.AGENT_BACKOFF_MS || "", 10);
|
|
30
|
+
return Number.isFinite(v) && v > 0 ? v : 30_000;
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
const MAX_PAUSE_MS = 120_000;
|
|
34
|
+
|
|
35
|
+
/**
|
|
36
|
+
* The pause before attempt number `attempt` (1-based).
|
|
37
|
+
*
|
|
38
|
+
* Grows, caps, and is monotonic: attempt 3 waits as long as attempt 4 does
|
|
39
|
+
* rather than going back down, because a model that has been silent longer is
|
|
40
|
+
* not about to answer in 30 seconds.
|
|
41
|
+
*
|
|
42
|
+
* @param {number} attempt — 1-based
|
|
43
|
+
* @param {number} [baseMs]
|
|
44
|
+
* @returns {number} milliseconds
|
|
45
|
+
*/
|
|
46
|
+
export function backoffMs(attempt, baseMs = backoffBaseMs()) {
|
|
47
|
+
const n = Math.max(1, attempt | 0);
|
|
48
|
+
return Math.min(MAX_PAUSE_MS, baseMs * Math.pow(2, n - 1));
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
/** m:ss, the way a countdown is read. 45s is "0:45", not "45". */
|
|
52
|
+
export function formatCountdown(ms) {
|
|
53
|
+
const total = Math.max(0, Math.ceil(ms / 1000));
|
|
54
|
+
const m = Math.floor(total / 60);
|
|
55
|
+
const s = total % 60;
|
|
56
|
+
return `${m}:${String(s).padStart(2, "0")}`;
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
/** The line the operator watches while Flint waits. */
|
|
60
|
+
export function waitNotice(reason, msLeft) {
|
|
61
|
+
return `${reason}, retrying in ${formatCountdown(msLeft)} (Esc to stop)`;
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
/**
|
|
65
|
+
* Wait, counting down out loud.
|
|
66
|
+
*
|
|
67
|
+
* Resolves true when the whole pause elapsed, false when it was cut short by
|
|
68
|
+
* the abort signal — the caller must not treat "aborted" as "waited", or a turn
|
|
69
|
+
* that was stopped with Esc carries on talking to the provider.
|
|
70
|
+
*
|
|
71
|
+
* @param {number} ms
|
|
72
|
+
* @param {object} opts - { signal, onTick(remainingMs), tickMs, reason }
|
|
73
|
+
* @returns {Promise<boolean>}
|
|
74
|
+
*/
|
|
75
|
+
export function sleepWithCountdown(ms, { signal, onTick, tickMs = 1000, reason = "the model is not answering" } = {}) {
|
|
76
|
+
if (!(ms > 0)) return Promise.resolve(true);
|
|
77
|
+
return new Promise((resolve) => {
|
|
78
|
+
let left = ms;
|
|
79
|
+
let done = false;
|
|
80
|
+
const finish = (waitedAll) => {
|
|
81
|
+
if (done) return;
|
|
82
|
+
done = true;
|
|
83
|
+
clearInterval(interval);
|
|
84
|
+
signal?.removeEventListener?.("abort", onAbort);
|
|
85
|
+
resolve(waitedAll);
|
|
86
|
+
};
|
|
87
|
+
const onAbort = () => finish(false);
|
|
88
|
+
const interval = setInterval(() => {
|
|
89
|
+
left -= tickMs;
|
|
90
|
+
if (left <= 0) {
|
|
91
|
+
onTick?.(0);
|
|
92
|
+
finish(true);
|
|
93
|
+
return;
|
|
94
|
+
}
|
|
95
|
+
onTick?.(left);
|
|
96
|
+
}, tickMs);
|
|
97
|
+
interval.unref?.();
|
|
98
|
+
if (signal) {
|
|
99
|
+
if (signal.aborted) { finish(false); return; }
|
|
100
|
+
signal.addEventListener("abort", onAbort, { once: true });
|
|
101
|
+
}
|
|
102
|
+
onTick?.(left);
|
|
103
|
+
});
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
/**
|
|
107
|
+
* Is this provider failure temporary — "not now" rather than "no"?
|
|
108
|
+
*
|
|
109
|
+
* The line that matters: a 400 with "rate-limited upstream" or "overloaded" in
|
|
110
|
+
* the body is an overloaded backend, not a malformed request. Flint used to
|
|
111
|
+
* treat a 400 as fatal in three places and to treat a 429 as fatal in the main
|
|
112
|
+
* loop, and both end up as "Stopped: API error" with nothing done.
|
|
113
|
+
*
|
|
114
|
+
* What is NOT temporary, and must still stop immediately: a bad key (401/403),
|
|
115
|
+
* an empty account (402), and our own budget refusal. Those need a human, and
|
|
116
|
+
* waiting does not change them.
|
|
117
|
+
*
|
|
118
|
+
* @param {Error & {statusCode?: number, isRateLimit?: boolean}} err
|
|
119
|
+
* @returns {boolean}
|
|
120
|
+
*/
|
|
121
|
+
export function isTemporaryProviderError(err) {
|
|
122
|
+
if (!err) return false;
|
|
123
|
+
if (err.isAuthError || err.isQuotaError || err.isBudgetError) return false;
|
|
124
|
+
if (err.name === "AbortError" || err.isStall) return false;
|
|
125
|
+
if (err.isRateLimit) return true;
|
|
126
|
+
const status = err.statusCode;
|
|
127
|
+
if (status === 408 || status === 409 || status === 429 || status === 529) return true;
|
|
128
|
+
if (typeof status === "number" && status >= 500) return true;
|
|
129
|
+
const body = String(err.message || "").toLowerCase();
|
|
130
|
+
if (status === 400 || status === 422) {
|
|
131
|
+
return /overload|rate[- ]?limit|upstream|try again|temporar|capacity|busy|no available|too many requests/.test(body);
|
|
132
|
+
}
|
|
133
|
+
return false;
|
|
134
|
+
}
|
|
135
|
+
|
|
136
|
+
/** How the failure is named in the countdown, so it is not a bare number. */
|
|
137
|
+
export function describeProviderError(err) {
|
|
138
|
+
if (err?.isRateLimit) return `the provider rate-limited the call (429)`;
|
|
139
|
+
const status = err?.statusCode;
|
|
140
|
+
const body = String(err?.message || "");
|
|
141
|
+
const kind = /overload|rate[- ]?limit|upstream/i.test(body) ? "it reported an overloaded upstream" : "it failed";
|
|
142
|
+
return status ? `the provider answered ${status} and ${kind}` : `the provider ${kind}`;
|
|
143
|
+
}
|