flint-agent 1.14.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.example +108 -0
- package/CHANGELOG.md +55 -0
- package/FEATURES.md +298 -0
- package/LICENSE +21 -0
- package/README.md +435 -0
- package/bin/flint.js +47 -0
- package/config/classifier-prompt.md +218 -0
- package/config/models-curated.json +4 -0
- package/config/providers.json +74 -0
- package/package.json +92 -0
- package/patches/ink+6.8.0.patch +78 -0
- package/profiles/desktop.md +65 -0
- package/profiles/generic.md +20 -0
- package/profiles/marketer.md +20 -0
- package/profiles/profiles.json +34 -0
- package/profiles/ux-reviewer.md +25 -0
- package/src/agent/agent.js +1743 -0
- package/src/agent/auto.js +346 -0
- package/src/agent/backoff.js +143 -0
- package/src/agent/compression.js +310 -0
- package/src/agent/content-resolver.js +180 -0
- package/src/agent/flow-controller.js +309 -0
- package/src/agent/intent-manifest.js +231 -0
- package/src/agent/intent-timeout.js +46 -0
- package/src/agent/intent.js +633 -0
- package/src/agent/knowledge.js +114 -0
- package/src/agent/learning.js +180 -0
- package/src/agent/modes.js +187 -0
- package/src/agent/outcome-ask.js +91 -0
- package/src/agent/project-context.js +76 -0
- package/src/agent/prompt-budget.js +117 -0
- package/src/agent/reflection-extractor.js +140 -0
- package/src/agent/steering.js +86 -0
- package/src/agent/supervisor.js +430 -0
- package/src/agent/swap.js +443 -0
- package/src/agent/system-prompt.js +446 -0
- package/src/agent/time-stamp.js +48 -0
- package/src/agent/tool-guard.js +201 -0
- package/src/agent/toolcall-text.js +162 -0
- package/src/agent/usage.js +297 -0
- package/src/agent/vision.js +94 -0
- package/src/agent/watchdog.js +139 -0
- package/src/agent/workspace-changes.js +177 -0
- package/src/api/address.js +14 -0
- package/src/api/client.js +280 -0
- package/src/api/server.js +535 -0
- package/src/api/stream-pipe.js +113 -0
- package/src/app-state.js +39 -0
- package/src/bootstrap.js +501 -0
- package/src/bus/drain-loop.js +497 -0
- package/src/bus/index.js +270 -0
- package/src/bus/plugins.js +65 -0
- package/src/child-idle.js +14 -0
- package/src/cli.js +118 -0
- package/src/commands/commands.js +1297 -0
- package/src/commands/registry.js +132 -0
- package/src/components/App.js +491 -0
- package/src/components/CarefulMenu.js +145 -0
- package/src/components/HistoryWriter.js +86 -0
- package/src/components/LineInput.js +69 -0
- package/src/components/LiveZone.js +294 -0
- package/src/components/OverlayMenu.js +179 -0
- package/src/components/SystemPanel.js +156 -0
- package/src/components/Table.js +54 -0
- package/src/config.js +249 -0
- package/src/free-models.js +230 -0
- package/src/index.js +1111 -0
- package/src/input-handler.js +13 -0
- package/src/input-text.js +123 -0
- package/src/launcher.js +129 -0
- package/src/logging/api-log.js +95 -0
- package/src/logging/chat-log-follower.js +113 -0
- package/src/logging/chat-log.js +15 -0
- package/src/logging/log-collector.js +182 -0
- package/src/logging/logger.js +112 -0
- package/src/logging/tool-log.js +20 -0
- package/src/mcp-client.js +314 -0
- package/src/memory/conversation-digest.js +113 -0
- package/src/memory/extract-facts.js +98 -0
- package/src/memory/facts.js +181 -0
- package/src/memory/inbox.js +63 -0
- package/src/memory/markdown.js +38 -0
- package/src/memory/patterns.js +185 -0
- package/src/memory/project.js +66 -0
- package/src/memory/reflections.js +74 -0
- package/src/memory/retrieval.js +84 -0
- package/src/memory/rules.js +105 -0
- package/src/memory/session-facts.js +125 -0
- package/src/memory/skills.js +191 -0
- package/src/memory/sqlite-store.js +653 -0
- package/src/memory/store.js +208 -0
- package/src/memory/tools.js +196 -0
- package/src/memory/user-model.js +86 -0
- package/src/message-handler.js +775 -0
- package/src/model-check.js +218 -0
- package/src/plugins/loader.js +120 -0
- package/src/plugins/manager.js +88 -0
- package/src/production-env.js +22 -0
- package/src/profiles.js +42 -0
- package/src/providers/adapters/anthropic.js +270 -0
- package/src/providers/adapters/openai.js +120 -0
- package/src/providers/keys-dpapi.js +41 -0
- package/src/providers/keys-fallback.js +31 -0
- package/src/providers/keys.js +132 -0
- package/src/providers/models.js +154 -0
- package/src/providers/registry.js +56 -0
- package/src/providers/state.js +56 -0
- package/src/registry.js +96 -0
- package/src/restart.js +29 -0
- package/src/sandbox/backend.js +130 -0
- package/src/security/api-auth.js +132 -0
- package/src/security/audit.js +98 -0
- package/src/security/child-policy.js +41 -0
- package/src/security/command-guard.js +173 -0
- package/src/security/content-fence.js +250 -0
- package/src/security/content-validator.js +132 -0
- package/src/security/index.js +143 -0
- package/src/security/network-guard.js +126 -0
- package/src/security/pairing.js +180 -0
- package/src/security/path-guard.js +140 -0
- package/src/security/persona-guard.js +67 -0
- package/src/security/policies.js +452 -0
- package/src/security/safety-constants.js +34 -0
- package/src/security/watchdog.js +107 -0
- package/src/sessions.js +130 -0
- package/src/spend.js +97 -0
- package/src/startup-watchdog.js +59 -0
- package/src/stdio/args.js +71 -0
- package/src/stdio/guard.js +59 -0
- package/src/stdio/protocol.js +167 -0
- package/src/stdio/run.js +106 -0
- package/src/stdio/session.js +180 -0
- package/src/store/agent-slice.js +306 -0
- package/src/store/dataset-slice.js +73 -0
- package/src/store/index.js +22 -0
- package/src/store/process-slice.js +135 -0
- package/src/store/session-slice.js +191 -0
- package/src/store/ui-slice.js +119 -0
- package/src/tasks/db.js +184 -0
- package/src/tasks/queries.js +589 -0
- package/src/tools/agent-tools.js +473 -0
- package/src/tools/checkpoint.js +152 -0
- package/src/tools/command-approvals.js +180 -0
- package/src/tools/dataset.js +50 -0
- package/src/tools/filesystem.js +682 -0
- package/src/tools/inbox-tools.js +48 -0
- package/src/tools/mesh.js +135 -0
- package/src/tools/own-env.js +136 -0
- package/src/tools/permissions.js +681 -0
- package/src/tools/plugin-tools.js +123 -0
- package/src/tools/process-tools.js +595 -0
- package/src/tools/registry.js +307 -0
- package/src/tools/swap-tools.js +72 -0
- package/src/tools/system.js +662 -0
- package/src/tools/tasks.js +532 -0
- package/src/tools/tool-search.js +171 -0
- package/src/ui/header.js +140 -0
- package/src/ui/input-cursor.js +23 -0
- package/src/ui/last-line.js +25 -0
- package/src/ui/line-edit.js +135 -0
- package/src/ui/output.js +399 -0
- package/src/ui/paste-tokens.js +131 -0
- package/src/ui/prompt-attention.js +134 -0
- package/src/ui/render-options.js +13 -0
- package/src/ui/replay.js +94 -0
- package/src/ui/splash.js +49 -0
- package/src/ui/status-level.js +36 -0
- package/src/ui/tool-ledger.js +203 -0
- package/src/ui/window-title.js +150 -0
- package/src/update.js +205 -0
- package/system.md +63 -0
|
@@ -0,0 +1,309 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Flow Controller — unified flow state machine
|
|
3
|
+
*
|
|
4
|
+
* Merges: loop-detector.js + auto-continue logic from drain-loop.js
|
|
5
|
+
* Owns all flow decisions: loop detection, plan state, continue/stop, learning triggers.
|
|
6
|
+
*
|
|
7
|
+
* Used by:
|
|
8
|
+
* - agent.js: checkLoop(), resetFlow()
|
|
9
|
+
* - drain-loop.js: shouldContinue()
|
|
10
|
+
*/
|
|
11
|
+
|
|
12
|
+
import { createLogger } from "../logging/logger.js";
|
|
13
|
+
|
|
14
|
+
const log = createLogger("flow");
|
|
15
|
+
|
|
16
|
+
// --- Configurable thresholds ---
|
|
17
|
+
export function getThresholds() { return { ...THRESHOLDS }; }
|
|
18
|
+
const THRESHOLDS = {
|
|
19
|
+
textRepeat: parseInt(process.env.AGENT_LOOP_TEXT_REPEAT || "5", 10),
|
|
20
|
+
toolRepeat: parseInt(process.env.AGENT_LOOP_TOOL_REPEAT || "5", 10),
|
|
21
|
+
toolRepeatScreenshot: parseInt(process.env.AGENT_LOOP_TOOL_SCREENSHOT || "7", 10),
|
|
22
|
+
toolRepeatCommand: parseInt(process.env.AGENT_LOOP_TOOL_COMMAND || "8", 10),
|
|
23
|
+
desktopObservations: parseInt(process.env.AGENT_LOOP_DESKTOP_OBS || "8", 10),
|
|
24
|
+
coordGridPx: parseInt(process.env.AGENT_LOOP_COORD_GRID || "50", 10),
|
|
25
|
+
maxRetries: 3,
|
|
26
|
+
maxPlanCompletions: 20,
|
|
27
|
+
historySize: 12,
|
|
28
|
+
textHistorySize: 5,
|
|
29
|
+
};
|
|
30
|
+
|
|
31
|
+
const NAV_KEYS = new Set(["tab", "shift+tab", "down", "up", "left", "right", "pagedown", "pageup", "home", "end"]);
|
|
32
|
+
const MAX_REPLANS = 3;
|
|
33
|
+
|
|
34
|
+
// --- State ---
|
|
35
|
+
let _recentTexts = [];
|
|
36
|
+
let _recentToolCalls = [];
|
|
37
|
+
let _consecutiveDesktopObs = 0;
|
|
38
|
+
let _replanCount = 0;
|
|
39
|
+
let _autoRetries = new Map();
|
|
40
|
+
let _planCompletions = 0;
|
|
41
|
+
let _userInterrupted = false; // set when user sends a message during auto-continue
|
|
42
|
+
|
|
43
|
+
// --- Loop Detection (from loop-detector.js) ---
|
|
44
|
+
|
|
45
|
+
function _fuzzyToolSig(name, args) {
|
|
46
|
+
if (!args || typeof args !== "object") return name;
|
|
47
|
+
if (name.includes("wait_stable") || name.includes("wait_change") || name === "run_background_command") {
|
|
48
|
+
return name + ":skip:" + Date.now();
|
|
49
|
+
}
|
|
50
|
+
if (name.includes("key") && args.keys && NAV_KEYS.has(String(args.keys).toLowerCase())) {
|
|
51
|
+
return name + ":nav:" + Date.now();
|
|
52
|
+
}
|
|
53
|
+
const key = {};
|
|
54
|
+
for (const [k, v] of Object.entries(args)) {
|
|
55
|
+
if (k === "x" || k === "y") {
|
|
56
|
+
key[k] = Math.round(Number(v) / THRESHOLDS.coordGridPx) * THRESHOLDS.coordGridPx;
|
|
57
|
+
} else if (k === "text" || k === "command") {
|
|
58
|
+
key[k] = String(v).slice(0, 120);
|
|
59
|
+
} else if (k === "desktop_id" || k === "action" || k === "keys" || k === "cell") {
|
|
60
|
+
key[k] = v;
|
|
61
|
+
} else {
|
|
62
|
+
// Bug fix 2026-04-12: default was slice(0, 40), which collapsed all
|
|
63
|
+
// nested paths like /tmp/flint-bench/l4-deep/003/level1/level2/... into
|
|
64
|
+
// a single signature and caused the loop detector to hard-stop legitimate
|
|
65
|
+
// sequential reads across 6 different deep files. 200 chars covers all
|
|
66
|
+
// realistic paths and file identifiers without meaningful memory cost.
|
|
67
|
+
key[k] = typeof v === "string" ? v.slice(0, 200) : v;
|
|
68
|
+
}
|
|
69
|
+
}
|
|
70
|
+
return name + ":" + JSON.stringify(key);
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
function _buildLoopResult(type, count, detail) {
|
|
74
|
+
_replanCount++;
|
|
75
|
+
if (_replanCount > MAX_REPLANS) {
|
|
76
|
+
return { type, count, action: "stop", message: `[Loop: ${detail} repeated ${count}x. Re-plan failed. Stopped.]` };
|
|
77
|
+
}
|
|
78
|
+
return { type, count, action: "replan", message: `[Loop: ${detail} repeated ${count}x. STOP. Create NEW plan with DIFFERENT strategy.]` };
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
export function checkTextLoop(text) {
|
|
82
|
+
const clean = (text || "").trim().toLowerCase();
|
|
83
|
+
if (clean.length === 0 || clean.length >= 100) return null;
|
|
84
|
+
_recentTexts.push(clean);
|
|
85
|
+
if (_recentTexts.length > THRESHOLDS.textHistorySize) _recentTexts.shift();
|
|
86
|
+
if (_recentTexts.length >= THRESHOLDS.textRepeat) {
|
|
87
|
+
const last = _recentTexts[_recentTexts.length - 1];
|
|
88
|
+
const count = _recentTexts.filter(t => t === last).length;
|
|
89
|
+
if (count >= THRESHOLDS.textRepeat) {
|
|
90
|
+
log.warn("text-loop", { text: clean.slice(0, 60), count });
|
|
91
|
+
return _buildLoopResult("text", count, text);
|
|
92
|
+
}
|
|
93
|
+
}
|
|
94
|
+
return null;
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
export function checkToolLoop(toolName, args) {
|
|
98
|
+
const sig = _fuzzyToolSig(toolName, args);
|
|
99
|
+
_recentToolCalls.push(sig);
|
|
100
|
+
if (_recentToolCalls.length > THRESHOLDS.historySize) _recentToolCalls.shift();
|
|
101
|
+
const count = _recentToolCalls.filter(t => t === sig).length;
|
|
102
|
+
const threshold = (toolName.includes("screenshot") || toolName.includes("look"))
|
|
103
|
+
? THRESHOLDS.toolRepeatScreenshot
|
|
104
|
+
: (toolName === "run_command" || toolName.includes("shell"))
|
|
105
|
+
? THRESHOLDS.toolRepeatCommand
|
|
106
|
+
: THRESHOLDS.toolRepeat;
|
|
107
|
+
if (count >= threshold) {
|
|
108
|
+
log.warn("tool-loop", { tool: toolName, count, sig });
|
|
109
|
+
return _buildLoopResult("tool", count, toolName);
|
|
110
|
+
}
|
|
111
|
+
return null;
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
export function checkDesktopLoop(toolCalls) {
|
|
115
|
+
const hasAction = toolCalls.some(tc => {
|
|
116
|
+
const n = tc.function?.name || tc.name || "";
|
|
117
|
+
if (!n.includes("desktop") && !n.includes("screenbox")) return true;
|
|
118
|
+
return n.includes("type") || n.includes("batch") || n.includes("shell")
|
|
119
|
+
|| n.includes("chrome") || n.includes("manage") || n.includes("file");
|
|
120
|
+
});
|
|
121
|
+
if (!hasAction) _consecutiveDesktopObs++;
|
|
122
|
+
else _consecutiveDesktopObs = 0;
|
|
123
|
+
if (_consecutiveDesktopObs >= THRESHOLDS.desktopObservations) {
|
|
124
|
+
log.warn("desktop-loop", { count: _consecutiveDesktopObs });
|
|
125
|
+
const result = _buildLoopResult("desktop", _consecutiveDesktopObs, "desktop observations");
|
|
126
|
+
_consecutiveDesktopObs = 0;
|
|
127
|
+
return result;
|
|
128
|
+
}
|
|
129
|
+
return null;
|
|
130
|
+
}
|
|
131
|
+
|
|
132
|
+
export function resetDesktopOnMeaningfulText(text) {
|
|
133
|
+
if ((text || "").toLowerCase().match(/\b(done|completed|finished|saved|created|failed|cannot|impossible)\b/)) {
|
|
134
|
+
_consecutiveDesktopObs = 0;
|
|
135
|
+
}
|
|
136
|
+
}
|
|
137
|
+
|
|
138
|
+
// --- What the provider's refusals mean ---
|
|
139
|
+
//
|
|
140
|
+
// Shared on purpose. There are two autonomous loops, /auto in agent/auto.js
|
|
141
|
+
// and the bus in bus/drain-loop.js, and they are accepted
|
|
142
|
+
// separately because they are different loops. What counts as a refusal must
|
|
143
|
+
// not be one of the things they can disagree about: a reason that is fatal on
|
|
144
|
+
// one door and survivable on the other is a hole with a door in front of it.
|
|
145
|
+
//
|
|
146
|
+
// NOTE for anyone moving these: the rate-limit COUNTER must not live in this
|
|
147
|
+
// module. resetFlow() runs at the top of every agent invocation
|
|
148
|
+
// (agent.js:248), so anything kept here is wiped once per turn and a counter
|
|
149
|
+
// of consecutive turns would never reach two.
|
|
150
|
+
|
|
151
|
+
/** The provider is not going to serve the next turn either. A human is needed. */
|
|
152
|
+
// "empty": the model answered with nothing three times running. Another
|
|
153
|
+
// autonomous turn would only buy more empty, billed answers.
|
|
154
|
+
export const FATAL_PROVIDER_REASONS = new Set(["auth", "quota", "error", "empty"]);
|
|
155
|
+
|
|
156
|
+
/** How many 429s in a row before a run treats the cap as a closed door. */
|
|
157
|
+
export const MAX_CONSECUTIVE_RATE_LIMITS = 3;
|
|
158
|
+
|
|
159
|
+
/** How long to wait out a 429 when the provider did not send retry-after. */
|
|
160
|
+
export const RATE_LIMIT_WAIT_MS = 20000;
|
|
161
|
+
|
|
162
|
+
/** One line per reason, for the human who has to decide what to do about it. */
|
|
163
|
+
export const STOP_REASON_TEXT = {
|
|
164
|
+
auth: "the provider rejected the API key",
|
|
165
|
+
quota: "the account is out of credits",
|
|
166
|
+
error: "the provider failed three calls in a row",
|
|
167
|
+
"rate-limit": "the provider kept rate limiting the run",
|
|
168
|
+
empty: "the model answered with nothing three times in a row",
|
|
169
|
+
};
|
|
170
|
+
|
|
171
|
+
// --- Auto-Continue / Plan State (from drain-loop.js) ---
|
|
172
|
+
|
|
173
|
+
/**
|
|
174
|
+
* Decide whether to continue after agent returns.
|
|
175
|
+
* @param {object} opts - { stopReason, plan, isApiScoped, apiGoalId, apiVerified }
|
|
176
|
+
* @returns {null | {action: "stop"|"continue"|"verify", message?: string, prompt?: string}}
|
|
177
|
+
*/
|
|
178
|
+
export function shouldContinue({ stopReason, plan, isApiScoped, apiGoalId }) {
|
|
179
|
+
// User interrupt: any non-self-continue message during auto-continue pauses the plan.
|
|
180
|
+
// The user typed something → they want control. Don't fight them with auto-Continue.
|
|
181
|
+
// Clear via resetFlow() (/new) or setUserInterrupt(false) (/resume).
|
|
182
|
+
if (_userInterrupted) {
|
|
183
|
+
log.info("flow: user interrupted auto-continue, pausing");
|
|
184
|
+
return { action: "stop", reason: "user_interrupt" };
|
|
185
|
+
}
|
|
186
|
+
|
|
187
|
+
const hasPending = plan && plan.tasks?.some(t => t.status === "pending" || t.status === "in_progress");
|
|
188
|
+
|
|
189
|
+
// No plan + done = simple task complete
|
|
190
|
+
if (!plan) {
|
|
191
|
+
log.info("flow: no plan + %s = stop", stopReason);
|
|
192
|
+
return { action: "stop" };
|
|
193
|
+
}
|
|
194
|
+
|
|
195
|
+
// Plan exists, all tasks done
|
|
196
|
+
if (!hasPending) {
|
|
197
|
+
_autoRetries.clear();
|
|
198
|
+
_planCompletions++;
|
|
199
|
+
|
|
200
|
+
// API scope: goal done = stop. Verification handled by agent loop's self-verify gate.
|
|
201
|
+
if (isApiScoped && plan?.goalId === apiGoalId) {
|
|
202
|
+
log.info("flow: API goal complete, stopping", { goalId: plan.goalId });
|
|
203
|
+
return { action: "stop", goalComplete: true };
|
|
204
|
+
}
|
|
205
|
+
|
|
206
|
+
// Safety limit
|
|
207
|
+
if (_planCompletions >= THRESHOLDS.maxPlanCompletions) {
|
|
208
|
+
_planCompletions = 0;
|
|
209
|
+
log.warn("flow: max plan completions reached");
|
|
210
|
+
return { action: "stop" };
|
|
211
|
+
}
|
|
212
|
+
|
|
213
|
+
// TUI: check other goals
|
|
214
|
+
return {
|
|
215
|
+
action: "continue",
|
|
216
|
+
prompt: "Your plan is complete. Check if your original task has more work. If everything is truly done, write a summary and stop.",
|
|
217
|
+
};
|
|
218
|
+
}
|
|
219
|
+
|
|
220
|
+
// Plan has pending tasks — find next
|
|
221
|
+
const pending = plan.tasks.filter(t => t.status === "pending" || t.status === "in_progress");
|
|
222
|
+
let nextTask = null;
|
|
223
|
+
for (const t of pending) {
|
|
224
|
+
const retries = _autoRetries.get(t.id) || 0;
|
|
225
|
+
if (retries < THRESHOLDS.maxRetries) {
|
|
226
|
+
nextTask = t;
|
|
227
|
+
break;
|
|
228
|
+
}
|
|
229
|
+
}
|
|
230
|
+
|
|
231
|
+
if (!nextTask) {
|
|
232
|
+
_autoRetries.clear();
|
|
233
|
+
return {
|
|
234
|
+
action: "continue",
|
|
235
|
+
prompt: "All remaining tasks exhausted retries. Mark them as skipped and move to the next phase.",
|
|
236
|
+
};
|
|
237
|
+
}
|
|
238
|
+
|
|
239
|
+
const retries = _autoRetries.get(nextTask.id) || 0;
|
|
240
|
+
_autoRetries.set(nextTask.id, retries + 1);
|
|
241
|
+
|
|
242
|
+
const retryNote = retries > 0
|
|
243
|
+
? ` (retry ${retries}/${THRESHOLDS.maxRetries} — try a DIFFERENT approach. If blocked, add an alternative task via add_task and skip this one.)`
|
|
244
|
+
: "";
|
|
245
|
+
return {
|
|
246
|
+
action: "continue",
|
|
247
|
+
prompt: `Continue: "${nextTask.title}" [#${nextTask.id}]${retryNote}. Use update_task to mark progress.`,
|
|
248
|
+
};
|
|
249
|
+
}
|
|
250
|
+
|
|
251
|
+
// --- Learning Trigger ---
|
|
252
|
+
|
|
253
|
+
/**
|
|
254
|
+
* Check if current state is a learning opportunity.
|
|
255
|
+
* Called by Verification Gate after successful task completion.
|
|
256
|
+
* @returns {boolean}
|
|
257
|
+
*/
|
|
258
|
+
export function isLearningOpportunity({ plan, toolCallCount }) {
|
|
259
|
+
// Multi-step task completed successfully = learning opportunity
|
|
260
|
+
if (plan && plan.tasks?.length >= 2) return true;
|
|
261
|
+
// Many tool calls = complex task worth learning from
|
|
262
|
+
if (toolCallCount > 5) return true;
|
|
263
|
+
return false;
|
|
264
|
+
}
|
|
265
|
+
|
|
266
|
+
// --- Reset ---
|
|
267
|
+
|
|
268
|
+
/**
|
|
269
|
+
* Set/clear user interrupt flag. Called from drain-loop when a user
|
|
270
|
+
* (non-self-continue) message arrives during an active plan.
|
|
271
|
+
*/
|
|
272
|
+
export function setUserInterrupt(interrupted) {
|
|
273
|
+
if (_userInterrupted !== interrupted) {
|
|
274
|
+
log.info("flow: user-interrupt", { interrupted });
|
|
275
|
+
}
|
|
276
|
+
_userInterrupted = interrupted;
|
|
277
|
+
}
|
|
278
|
+
|
|
279
|
+
export function isUserInterrupted() {
|
|
280
|
+
return _userInterrupted;
|
|
281
|
+
}
|
|
282
|
+
|
|
283
|
+
/**
|
|
284
|
+
* Per-turn reset: the loop detectors only. Called at the top of every agent
|
|
285
|
+
* invocation.
|
|
286
|
+
*
|
|
287
|
+
* It used to be resetFlow(), which also cleared the run-level state below.
|
|
288
|
+
* That ran once per turn, so the user interrupt the drain loop had just set
|
|
289
|
+
* was gone before shouldContinue could read it, the per-task retry cap never
|
|
290
|
+
* got past one, and the plan-completion cap never got past one either.
|
|
291
|
+
*/
|
|
292
|
+
export function resetTurn() {
|
|
293
|
+
_recentTexts = [];
|
|
294
|
+
_recentToolCalls = [];
|
|
295
|
+
_consecutiveDesktopObs = 0;
|
|
296
|
+
_replanCount = 0;
|
|
297
|
+
}
|
|
298
|
+
|
|
299
|
+
/**
|
|
300
|
+
* Full reset: the loop detectors plus the run-level state (retries, plan
|
|
301
|
+
* completions, user interrupt). For the start of a new run and /new.
|
|
302
|
+
*/
|
|
303
|
+
export function resetFlow() {
|
|
304
|
+
resetTurn();
|
|
305
|
+
_autoRetries.clear();
|
|
306
|
+
_planCompletions = 0;
|
|
307
|
+
_userInterrupted = false;
|
|
308
|
+
}
|
|
309
|
+
|
|
@@ -0,0 +1,231 @@
|
|
|
1
|
+
// Intent manifest — catalog of intent classes.
|
|
2
|
+
//
|
|
3
|
+
// Each intent describes a *class of request* — how many steps it usually needs,
|
|
4
|
+
// what the output looks like, and whether tools are required at all. Tool
|
|
5
|
+
// selection itself is delegated to the classifier at runtime, which picks
|
|
6
|
+
// concrete tools from the live registry. This keeps the manifest agnostic to
|
|
7
|
+
// specific tool names — adding a new tool in registry does not require editing
|
|
8
|
+
// this file.
|
|
9
|
+
//
|
|
10
|
+
// Fields:
|
|
11
|
+
// description — one-line hint for the classifier prompt
|
|
12
|
+
// needsTools — false = pure text/knowledge answer (empty tool list enforced)
|
|
13
|
+
// true = classifier will pick tools from the live registry
|
|
14
|
+
// max_steps — hard cap on iterations for this intent
|
|
15
|
+
// expected — shape of final output: "text" | "tool_call" | "tool_and_summary" | "mixed"
|
|
16
|
+
// changes : does a request of this class end with something different
|
|
17
|
+
// on disk? "yes" when that is the whole point of the class,
|
|
18
|
+
// "no" when it is a question, "maybe" when the class covers
|
|
19
|
+
// both. The no-change check uses it to decide whether a turn that changed
|
|
20
|
+
// nothing owes the operator an explanation. It is a property
|
|
21
|
+
// of the CLASS, written down once where the classes are
|
|
22
|
+
// described and reviewable there, not a guess about the
|
|
23
|
+
// wording of any one request.
|
|
24
|
+
|
|
25
|
+
export const INTENTS = {
|
|
26
|
+
// --- Pure text: model answers from own knowledge, no tools at all ---
|
|
27
|
+
creative_text: {
|
|
28
|
+
description: "Generate text from scratch and reply in chat: poems, stories, summaries, explanations, jokes, translations, code snippets (no execution). DO NOT pick this if the user asks to SAVE/WRITE the result to a file or path — that is file_write.",
|
|
29
|
+
needsTools: false,
|
|
30
|
+
max_steps: 1,
|
|
31
|
+
changes: "no",
|
|
32
|
+
expected: "text",
|
|
33
|
+
},
|
|
34
|
+
|
|
35
|
+
knowledge_qa: {
|
|
36
|
+
description: "Factual question, definition, or explanation that the model can answer directly without tools",
|
|
37
|
+
needsTools: false,
|
|
38
|
+
max_steps: 1,
|
|
39
|
+
changes: "no",
|
|
40
|
+
expected: "text",
|
|
41
|
+
},
|
|
42
|
+
|
|
43
|
+
reasoning: {
|
|
44
|
+
description: "Math, logic, puzzles, analysis of data provided in the message itself",
|
|
45
|
+
needsTools: false,
|
|
46
|
+
max_steps: 2,
|
|
47
|
+
changes: "no",
|
|
48
|
+
expected: "text",
|
|
49
|
+
},
|
|
50
|
+
|
|
51
|
+
chat: {
|
|
52
|
+
description: "Small talk, greetings, acknowledgements, clarification questions, meta discussion about the conversation",
|
|
53
|
+
needsTools: false,
|
|
54
|
+
max_steps: 1,
|
|
55
|
+
changes: "no",
|
|
56
|
+
expected: "text",
|
|
57
|
+
},
|
|
58
|
+
|
|
59
|
+
// --- Tool-backed: classifier picks concrete tools from live registry ---
|
|
60
|
+
shell_command: {
|
|
61
|
+
description: "Run a single shell command and report output",
|
|
62
|
+
needsTools: true,
|
|
63
|
+
// Was 5; bumped to 12 on 2026-04-21. The 5-step cap was too tight for
|
|
64
|
+
// any shell work involving curl + JSON escape or HTTP-payload work —
|
|
65
|
+
// observed during the readiness-test mirror experiment where the agent
|
|
66
|
+
// burned every iteration on quoting attempts before giving up.
|
|
67
|
+
max_steps: 12,
|
|
68
|
+
changes: "maybe",
|
|
69
|
+
expected: "tool_and_summary",
|
|
70
|
+
tool_pattern: /^(run_command|run_background_command|peek_process|list_processes|kill_process|read_file)$/,
|
|
71
|
+
},
|
|
72
|
+
|
|
73
|
+
shell_multi: {
|
|
74
|
+
description: "Run several shell commands to accomplish a goal (install, diagnose, pipe-chain)",
|
|
75
|
+
needsTools: true,
|
|
76
|
+
// Was 8; bumped to 16 on 2026-04-21 for consistency with shell_command.
|
|
77
|
+
// Multi-shell pipelines on Windows (bash-on-windows escape quirks)
|
|
78
|
+
// often need several extra diagnostic steps.
|
|
79
|
+
max_steps: 16,
|
|
80
|
+
changes: "maybe",
|
|
81
|
+
expected: "tool_and_summary",
|
|
82
|
+
tool_pattern: /^(run_command|run_background_command|peek_process|list_processes|kill_process|read_file|write_file|edit_file)$/,
|
|
83
|
+
},
|
|
84
|
+
|
|
85
|
+
file_read: {
|
|
86
|
+
description: "Read the contents of a file or list a directory",
|
|
87
|
+
needsTools: true,
|
|
88
|
+
max_steps: 3,
|
|
89
|
+
changes: "no",
|
|
90
|
+
expected: "tool_and_summary",
|
|
91
|
+
},
|
|
92
|
+
|
|
93
|
+
file_write: {
|
|
94
|
+
description: "Create or overwrite a file with specific content. ALWAYS pick this when the user asks to SAVE/WRITE/STORE generated text to a file or path (e.g. 'write a tweet and save to /tmp/x.txt', 'generate an outline and save it to ...'). The text-generation part does NOT make it creative_text once a file destination is named.",
|
|
95
|
+
needsTools: true,
|
|
96
|
+
max_steps: 3,
|
|
97
|
+
changes: "yes",
|
|
98
|
+
expected: "tool_and_summary",
|
|
99
|
+
},
|
|
100
|
+
|
|
101
|
+
file_edit: {
|
|
102
|
+
description: "Modify an existing file: read it, change parts, write it back",
|
|
103
|
+
needsTools: true,
|
|
104
|
+
max_steps: 5,
|
|
105
|
+
changes: "yes",
|
|
106
|
+
expected: "tool_and_summary",
|
|
107
|
+
},
|
|
108
|
+
|
|
109
|
+
file_manage: {
|
|
110
|
+
description: "Copy, move, delete, search across files",
|
|
111
|
+
needsTools: true,
|
|
112
|
+
max_steps: 6,
|
|
113
|
+
changes: "yes",
|
|
114
|
+
expected: "tool_and_summary",
|
|
115
|
+
},
|
|
116
|
+
|
|
117
|
+
web_fetch: {
|
|
118
|
+
description: "Fetch a specific URL and report/parse its content",
|
|
119
|
+
needsTools: true,
|
|
120
|
+
max_steps: 3,
|
|
121
|
+
changes: "no",
|
|
122
|
+
expected: "tool_and_summary",
|
|
123
|
+
// Widen: when user asks to read a page, include browser tools too —
|
|
124
|
+
// search-only gets stuck on JS-heavy sites (observed: linkedin/twitter/github).
|
|
125
|
+
tool_pattern: /^(web_fetch|web_search|mcp_browser_)/,
|
|
126
|
+
},
|
|
127
|
+
|
|
128
|
+
web_search: {
|
|
129
|
+
description: "Search the web for information on a topic, including 'open <site>', 'navigate to', 'search on <site>' — these are tool-backed browsing, not just a search engine query",
|
|
130
|
+
needsTools: true,
|
|
131
|
+
max_steps: 8,
|
|
132
|
+
changes: "no",
|
|
133
|
+
expected: "tool_and_summary",
|
|
134
|
+
// Same widening, so "go to LinkedIn" no longer gets only web_search.
|
|
135
|
+
tool_pattern: /^(web_search|web_fetch|mcp_browser_)/,
|
|
136
|
+
},
|
|
137
|
+
|
|
138
|
+
task_create: {
|
|
139
|
+
description: "Create a new plan with tasks, or add a task to an existing plan",
|
|
140
|
+
needsTools: true,
|
|
141
|
+
max_steps: 3,
|
|
142
|
+
changes: "maybe",
|
|
143
|
+
expected: "tool_and_summary",
|
|
144
|
+
},
|
|
145
|
+
|
|
146
|
+
task_update: {
|
|
147
|
+
description: "Change the status of a task (done, skipped, in_progress), add notes, mark progress",
|
|
148
|
+
needsTools: true,
|
|
149
|
+
max_steps: 3,
|
|
150
|
+
changes: "maybe",
|
|
151
|
+
expected: "tool_and_summary",
|
|
152
|
+
},
|
|
153
|
+
|
|
154
|
+
task_view: {
|
|
155
|
+
description: "Show the current plan, list tasks, check progress",
|
|
156
|
+
needsTools: true,
|
|
157
|
+
max_steps: 2,
|
|
158
|
+
changes: "no",
|
|
159
|
+
expected: "tool_and_summary",
|
|
160
|
+
},
|
|
161
|
+
|
|
162
|
+
memory_write: {
|
|
163
|
+
description: "Remember something for later: save a fact, note, or preference",
|
|
164
|
+
needsTools: true,
|
|
165
|
+
max_steps: 3,
|
|
166
|
+
changes: "maybe",
|
|
167
|
+
expected: "tool_and_summary",
|
|
168
|
+
},
|
|
169
|
+
|
|
170
|
+
memory_read: {
|
|
171
|
+
description: "Recall something from memory: search past notes, find saved facts",
|
|
172
|
+
needsTools: true,
|
|
173
|
+
max_steps: 4,
|
|
174
|
+
changes: "no",
|
|
175
|
+
expected: "tool_and_summary",
|
|
176
|
+
},
|
|
177
|
+
|
|
178
|
+
memory_manage: {
|
|
179
|
+
description: "Write AND read memory in the same task: save then verify, delete then confirm, or similar multi-step memory flows",
|
|
180
|
+
needsTools: true,
|
|
181
|
+
max_steps: 12,
|
|
182
|
+
changes: "maybe",
|
|
183
|
+
expected: "tool_and_summary",
|
|
184
|
+
},
|
|
185
|
+
|
|
186
|
+
agent_spawn: {
|
|
187
|
+
description: "Delegate work to a child agent, ask another agent, or wait for agents to finish",
|
|
188
|
+
needsTools: true,
|
|
189
|
+
max_steps: 5,
|
|
190
|
+
changes: "maybe",
|
|
191
|
+
expected: "tool_and_summary",
|
|
192
|
+
},
|
|
193
|
+
|
|
194
|
+
desktop: {
|
|
195
|
+
description: "Interact with a remote desktop: screenshot, click, type, open apps, browse, run desktop shell",
|
|
196
|
+
needsTools: true,
|
|
197
|
+
max_steps: 20,
|
|
198
|
+
changes: "maybe",
|
|
199
|
+
expected: "tool_and_summary",
|
|
200
|
+
// Desktop + browser both can handle screenshots/clicks; expose both families.
|
|
201
|
+
tool_pattern: /^(mcp_screenbox_|mcp_browser_|view_image)/,
|
|
202
|
+
},
|
|
203
|
+
|
|
204
|
+
system_info: {
|
|
205
|
+
description: "Report info about Flint itself: balance, models, providers, active plan, think out loud",
|
|
206
|
+
needsTools: true,
|
|
207
|
+
max_steps: 2,
|
|
208
|
+
changes: "no",
|
|
209
|
+
expected: "tool_and_summary",
|
|
210
|
+
},
|
|
211
|
+
|
|
212
|
+
complex_multi: {
|
|
213
|
+
description: "Multi-step task combining several categories (web + file + shell, plan + execute, etc.) — requires full tool surface",
|
|
214
|
+
needsTools: true,
|
|
215
|
+
max_steps: 30,
|
|
216
|
+
changes: "maybe",
|
|
217
|
+
expected: "mixed",
|
|
218
|
+
},
|
|
219
|
+
};
|
|
220
|
+
|
|
221
|
+
// Compact list for classifier prompt — just name + description
|
|
222
|
+
export function formatIntentCatalog() {
|
|
223
|
+
return Object.entries(INTENTS)
|
|
224
|
+
.map(([name, spec]) => ` ${name}: ${spec.description}`)
|
|
225
|
+
.join("\n");
|
|
226
|
+
}
|
|
227
|
+
|
|
228
|
+
// Resolve intent name → full spec with defaults
|
|
229
|
+
export function resolveIntent(name) {
|
|
230
|
+
return INTENTS[name] || INTENTS.complex_multi;
|
|
231
|
+
}
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
// The classifier's per-attempt budget.
|
|
2
|
+
//
|
|
3
|
+
// Item 10: operator-visibility.test.js failed on a loaded machine, and one of
|
|
4
|
+
// the causes was a hard 15 s wall-clock deadline inside the classifier call that
|
|
5
|
+
// no caller could see or control. Under load that deadline fired before the
|
|
6
|
+
// mocked response was processed, and `classifyIntent` — which catches its own
|
|
7
|
+
// errors — returned the all-tools fallback. Nothing said so, which is how a test
|
|
8
|
+
// came to depend on how busy the machine was: the code under the assertion
|
|
9
|
+
// quietly changed its mind, and a fallback is indistinguishable from a real
|
|
10
|
+
// classification in the output.
|
|
11
|
+
//
|
|
12
|
+
// Its own module rather than a line in intent.js, because intent.js pulls in
|
|
13
|
+
// the config, the API client and the tool manifest: importing it to test one
|
|
14
|
+
// arithmetic decision means paying for a network client and an env-dependent
|
|
15
|
+
// config load, and the failure mode of that is a test that cannot run in CI at
|
|
16
|
+
// all. This is a pure function of the environment and is tested as one.
|
|
17
|
+
//
|
|
18
|
+
// The project rule is no silent fallbacks — a required value that is absent
|
|
19
|
+
// throws at config load rather than becoming a default. An override gets the
|
|
20
|
+
// same treatment. A caller that sets INTENT_TIMEOUT_MS=0 and silently receives
|
|
21
|
+
// 15 s has the original problem in a new place: a budget that is not what was
|
|
22
|
+
// asked for, with nothing said about it.
|
|
23
|
+
|
|
24
|
+
/** The production default, in ms. */
|
|
25
|
+
export const DEFAULT_INTENT_TIMEOUT_MS = 15000;
|
|
26
|
+
|
|
27
|
+
/**
|
|
28
|
+
* The per-attempt budget for a classifier call.
|
|
29
|
+
*
|
|
30
|
+
* @returns {number} milliseconds
|
|
31
|
+
* @throws {Error} if INTENT_TIMEOUT_MS is set to something that is not a
|
|
32
|
+
* positive number of milliseconds
|
|
33
|
+
*/
|
|
34
|
+
export function intentTimeoutMs() {
|
|
35
|
+
const raw = process.env.INTENT_TIMEOUT_MS;
|
|
36
|
+
// Unset, or set to empty: the documented default. An empty value in a .env is
|
|
37
|
+
// a real thing to find and it means "not set", not "zero".
|
|
38
|
+
if (!raw) return DEFAULT_INTENT_TIMEOUT_MS;
|
|
39
|
+
const n = Number(raw);
|
|
40
|
+
if (!Number.isFinite(n) || n <= 0) {
|
|
41
|
+
throw new Error(
|
|
42
|
+
`INTENT_TIMEOUT_MS must be a positive number of milliseconds, got: ${raw}`,
|
|
43
|
+
);
|
|
44
|
+
}
|
|
45
|
+
return n;
|
|
46
|
+
}
|