flint-agent 1.14.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (171) hide show
  1. package/.env.example +108 -0
  2. package/CHANGELOG.md +55 -0
  3. package/FEATURES.md +298 -0
  4. package/LICENSE +21 -0
  5. package/README.md +435 -0
  6. package/bin/flint.js +47 -0
  7. package/config/classifier-prompt.md +218 -0
  8. package/config/models-curated.json +4 -0
  9. package/config/providers.json +74 -0
  10. package/package.json +92 -0
  11. package/patches/ink+6.8.0.patch +78 -0
  12. package/profiles/desktop.md +65 -0
  13. package/profiles/generic.md +20 -0
  14. package/profiles/marketer.md +20 -0
  15. package/profiles/profiles.json +34 -0
  16. package/profiles/ux-reviewer.md +25 -0
  17. package/src/agent/agent.js +1743 -0
  18. package/src/agent/auto.js +346 -0
  19. package/src/agent/backoff.js +143 -0
  20. package/src/agent/compression.js +310 -0
  21. package/src/agent/content-resolver.js +180 -0
  22. package/src/agent/flow-controller.js +309 -0
  23. package/src/agent/intent-manifest.js +231 -0
  24. package/src/agent/intent-timeout.js +46 -0
  25. package/src/agent/intent.js +633 -0
  26. package/src/agent/knowledge.js +114 -0
  27. package/src/agent/learning.js +180 -0
  28. package/src/agent/modes.js +187 -0
  29. package/src/agent/outcome-ask.js +91 -0
  30. package/src/agent/project-context.js +76 -0
  31. package/src/agent/prompt-budget.js +117 -0
  32. package/src/agent/reflection-extractor.js +140 -0
  33. package/src/agent/steering.js +86 -0
  34. package/src/agent/supervisor.js +430 -0
  35. package/src/agent/swap.js +443 -0
  36. package/src/agent/system-prompt.js +446 -0
  37. package/src/agent/time-stamp.js +48 -0
  38. package/src/agent/tool-guard.js +201 -0
  39. package/src/agent/toolcall-text.js +162 -0
  40. package/src/agent/usage.js +297 -0
  41. package/src/agent/vision.js +94 -0
  42. package/src/agent/watchdog.js +139 -0
  43. package/src/agent/workspace-changes.js +177 -0
  44. package/src/api/address.js +14 -0
  45. package/src/api/client.js +280 -0
  46. package/src/api/server.js +535 -0
  47. package/src/api/stream-pipe.js +113 -0
  48. package/src/app-state.js +39 -0
  49. package/src/bootstrap.js +501 -0
  50. package/src/bus/drain-loop.js +497 -0
  51. package/src/bus/index.js +270 -0
  52. package/src/bus/plugins.js +65 -0
  53. package/src/child-idle.js +14 -0
  54. package/src/cli.js +118 -0
  55. package/src/commands/commands.js +1297 -0
  56. package/src/commands/registry.js +132 -0
  57. package/src/components/App.js +491 -0
  58. package/src/components/CarefulMenu.js +145 -0
  59. package/src/components/HistoryWriter.js +86 -0
  60. package/src/components/LineInput.js +69 -0
  61. package/src/components/LiveZone.js +294 -0
  62. package/src/components/OverlayMenu.js +179 -0
  63. package/src/components/SystemPanel.js +156 -0
  64. package/src/components/Table.js +54 -0
  65. package/src/config.js +249 -0
  66. package/src/free-models.js +230 -0
  67. package/src/index.js +1111 -0
  68. package/src/input-handler.js +13 -0
  69. package/src/input-text.js +123 -0
  70. package/src/launcher.js +129 -0
  71. package/src/logging/api-log.js +95 -0
  72. package/src/logging/chat-log-follower.js +113 -0
  73. package/src/logging/chat-log.js +15 -0
  74. package/src/logging/log-collector.js +182 -0
  75. package/src/logging/logger.js +112 -0
  76. package/src/logging/tool-log.js +20 -0
  77. package/src/mcp-client.js +314 -0
  78. package/src/memory/conversation-digest.js +113 -0
  79. package/src/memory/extract-facts.js +98 -0
  80. package/src/memory/facts.js +181 -0
  81. package/src/memory/inbox.js +63 -0
  82. package/src/memory/markdown.js +38 -0
  83. package/src/memory/patterns.js +185 -0
  84. package/src/memory/project.js +66 -0
  85. package/src/memory/reflections.js +74 -0
  86. package/src/memory/retrieval.js +84 -0
  87. package/src/memory/rules.js +105 -0
  88. package/src/memory/session-facts.js +125 -0
  89. package/src/memory/skills.js +191 -0
  90. package/src/memory/sqlite-store.js +653 -0
  91. package/src/memory/store.js +208 -0
  92. package/src/memory/tools.js +196 -0
  93. package/src/memory/user-model.js +86 -0
  94. package/src/message-handler.js +775 -0
  95. package/src/model-check.js +218 -0
  96. package/src/plugins/loader.js +120 -0
  97. package/src/plugins/manager.js +88 -0
  98. package/src/production-env.js +22 -0
  99. package/src/profiles.js +42 -0
  100. package/src/providers/adapters/anthropic.js +270 -0
  101. package/src/providers/adapters/openai.js +120 -0
  102. package/src/providers/keys-dpapi.js +41 -0
  103. package/src/providers/keys-fallback.js +31 -0
  104. package/src/providers/keys.js +132 -0
  105. package/src/providers/models.js +154 -0
  106. package/src/providers/registry.js +56 -0
  107. package/src/providers/state.js +56 -0
  108. package/src/registry.js +96 -0
  109. package/src/restart.js +29 -0
  110. package/src/sandbox/backend.js +130 -0
  111. package/src/security/api-auth.js +132 -0
  112. package/src/security/audit.js +98 -0
  113. package/src/security/child-policy.js +41 -0
  114. package/src/security/command-guard.js +173 -0
  115. package/src/security/content-fence.js +250 -0
  116. package/src/security/content-validator.js +132 -0
  117. package/src/security/index.js +143 -0
  118. package/src/security/network-guard.js +126 -0
  119. package/src/security/pairing.js +180 -0
  120. package/src/security/path-guard.js +140 -0
  121. package/src/security/persona-guard.js +67 -0
  122. package/src/security/policies.js +452 -0
  123. package/src/security/safety-constants.js +34 -0
  124. package/src/security/watchdog.js +107 -0
  125. package/src/sessions.js +130 -0
  126. package/src/spend.js +97 -0
  127. package/src/startup-watchdog.js +59 -0
  128. package/src/stdio/args.js +71 -0
  129. package/src/stdio/guard.js +59 -0
  130. package/src/stdio/protocol.js +167 -0
  131. package/src/stdio/run.js +106 -0
  132. package/src/stdio/session.js +180 -0
  133. package/src/store/agent-slice.js +306 -0
  134. package/src/store/dataset-slice.js +73 -0
  135. package/src/store/index.js +22 -0
  136. package/src/store/process-slice.js +135 -0
  137. package/src/store/session-slice.js +191 -0
  138. package/src/store/ui-slice.js +119 -0
  139. package/src/tasks/db.js +184 -0
  140. package/src/tasks/queries.js +589 -0
  141. package/src/tools/agent-tools.js +473 -0
  142. package/src/tools/checkpoint.js +152 -0
  143. package/src/tools/command-approvals.js +180 -0
  144. package/src/tools/dataset.js +50 -0
  145. package/src/tools/filesystem.js +682 -0
  146. package/src/tools/inbox-tools.js +48 -0
  147. package/src/tools/mesh.js +135 -0
  148. package/src/tools/own-env.js +136 -0
  149. package/src/tools/permissions.js +681 -0
  150. package/src/tools/plugin-tools.js +123 -0
  151. package/src/tools/process-tools.js +595 -0
  152. package/src/tools/registry.js +307 -0
  153. package/src/tools/swap-tools.js +72 -0
  154. package/src/tools/system.js +662 -0
  155. package/src/tools/tasks.js +532 -0
  156. package/src/tools/tool-search.js +171 -0
  157. package/src/ui/header.js +140 -0
  158. package/src/ui/input-cursor.js +23 -0
  159. package/src/ui/last-line.js +25 -0
  160. package/src/ui/line-edit.js +135 -0
  161. package/src/ui/output.js +399 -0
  162. package/src/ui/paste-tokens.js +131 -0
  163. package/src/ui/prompt-attention.js +134 -0
  164. package/src/ui/render-options.js +13 -0
  165. package/src/ui/replay.js +94 -0
  166. package/src/ui/splash.js +49 -0
  167. package/src/ui/status-level.js +36 -0
  168. package/src/ui/tool-ledger.js +203 -0
  169. package/src/ui/window-title.js +150 -0
  170. package/src/update.js +205 -0
  171. package/system.md +63 -0
@@ -0,0 +1,309 @@
1
+ /**
2
+ * Flow Controller — unified flow state machine
3
+ *
4
+ * Merges: loop-detector.js + auto-continue logic from drain-loop.js
5
+ * Owns all flow decisions: loop detection, plan state, continue/stop, learning triggers.
6
+ *
7
+ * Used by:
8
+ * - agent.js: checkLoop(), resetFlow()
9
+ * - drain-loop.js: shouldContinue()
10
+ */
11
+
12
+ import { createLogger } from "../logging/logger.js";
13
+
14
+ const log = createLogger("flow");
15
+
16
+ // --- Configurable thresholds ---
17
+ export function getThresholds() { return { ...THRESHOLDS }; }
18
+ const THRESHOLDS = {
19
+ textRepeat: parseInt(process.env.AGENT_LOOP_TEXT_REPEAT || "5", 10),
20
+ toolRepeat: parseInt(process.env.AGENT_LOOP_TOOL_REPEAT || "5", 10),
21
+ toolRepeatScreenshot: parseInt(process.env.AGENT_LOOP_TOOL_SCREENSHOT || "7", 10),
22
+ toolRepeatCommand: parseInt(process.env.AGENT_LOOP_TOOL_COMMAND || "8", 10),
23
+ desktopObservations: parseInt(process.env.AGENT_LOOP_DESKTOP_OBS || "8", 10),
24
+ coordGridPx: parseInt(process.env.AGENT_LOOP_COORD_GRID || "50", 10),
25
+ maxRetries: 3,
26
+ maxPlanCompletions: 20,
27
+ historySize: 12,
28
+ textHistorySize: 5,
29
+ };
30
+
31
+ const NAV_KEYS = new Set(["tab", "shift+tab", "down", "up", "left", "right", "pagedown", "pageup", "home", "end"]);
32
+ const MAX_REPLANS = 3;
33
+
34
+ // --- State ---
35
+ let _recentTexts = [];
36
+ let _recentToolCalls = [];
37
+ let _consecutiveDesktopObs = 0;
38
+ let _replanCount = 0;
39
+ let _autoRetries = new Map();
40
+ let _planCompletions = 0;
41
+ let _userInterrupted = false; // set when user sends a message during auto-continue
42
+
43
+ // --- Loop Detection (from loop-detector.js) ---
44
+
45
+ function _fuzzyToolSig(name, args) {
46
+ if (!args || typeof args !== "object") return name;
47
+ if (name.includes("wait_stable") || name.includes("wait_change") || name === "run_background_command") {
48
+ return name + ":skip:" + Date.now();
49
+ }
50
+ if (name.includes("key") && args.keys && NAV_KEYS.has(String(args.keys).toLowerCase())) {
51
+ return name + ":nav:" + Date.now();
52
+ }
53
+ const key = {};
54
+ for (const [k, v] of Object.entries(args)) {
55
+ if (k === "x" || k === "y") {
56
+ key[k] = Math.round(Number(v) / THRESHOLDS.coordGridPx) * THRESHOLDS.coordGridPx;
57
+ } else if (k === "text" || k === "command") {
58
+ key[k] = String(v).slice(0, 120);
59
+ } else if (k === "desktop_id" || k === "action" || k === "keys" || k === "cell") {
60
+ key[k] = v;
61
+ } else {
62
+ // Bug fix 2026-04-12: default was slice(0, 40), which collapsed all
63
+ // nested paths like /tmp/flint-bench/l4-deep/003/level1/level2/... into
64
+ // a single signature and caused the loop detector to hard-stop legitimate
65
+ // sequential reads across 6 different deep files. 200 chars covers all
66
+ // realistic paths and file identifiers without meaningful memory cost.
67
+ key[k] = typeof v === "string" ? v.slice(0, 200) : v;
68
+ }
69
+ }
70
+ return name + ":" + JSON.stringify(key);
71
+ }
72
+
73
+ function _buildLoopResult(type, count, detail) {
74
+ _replanCount++;
75
+ if (_replanCount > MAX_REPLANS) {
76
+ return { type, count, action: "stop", message: `[Loop: ${detail} repeated ${count}x. Re-plan failed. Stopped.]` };
77
+ }
78
+ return { type, count, action: "replan", message: `[Loop: ${detail} repeated ${count}x. STOP. Create NEW plan with DIFFERENT strategy.]` };
79
+ }
80
+
81
+ export function checkTextLoop(text) {
82
+ const clean = (text || "").trim().toLowerCase();
83
+ if (clean.length === 0 || clean.length >= 100) return null;
84
+ _recentTexts.push(clean);
85
+ if (_recentTexts.length > THRESHOLDS.textHistorySize) _recentTexts.shift();
86
+ if (_recentTexts.length >= THRESHOLDS.textRepeat) {
87
+ const last = _recentTexts[_recentTexts.length - 1];
88
+ const count = _recentTexts.filter(t => t === last).length;
89
+ if (count >= THRESHOLDS.textRepeat) {
90
+ log.warn("text-loop", { text: clean.slice(0, 60), count });
91
+ return _buildLoopResult("text", count, text);
92
+ }
93
+ }
94
+ return null;
95
+ }
96
+
97
+ export function checkToolLoop(toolName, args) {
98
+ const sig = _fuzzyToolSig(toolName, args);
99
+ _recentToolCalls.push(sig);
100
+ if (_recentToolCalls.length > THRESHOLDS.historySize) _recentToolCalls.shift();
101
+ const count = _recentToolCalls.filter(t => t === sig).length;
102
+ const threshold = (toolName.includes("screenshot") || toolName.includes("look"))
103
+ ? THRESHOLDS.toolRepeatScreenshot
104
+ : (toolName === "run_command" || toolName.includes("shell"))
105
+ ? THRESHOLDS.toolRepeatCommand
106
+ : THRESHOLDS.toolRepeat;
107
+ if (count >= threshold) {
108
+ log.warn("tool-loop", { tool: toolName, count, sig });
109
+ return _buildLoopResult("tool", count, toolName);
110
+ }
111
+ return null;
112
+ }
113
+
114
+ export function checkDesktopLoop(toolCalls) {
115
+ const hasAction = toolCalls.some(tc => {
116
+ const n = tc.function?.name || tc.name || "";
117
+ if (!n.includes("desktop") && !n.includes("screenbox")) return true;
118
+ return n.includes("type") || n.includes("batch") || n.includes("shell")
119
+ || n.includes("chrome") || n.includes("manage") || n.includes("file");
120
+ });
121
+ if (!hasAction) _consecutiveDesktopObs++;
122
+ else _consecutiveDesktopObs = 0;
123
+ if (_consecutiveDesktopObs >= THRESHOLDS.desktopObservations) {
124
+ log.warn("desktop-loop", { count: _consecutiveDesktopObs });
125
+ const result = _buildLoopResult("desktop", _consecutiveDesktopObs, "desktop observations");
126
+ _consecutiveDesktopObs = 0;
127
+ return result;
128
+ }
129
+ return null;
130
+ }
131
+
132
+ export function resetDesktopOnMeaningfulText(text) {
133
+ if ((text || "").toLowerCase().match(/\b(done|completed|finished|saved|created|failed|cannot|impossible)\b/)) {
134
+ _consecutiveDesktopObs = 0;
135
+ }
136
+ }
137
+
138
+ // --- What the provider's refusals mean ---
139
+ //
140
+ // Shared on purpose. There are two autonomous loops, /auto in agent/auto.js
141
+ // and the bus in bus/drain-loop.js, and they are accepted
142
+ // separately because they are different loops. What counts as a refusal must
143
+ // not be one of the things they can disagree about: a reason that is fatal on
144
+ // one door and survivable on the other is a hole with a door in front of it.
145
+ //
146
+ // NOTE for anyone moving these: the rate-limit COUNTER must not live in this
147
+ // module. resetFlow() runs at the top of every agent invocation
148
+ // (agent.js:248), so anything kept here is wiped once per turn and a counter
149
+ // of consecutive turns would never reach two.
150
+
151
+ /** The provider is not going to serve the next turn either. A human is needed. */
152
+ // "empty": the model answered with nothing three times running. Another
153
+ // autonomous turn would only buy more empty, billed answers.
154
+ export const FATAL_PROVIDER_REASONS = new Set(["auth", "quota", "error", "empty"]);
155
+
156
+ /** How many 429s in a row before a run treats the cap as a closed door. */
157
+ export const MAX_CONSECUTIVE_RATE_LIMITS = 3;
158
+
159
+ /** How long to wait out a 429 when the provider did not send retry-after. */
160
+ export const RATE_LIMIT_WAIT_MS = 20000;
161
+
162
+ /** One line per reason, for the human who has to decide what to do about it. */
163
+ export const STOP_REASON_TEXT = {
164
+ auth: "the provider rejected the API key",
165
+ quota: "the account is out of credits",
166
+ error: "the provider failed three calls in a row",
167
+ "rate-limit": "the provider kept rate limiting the run",
168
+ empty: "the model answered with nothing three times in a row",
169
+ };
170
+
171
+ // --- Auto-Continue / Plan State (from drain-loop.js) ---
172
+
173
+ /**
174
+ * Decide whether to continue after agent returns.
175
+ * @param {object} opts - { stopReason, plan, isApiScoped, apiGoalId, apiVerified }
176
+ * @returns {null | {action: "stop"|"continue"|"verify", message?: string, prompt?: string}}
177
+ */
178
+ export function shouldContinue({ stopReason, plan, isApiScoped, apiGoalId }) {
179
+ // User interrupt: any non-self-continue message during auto-continue pauses the plan.
180
+ // The user typed something → they want control. Don't fight them with auto-Continue.
181
+ // Clear via resetFlow() (/new) or setUserInterrupt(false) (/resume).
182
+ if (_userInterrupted) {
183
+ log.info("flow: user interrupted auto-continue, pausing");
184
+ return { action: "stop", reason: "user_interrupt" };
185
+ }
186
+
187
+ const hasPending = plan && plan.tasks?.some(t => t.status === "pending" || t.status === "in_progress");
188
+
189
+ // No plan + done = simple task complete
190
+ if (!plan) {
191
+ log.info("flow: no plan + %s = stop", stopReason);
192
+ return { action: "stop" };
193
+ }
194
+
195
+ // Plan exists, all tasks done
196
+ if (!hasPending) {
197
+ _autoRetries.clear();
198
+ _planCompletions++;
199
+
200
+ // API scope: goal done = stop. Verification handled by agent loop's self-verify gate.
201
+ if (isApiScoped && plan?.goalId === apiGoalId) {
202
+ log.info("flow: API goal complete, stopping", { goalId: plan.goalId });
203
+ return { action: "stop", goalComplete: true };
204
+ }
205
+
206
+ // Safety limit
207
+ if (_planCompletions >= THRESHOLDS.maxPlanCompletions) {
208
+ _planCompletions = 0;
209
+ log.warn("flow: max plan completions reached");
210
+ return { action: "stop" };
211
+ }
212
+
213
+ // TUI: check other goals
214
+ return {
215
+ action: "continue",
216
+ prompt: "Your plan is complete. Check if your original task has more work. If everything is truly done, write a summary and stop.",
217
+ };
218
+ }
219
+
220
+ // Plan has pending tasks — find next
221
+ const pending = plan.tasks.filter(t => t.status === "pending" || t.status === "in_progress");
222
+ let nextTask = null;
223
+ for (const t of pending) {
224
+ const retries = _autoRetries.get(t.id) || 0;
225
+ if (retries < THRESHOLDS.maxRetries) {
226
+ nextTask = t;
227
+ break;
228
+ }
229
+ }
230
+
231
+ if (!nextTask) {
232
+ _autoRetries.clear();
233
+ return {
234
+ action: "continue",
235
+ prompt: "All remaining tasks exhausted retries. Mark them as skipped and move to the next phase.",
236
+ };
237
+ }
238
+
239
+ const retries = _autoRetries.get(nextTask.id) || 0;
240
+ _autoRetries.set(nextTask.id, retries + 1);
241
+
242
+ const retryNote = retries > 0
243
+ ? ` (retry ${retries}/${THRESHOLDS.maxRetries} — try a DIFFERENT approach. If blocked, add an alternative task via add_task and skip this one.)`
244
+ : "";
245
+ return {
246
+ action: "continue",
247
+ prompt: `Continue: "${nextTask.title}" [#${nextTask.id}]${retryNote}. Use update_task to mark progress.`,
248
+ };
249
+ }
250
+
251
+ // --- Learning Trigger ---
252
+
253
+ /**
254
+ * Check if current state is a learning opportunity.
255
+ * Called by Verification Gate after successful task completion.
256
+ * @returns {boolean}
257
+ */
258
+ export function isLearningOpportunity({ plan, toolCallCount }) {
259
+ // Multi-step task completed successfully = learning opportunity
260
+ if (plan && plan.tasks?.length >= 2) return true;
261
+ // Many tool calls = complex task worth learning from
262
+ if (toolCallCount > 5) return true;
263
+ return false;
264
+ }
265
+
266
+ // --- Reset ---
267
+
268
+ /**
269
+ * Set/clear user interrupt flag. Called from drain-loop when a user
270
+ * (non-self-continue) message arrives during an active plan.
271
+ */
272
+ export function setUserInterrupt(interrupted) {
273
+ if (_userInterrupted !== interrupted) {
274
+ log.info("flow: user-interrupt", { interrupted });
275
+ }
276
+ _userInterrupted = interrupted;
277
+ }
278
+
279
+ export function isUserInterrupted() {
280
+ return _userInterrupted;
281
+ }
282
+
283
+ /**
284
+ * Per-turn reset: the loop detectors only. Called at the top of every agent
285
+ * invocation.
286
+ *
287
+ * It used to be resetFlow(), which also cleared the run-level state below.
288
+ * That ran once per turn, so the user interrupt the drain loop had just set
289
+ * was gone before shouldContinue could read it, the per-task retry cap never
290
+ * got past one, and the plan-completion cap never got past one either.
291
+ */
292
+ export function resetTurn() {
293
+ _recentTexts = [];
294
+ _recentToolCalls = [];
295
+ _consecutiveDesktopObs = 0;
296
+ _replanCount = 0;
297
+ }
298
+
299
+ /**
300
+ * Full reset: the loop detectors plus the run-level state (retries, plan
301
+ * completions, user interrupt). For the start of a new run and /new.
302
+ */
303
+ export function resetFlow() {
304
+ resetTurn();
305
+ _autoRetries.clear();
306
+ _planCompletions = 0;
307
+ _userInterrupted = false;
308
+ }
309
+
@@ -0,0 +1,231 @@
1
+ // Intent manifest — catalog of intent classes.
2
+ //
3
+ // Each intent describes a *class of request* — how many steps it usually needs,
4
+ // what the output looks like, and whether tools are required at all. Tool
5
+ // selection itself is delegated to the classifier at runtime, which picks
6
+ // concrete tools from the live registry. This keeps the manifest agnostic to
7
+ // specific tool names — adding a new tool in registry does not require editing
8
+ // this file.
9
+ //
10
+ // Fields:
11
+ // description — one-line hint for the classifier prompt
12
+ // needsTools — false = pure text/knowledge answer (empty tool list enforced)
13
+ // true = classifier will pick tools from the live registry
14
+ // max_steps — hard cap on iterations for this intent
15
+ // expected — shape of final output: "text" | "tool_call" | "tool_and_summary" | "mixed"
16
+ // changes : does a request of this class end with something different
17
+ // on disk? "yes" when that is the whole point of the class,
18
+ // "no" when it is a question, "maybe" when the class covers
19
+ // both. The no-change check uses it to decide whether a turn that changed
20
+ // nothing owes the operator an explanation. It is a property
21
+ // of the CLASS, written down once where the classes are
22
+ // described and reviewable there, not a guess about the
23
+ // wording of any one request.
24
+
25
+ export const INTENTS = {
26
+ // --- Pure text: model answers from own knowledge, no tools at all ---
27
+ creative_text: {
28
+ description: "Generate text from scratch and reply in chat: poems, stories, summaries, explanations, jokes, translations, code snippets (no execution). DO NOT pick this if the user asks to SAVE/WRITE the result to a file or path — that is file_write.",
29
+ needsTools: false,
30
+ max_steps: 1,
31
+ changes: "no",
32
+ expected: "text",
33
+ },
34
+
35
+ knowledge_qa: {
36
+ description: "Factual question, definition, or explanation that the model can answer directly without tools",
37
+ needsTools: false,
38
+ max_steps: 1,
39
+ changes: "no",
40
+ expected: "text",
41
+ },
42
+
43
+ reasoning: {
44
+ description: "Math, logic, puzzles, analysis of data provided in the message itself",
45
+ needsTools: false,
46
+ max_steps: 2,
47
+ changes: "no",
48
+ expected: "text",
49
+ },
50
+
51
+ chat: {
52
+ description: "Small talk, greetings, acknowledgements, clarification questions, meta discussion about the conversation",
53
+ needsTools: false,
54
+ max_steps: 1,
55
+ changes: "no",
56
+ expected: "text",
57
+ },
58
+
59
+ // --- Tool-backed: classifier picks concrete tools from live registry ---
60
+ shell_command: {
61
+ description: "Run a single shell command and report output",
62
+ needsTools: true,
63
+ // Was 5; bumped to 12 on 2026-04-21. The 5-step cap was too tight for
64
+ // any shell work involving curl + JSON escape or HTTP-payload work —
65
+ // observed during the readiness-test mirror experiment where the agent
66
+ // burned every iteration on quoting attempts before giving up.
67
+ max_steps: 12,
68
+ changes: "maybe",
69
+ expected: "tool_and_summary",
70
+ tool_pattern: /^(run_command|run_background_command|peek_process|list_processes|kill_process|read_file)$/,
71
+ },
72
+
73
+ shell_multi: {
74
+ description: "Run several shell commands to accomplish a goal (install, diagnose, pipe-chain)",
75
+ needsTools: true,
76
+ // Was 8; bumped to 16 on 2026-04-21 for consistency with shell_command.
77
+ // Multi-shell pipelines on Windows (bash-on-windows escape quirks)
78
+ // often need several extra diagnostic steps.
79
+ max_steps: 16,
80
+ changes: "maybe",
81
+ expected: "tool_and_summary",
82
+ tool_pattern: /^(run_command|run_background_command|peek_process|list_processes|kill_process|read_file|write_file|edit_file)$/,
83
+ },
84
+
85
+ file_read: {
86
+ description: "Read the contents of a file or list a directory",
87
+ needsTools: true,
88
+ max_steps: 3,
89
+ changes: "no",
90
+ expected: "tool_and_summary",
91
+ },
92
+
93
+ file_write: {
94
+ description: "Create or overwrite a file with specific content. ALWAYS pick this when the user asks to SAVE/WRITE/STORE generated text to a file or path (e.g. 'write a tweet and save to /tmp/x.txt', 'generate an outline and save it to ...'). The text-generation part does NOT make it creative_text once a file destination is named.",
95
+ needsTools: true,
96
+ max_steps: 3,
97
+ changes: "yes",
98
+ expected: "tool_and_summary",
99
+ },
100
+
101
+ file_edit: {
102
+ description: "Modify an existing file: read it, change parts, write it back",
103
+ needsTools: true,
104
+ max_steps: 5,
105
+ changes: "yes",
106
+ expected: "tool_and_summary",
107
+ },
108
+
109
+ file_manage: {
110
+ description: "Copy, move, delete, search across files",
111
+ needsTools: true,
112
+ max_steps: 6,
113
+ changes: "yes",
114
+ expected: "tool_and_summary",
115
+ },
116
+
117
+ web_fetch: {
118
+ description: "Fetch a specific URL and report/parse its content",
119
+ needsTools: true,
120
+ max_steps: 3,
121
+ changes: "no",
122
+ expected: "tool_and_summary",
123
+ // Widen: when user asks to read a page, include browser tools too —
124
+ // search-only gets stuck on JS-heavy sites (observed: linkedin/twitter/github).
125
+ tool_pattern: /^(web_fetch|web_search|mcp_browser_)/,
126
+ },
127
+
128
+ web_search: {
129
+ description: "Search the web for information on a topic, including 'open <site>', 'navigate to', 'search on <site>' — these are tool-backed browsing, not just a search engine query",
130
+ needsTools: true,
131
+ max_steps: 8,
132
+ changes: "no",
133
+ expected: "tool_and_summary",
134
+ // Same widening, so "go to LinkedIn" no longer gets only web_search.
135
+ tool_pattern: /^(web_search|web_fetch|mcp_browser_)/,
136
+ },
137
+
138
+ task_create: {
139
+ description: "Create a new plan with tasks, or add a task to an existing plan",
140
+ needsTools: true,
141
+ max_steps: 3,
142
+ changes: "maybe",
143
+ expected: "tool_and_summary",
144
+ },
145
+
146
+ task_update: {
147
+ description: "Change the status of a task (done, skipped, in_progress), add notes, mark progress",
148
+ needsTools: true,
149
+ max_steps: 3,
150
+ changes: "maybe",
151
+ expected: "tool_and_summary",
152
+ },
153
+
154
+ task_view: {
155
+ description: "Show the current plan, list tasks, check progress",
156
+ needsTools: true,
157
+ max_steps: 2,
158
+ changes: "no",
159
+ expected: "tool_and_summary",
160
+ },
161
+
162
+ memory_write: {
163
+ description: "Remember something for later: save a fact, note, or preference",
164
+ needsTools: true,
165
+ max_steps: 3,
166
+ changes: "maybe",
167
+ expected: "tool_and_summary",
168
+ },
169
+
170
+ memory_read: {
171
+ description: "Recall something from memory: search past notes, find saved facts",
172
+ needsTools: true,
173
+ max_steps: 4,
174
+ changes: "no",
175
+ expected: "tool_and_summary",
176
+ },
177
+
178
+ memory_manage: {
179
+ description: "Write AND read memory in the same task: save then verify, delete then confirm, or similar multi-step memory flows",
180
+ needsTools: true,
181
+ max_steps: 12,
182
+ changes: "maybe",
183
+ expected: "tool_and_summary",
184
+ },
185
+
186
+ agent_spawn: {
187
+ description: "Delegate work to a child agent, ask another agent, or wait for agents to finish",
188
+ needsTools: true,
189
+ max_steps: 5,
190
+ changes: "maybe",
191
+ expected: "tool_and_summary",
192
+ },
193
+
194
+ desktop: {
195
+ description: "Interact with a remote desktop: screenshot, click, type, open apps, browse, run desktop shell",
196
+ needsTools: true,
197
+ max_steps: 20,
198
+ changes: "maybe",
199
+ expected: "tool_and_summary",
200
+ // Desktop + browser both can handle screenshots/clicks; expose both families.
201
+ tool_pattern: /^(mcp_screenbox_|mcp_browser_|view_image)/,
202
+ },
203
+
204
+ system_info: {
205
+ description: "Report info about Flint itself: balance, models, providers, active plan, think out loud",
206
+ needsTools: true,
207
+ max_steps: 2,
208
+ changes: "no",
209
+ expected: "tool_and_summary",
210
+ },
211
+
212
+ complex_multi: {
213
+ description: "Multi-step task combining several categories (web + file + shell, plan + execute, etc.) — requires full tool surface",
214
+ needsTools: true,
215
+ max_steps: 30,
216
+ changes: "maybe",
217
+ expected: "mixed",
218
+ },
219
+ };
220
+
221
+ // Compact list for classifier prompt — just name + description
222
+ export function formatIntentCatalog() {
223
+ return Object.entries(INTENTS)
224
+ .map(([name, spec]) => ` ${name}: ${spec.description}`)
225
+ .join("\n");
226
+ }
227
+
228
+ // Resolve intent name → full spec with defaults
229
+ export function resolveIntent(name) {
230
+ return INTENTS[name] || INTENTS.complex_multi;
231
+ }
@@ -0,0 +1,46 @@
1
+ // The classifier's per-attempt budget.
2
+ //
3
+ // Item 10: operator-visibility.test.js failed on a loaded machine, and one of
4
+ // the causes was a hard 15 s wall-clock deadline inside the classifier call that
5
+ // no caller could see or control. Under load that deadline fired before the
6
+ // mocked response was processed, and `classifyIntent` — which catches its own
7
+ // errors — returned the all-tools fallback. Nothing said so, which is how a test
8
+ // came to depend on how busy the machine was: the code under the assertion
9
+ // quietly changed its mind, and a fallback is indistinguishable from a real
10
+ // classification in the output.
11
+ //
12
+ // Its own module rather than a line in intent.js, because intent.js pulls in
13
+ // the config, the API client and the tool manifest: importing it to test one
14
+ // arithmetic decision means paying for a network client and an env-dependent
15
+ // config load, and the failure mode of that is a test that cannot run in CI at
16
+ // all. This is a pure function of the environment and is tested as one.
17
+ //
18
+ // The project rule is no silent fallbacks — a required value that is absent
19
+ // throws at config load rather than becoming a default. An override gets the
20
+ // same treatment. A caller that sets INTENT_TIMEOUT_MS=0 and silently receives
21
+ // 15 s has the original problem in a new place: a budget that is not what was
22
+ // asked for, with nothing said about it.
23
+
24
+ /** The production default, in ms. */
25
+ export const DEFAULT_INTENT_TIMEOUT_MS = 15000;
26
+
27
+ /**
28
+ * The per-attempt budget for a classifier call.
29
+ *
30
+ * @returns {number} milliseconds
31
+ * @throws {Error} if INTENT_TIMEOUT_MS is set to something that is not a
32
+ * positive number of milliseconds
33
+ */
34
+ export function intentTimeoutMs() {
35
+ const raw = process.env.INTENT_TIMEOUT_MS;
36
+ // Unset, or set to empty: the documented default. An empty value in a .env is
37
+ // a real thing to find and it means "not set", not "zero".
38
+ if (!raw) return DEFAULT_INTENT_TIMEOUT_MS;
39
+ const n = Number(raw);
40
+ if (!Number.isFinite(n) || n <= 0) {
41
+ throw new Error(
42
+ `INTENT_TIMEOUT_MS must be a positive number of milliseconds, got: ${raw}`,
43
+ );
44
+ }
45
+ return n;
46
+ }