flint-agent 1.14.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (171) hide show
  1. package/.env.example +108 -0
  2. package/CHANGELOG.md +55 -0
  3. package/FEATURES.md +298 -0
  4. package/LICENSE +21 -0
  5. package/README.md +435 -0
  6. package/bin/flint.js +47 -0
  7. package/config/classifier-prompt.md +218 -0
  8. package/config/models-curated.json +4 -0
  9. package/config/providers.json +74 -0
  10. package/package.json +92 -0
  11. package/patches/ink+6.8.0.patch +78 -0
  12. package/profiles/desktop.md +65 -0
  13. package/profiles/generic.md +20 -0
  14. package/profiles/marketer.md +20 -0
  15. package/profiles/profiles.json +34 -0
  16. package/profiles/ux-reviewer.md +25 -0
  17. package/src/agent/agent.js +1743 -0
  18. package/src/agent/auto.js +346 -0
  19. package/src/agent/backoff.js +143 -0
  20. package/src/agent/compression.js +310 -0
  21. package/src/agent/content-resolver.js +180 -0
  22. package/src/agent/flow-controller.js +309 -0
  23. package/src/agent/intent-manifest.js +231 -0
  24. package/src/agent/intent-timeout.js +46 -0
  25. package/src/agent/intent.js +633 -0
  26. package/src/agent/knowledge.js +114 -0
  27. package/src/agent/learning.js +180 -0
  28. package/src/agent/modes.js +187 -0
  29. package/src/agent/outcome-ask.js +91 -0
  30. package/src/agent/project-context.js +76 -0
  31. package/src/agent/prompt-budget.js +117 -0
  32. package/src/agent/reflection-extractor.js +140 -0
  33. package/src/agent/steering.js +86 -0
  34. package/src/agent/supervisor.js +430 -0
  35. package/src/agent/swap.js +443 -0
  36. package/src/agent/system-prompt.js +446 -0
  37. package/src/agent/time-stamp.js +48 -0
  38. package/src/agent/tool-guard.js +201 -0
  39. package/src/agent/toolcall-text.js +162 -0
  40. package/src/agent/usage.js +297 -0
  41. package/src/agent/vision.js +94 -0
  42. package/src/agent/watchdog.js +139 -0
  43. package/src/agent/workspace-changes.js +177 -0
  44. package/src/api/address.js +14 -0
  45. package/src/api/client.js +280 -0
  46. package/src/api/server.js +535 -0
  47. package/src/api/stream-pipe.js +113 -0
  48. package/src/app-state.js +39 -0
  49. package/src/bootstrap.js +501 -0
  50. package/src/bus/drain-loop.js +497 -0
  51. package/src/bus/index.js +270 -0
  52. package/src/bus/plugins.js +65 -0
  53. package/src/child-idle.js +14 -0
  54. package/src/cli.js +118 -0
  55. package/src/commands/commands.js +1297 -0
  56. package/src/commands/registry.js +132 -0
  57. package/src/components/App.js +491 -0
  58. package/src/components/CarefulMenu.js +145 -0
  59. package/src/components/HistoryWriter.js +86 -0
  60. package/src/components/LineInput.js +69 -0
  61. package/src/components/LiveZone.js +294 -0
  62. package/src/components/OverlayMenu.js +179 -0
  63. package/src/components/SystemPanel.js +156 -0
  64. package/src/components/Table.js +54 -0
  65. package/src/config.js +249 -0
  66. package/src/free-models.js +230 -0
  67. package/src/index.js +1111 -0
  68. package/src/input-handler.js +13 -0
  69. package/src/input-text.js +123 -0
  70. package/src/launcher.js +129 -0
  71. package/src/logging/api-log.js +95 -0
  72. package/src/logging/chat-log-follower.js +113 -0
  73. package/src/logging/chat-log.js +15 -0
  74. package/src/logging/log-collector.js +182 -0
  75. package/src/logging/logger.js +112 -0
  76. package/src/logging/tool-log.js +20 -0
  77. package/src/mcp-client.js +314 -0
  78. package/src/memory/conversation-digest.js +113 -0
  79. package/src/memory/extract-facts.js +98 -0
  80. package/src/memory/facts.js +181 -0
  81. package/src/memory/inbox.js +63 -0
  82. package/src/memory/markdown.js +38 -0
  83. package/src/memory/patterns.js +185 -0
  84. package/src/memory/project.js +66 -0
  85. package/src/memory/reflections.js +74 -0
  86. package/src/memory/retrieval.js +84 -0
  87. package/src/memory/rules.js +105 -0
  88. package/src/memory/session-facts.js +125 -0
  89. package/src/memory/skills.js +191 -0
  90. package/src/memory/sqlite-store.js +653 -0
  91. package/src/memory/store.js +208 -0
  92. package/src/memory/tools.js +196 -0
  93. package/src/memory/user-model.js +86 -0
  94. package/src/message-handler.js +775 -0
  95. package/src/model-check.js +218 -0
  96. package/src/plugins/loader.js +120 -0
  97. package/src/plugins/manager.js +88 -0
  98. package/src/production-env.js +22 -0
  99. package/src/profiles.js +42 -0
  100. package/src/providers/adapters/anthropic.js +270 -0
  101. package/src/providers/adapters/openai.js +120 -0
  102. package/src/providers/keys-dpapi.js +41 -0
  103. package/src/providers/keys-fallback.js +31 -0
  104. package/src/providers/keys.js +132 -0
  105. package/src/providers/models.js +154 -0
  106. package/src/providers/registry.js +56 -0
  107. package/src/providers/state.js +56 -0
  108. package/src/registry.js +96 -0
  109. package/src/restart.js +29 -0
  110. package/src/sandbox/backend.js +130 -0
  111. package/src/security/api-auth.js +132 -0
  112. package/src/security/audit.js +98 -0
  113. package/src/security/child-policy.js +41 -0
  114. package/src/security/command-guard.js +173 -0
  115. package/src/security/content-fence.js +250 -0
  116. package/src/security/content-validator.js +132 -0
  117. package/src/security/index.js +143 -0
  118. package/src/security/network-guard.js +126 -0
  119. package/src/security/pairing.js +180 -0
  120. package/src/security/path-guard.js +140 -0
  121. package/src/security/persona-guard.js +67 -0
  122. package/src/security/policies.js +452 -0
  123. package/src/security/safety-constants.js +34 -0
  124. package/src/security/watchdog.js +107 -0
  125. package/src/sessions.js +130 -0
  126. package/src/spend.js +97 -0
  127. package/src/startup-watchdog.js +59 -0
  128. package/src/stdio/args.js +71 -0
  129. package/src/stdio/guard.js +59 -0
  130. package/src/stdio/protocol.js +167 -0
  131. package/src/stdio/run.js +106 -0
  132. package/src/stdio/session.js +180 -0
  133. package/src/store/agent-slice.js +306 -0
  134. package/src/store/dataset-slice.js +73 -0
  135. package/src/store/index.js +22 -0
  136. package/src/store/process-slice.js +135 -0
  137. package/src/store/session-slice.js +191 -0
  138. package/src/store/ui-slice.js +119 -0
  139. package/src/tasks/db.js +184 -0
  140. package/src/tasks/queries.js +589 -0
  141. package/src/tools/agent-tools.js +473 -0
  142. package/src/tools/checkpoint.js +152 -0
  143. package/src/tools/command-approvals.js +180 -0
  144. package/src/tools/dataset.js +50 -0
  145. package/src/tools/filesystem.js +682 -0
  146. package/src/tools/inbox-tools.js +48 -0
  147. package/src/tools/mesh.js +135 -0
  148. package/src/tools/own-env.js +136 -0
  149. package/src/tools/permissions.js +681 -0
  150. package/src/tools/plugin-tools.js +123 -0
  151. package/src/tools/process-tools.js +595 -0
  152. package/src/tools/registry.js +307 -0
  153. package/src/tools/swap-tools.js +72 -0
  154. package/src/tools/system.js +662 -0
  155. package/src/tools/tasks.js +532 -0
  156. package/src/tools/tool-search.js +171 -0
  157. package/src/ui/header.js +140 -0
  158. package/src/ui/input-cursor.js +23 -0
  159. package/src/ui/last-line.js +25 -0
  160. package/src/ui/line-edit.js +135 -0
  161. package/src/ui/output.js +399 -0
  162. package/src/ui/paste-tokens.js +131 -0
  163. package/src/ui/prompt-attention.js +134 -0
  164. package/src/ui/render-options.js +13 -0
  165. package/src/ui/replay.js +94 -0
  166. package/src/ui/splash.js +49 -0
  167. package/src/ui/status-level.js +36 -0
  168. package/src/ui/tool-ledger.js +203 -0
  169. package/src/ui/window-title.js +150 -0
  170. package/src/update.js +205 -0
  171. package/system.md +63 -0
@@ -0,0 +1,430 @@
1
+ // Real-time Supervisor — observes tool calls and injects hints
2
+ // Toggle: /supervisor on|off
3
+ // Zero cost — local pattern matching + knowledge search, no API calls
4
+
5
+ import { createLogger } from "../logging/logger.js";
6
+
7
+ const log = createLogger("supervisor");
8
+
9
+ let _enabled = false; // toggled by /supervisor or auto-enabled for API messages
10
+ let _knowledgeSearchFn = null; // injected from outside
11
+ let _lastTools = []; // last N tool calls for pattern detection
12
+ let _hintCounts = {}; // per-rule escalation counter
13
+ let _lastExpect = null; // EXPECT string from last assistant message
14
+
15
+ export function setSupervisorEnabled(enabled) {
16
+ _enabled = enabled;
17
+ log.info("supervisor", { enabled });
18
+ }
19
+
20
+ export function isSupervisorEnabled() {
21
+ return _enabled;
22
+ }
23
+
24
+ export function setSupervisorKnowledgeFn(fn) {
25
+ _knowledgeSearchFn = fn;
26
+ }
27
+
28
+ /**
29
+ * Called after each tool execution, before next API call.
30
+ * Returns hint string to inject as system message, or null if no hint needed.
31
+ */
32
+ export function evaluateToolCall(toolName, args, result, context = {}) {
33
+ if (!_enabled) return null;
34
+
35
+ const hints = [];
36
+
37
+ // Track recent tools
38
+ _lastTools.push({ name: toolName, args, ts: Date.now() });
39
+ if (_lastTools.length > 10) _lastTools.shift();
40
+
41
+ // --- Rule: click without prior look ---
42
+ if (toolName.includes("click") && !toolName.includes("chrome")) {
43
+ const lastLook = _lastTools.slice(0, -1).reverse().find(t => t.name.includes("look"));
44
+ const lastScreenshot = _lastTools.slice(0, -1).reverse().find(t => t.name.includes("screenshot"));
45
+ if (!lastLook || (lastScreenshot && lastScreenshot.ts > lastLook.ts)) {
46
+ hints.push("HINT: Always use desktop_look(cell=N) BEFORE clicking to get precise coordinates. Screenshot coordinates are wrong (image is resized).");
47
+ }
48
+ }
49
+
50
+ // --- Rule: clicking same coordinates repeatedly ---
51
+ if (toolName.includes("click")) {
52
+ const recentClicks = _lastTools.filter(t => t.name.includes("click")).slice(-3);
53
+ if (recentClicks.length >= 3) {
54
+ const coords = recentClicks.map(t => `${t.args?.x},${t.args?.y}`);
55
+ if (coords[0] === coords[1] && coords[1] === coords[2]) {
56
+ hints.push("HINT: You clicked the same coordinates 3 times. The click is missing its target. Use desktop_look on a DIFFERENT cell to find the element, or try keyboard shortcut instead (Escape, Alt+N, Alt+Y, Tab+Enter).");
57
+ }
58
+ }
59
+ }
60
+
61
+ // --- Rule: local tool used when remote desktop tools were active ---
62
+ const LOCAL_TOOLS = new Set(["run_command", "write_file", "edit_file", "delete_file", "copy_file", "move_file"]);
63
+ if (LOCAL_TOOLS.has(toolName)) {
64
+ const hasRemoteDesktop = _lastTools.some(t => t.name.includes("desktop_") && t.name.startsWith("mcp_"));
65
+ if (hasRemoteDesktop) {
66
+ const remoteShell = _lastTools.find(t => t.name.includes("desktop_shell"))?.name || "desktop_shell";
67
+ hints.push(`HINT: You used a LOCAL tool (${toolName}) but this session has remote desktop tools active. Use ${remoteShell} for commands on the remote desktop. Local tools execute on YOUR machine, not the remote desktop.`);
68
+ }
69
+ }
70
+
71
+ // --- Rule: GUI app launched via shell — remind to activate window ---
72
+ if (toolName.includes("shell")) {
73
+ const cmd = args?.command || "";
74
+ const guiApps = ["soffice", "libreoffice", "gimp", "inkscape", "mousepad", "xfce4-terminal", "thunar", "firefox", "chromium"];
75
+ const launchedGui = guiApps.some(app => cmd.includes(app)) && cmd.includes("&");
76
+ if (launchedGui) {
77
+ hints.push("HINT: You launched a GUI app in background. It may NOT have focus. Before typing or pasting, use desktop_window(action='list') to find the new window, then desktop_window(action='activate', window_id=...) to give it focus. Otherwise keystrokes go to the wrong window.");
78
+ }
79
+ }
80
+
81
+ // --- Rule: paste/type right after shell launch without window activation ---
82
+ if (toolName.includes("key") || toolName.includes("type")) {
83
+ const lastShell = [..._lastTools].reverse().find(t => t.name.includes("shell"));
84
+ const lastActivate = [..._lastTools].reverse().find(t => t.name.includes("window") && t.args?.action === "activate");
85
+ if (lastShell) {
86
+ const shellCmd = lastShell.args?.command || "";
87
+ const isGuiLaunch = shellCmd.includes("&") && (shellCmd.includes("soffice") || shellCmd.includes("mousepad") || shellCmd.includes("gimp") || shellCmd.includes("terminal"));
88
+ if (isGuiLaunch && (!lastActivate || lastActivate.ts < lastShell.ts)) {
89
+ hints.push("HINT: You are typing/pasting but the last shell command launched a GUI app. You have NOT activated that window yet — keystrokes may go to the WRONG window. Use desktop_window(action='activate') first.");
90
+ }
91
+ }
92
+ }
93
+
94
+ // --- Rule: apt-get / sudo on screenbox ---
95
+ if (toolName.includes("shell")) {
96
+ const cmd = args?.command || "";
97
+ if (cmd.includes("apt-get") || cmd.includes("sudo")) {
98
+ hints.push("HINT: This desktop does not have root/sudo access. To install apps use: desktop_manage(action=\"install\", app=\"appname\"). To check available apps: desktop_manage(action=\"status\").");
99
+ }
100
+ }
101
+
102
+ // --- Rule: libreoffice without DISPLAY ---
103
+ if (toolName.includes("shell")) {
104
+ const cmd = args?.command || "";
105
+ if ((cmd.includes("libreoffice") || cmd.includes("soffice") || cmd.includes("localc")) && !cmd.includes("DISPLAY") && !cmd.includes("--headless")) {
106
+ hints.push("HINT: GUI apps need DISPLAY=:99. Use: desktop_shell(command=\"DISPLAY=:99 soffice --calc &\"). The & runs it in background so the command returns immediately.");
107
+ }
108
+ }
109
+
110
+ // --- Rule: libreoffice headless while GUI is open ---
111
+ if (toolName.includes("shell")) {
112
+ const cmd = args?.command || "";
113
+ const resultStr = typeof result === "string" ? result : JSON.stringify(result);
114
+ if (cmd.includes("--headless") && cmd.includes("convert") && (resultStr.includes("error") || resultStr.includes("timeout") || resultStr.includes("exit_code\": 1") || resultStr.includes("exit_code\":1"))) {
115
+ hints.push("HINT: Headless conversion fails when LibreOffice GUI is running. First: desktop_shell(command=\"pkill -f soffice\"), wait 2 seconds, then retry the conversion.");
116
+ }
117
+ }
118
+
119
+ // --- Rule: Save As workflow ---
120
+ if (toolName.includes("key")) {
121
+ const keys = args?.keys || "";
122
+ if (keys.toLowerCase() === "ctrl+shift+s") {
123
+ hints.push("HINT: Save As dialog opened. Steps: (1) Type filename in the File name field (Ctrl+A to select existing, then type new name). (2) Change file type dropdown if needed. (3) Navigate to Desktop folder. (4) Click Save. (5) If format confirmation appears, click 'Use Excel Format' or press Alt+Y. If you can't find buttons, use keyboard: Tab to navigate between fields, Enter to confirm.");
124
+ }
125
+ }
126
+
127
+ // --- Rule: Recovery/Discard dialog ---
128
+ if (_isScreenObserver(toolName) && typeof result === "string" && (result.includes("recover") || result.includes("Recover") || result.includes("Discard") || result.includes("recovery") || result.includes("Recovery"))) {
129
+ hints.push("HINT: Document Recovery dialog detected. Press alt+d to DISCARD (not Escape — Escape does NOT close this dialog). If alt+d doesn't work, use desktop_look to find the Discard button and click it.");
130
+ }
131
+
132
+ // --- Rule: Tip of the Day dialog ---
133
+ if (_isScreenObserver(toolName) && typeof result === "string" && result.includes("Tip of the Day")) {
134
+ hints.push("HINT: 'Tip of the Day' dialog — press Enter or Return to close it.");
135
+ }
136
+
137
+ // --- Rule: Text Import dialog ---
138
+ if (_isScreenObserver(toolName) && typeof result === "string" && (result.includes("Text Import") || result.includes("Separator"))) {
139
+ hints.push("HINT: Text Import dialog — Tab separator is usually correct. Press Enter/Return to accept.");
140
+ }
141
+
142
+ // --- Rule: unverified write action ---
143
+ // After a write/create action, if next tool is NOT a read/verify, remind to verify
144
+ if (_lastTools.length >= 2) {
145
+ const prev = _lastTools[_lastTools.length - 2];
146
+ const curr = _lastTools[_lastTools.length - 1];
147
+ const isWrite = _isWriteAction(prev.name, prev.args);
148
+ const isVerify = _isVerifyAction(curr.name, curr.args, prev.args);
149
+ if (isWrite && !isVerify) {
150
+ hints.push("HINT: You just performed a write action but did NOT verify the result. Read the file back, check the output, or take a screenshot to confirm success. Never claim success without verification.");
151
+ }
152
+ }
153
+
154
+ // --- Rule: same tool repeated WITH SAME ARGS ---
155
+ // Only flag if same tool AND same args across last 3 calls. Different args
156
+ // (e.g. read_file on different paths) is legitimate sequential work, not
157
+ // a loop. Bug fix 2026-04-12: previously only checked name, which flagged
158
+ // every multi-file read/write operation as a false-positive loop.
159
+ if (_lastTools.length >= 3) {
160
+ const last3 = _lastTools.slice(-3);
161
+ const sameToolName = last3.every(t => t.name === last3[0].name);
162
+ if (sameToolName) {
163
+ const argSigs = last3.map(t => {
164
+ try { return JSON.stringify(t.args || {}); } catch { return ""; }
165
+ });
166
+ const sameArgs = argSigs[0] === argSigs[1] && argSigs[1] === argSigs[2];
167
+ if (sameArgs) {
168
+ hints.push("HINT: You called the same tool 3 times in a row with the SAME arguments. The state did not change. Try a different approach or check if the operation is actually needed.");
169
+ }
170
+ }
171
+ }
172
+
173
+ // --- Knowledge base lookup ---
174
+ if (_knowledgeSearchFn && hints.length === 0) {
175
+ // Auto-search knowledge when model seems stuck (same tool with same args 2+ times).
176
+ // Bug fix 2026-04-12: previously checked name only, triggering KB search
177
+ // on every legitimate sequential operation.
178
+ const last2 = _lastTools.slice(-2);
179
+ const sameCall = last2.length === 2 && last2[0].name === last2[1].name &&
180
+ (() => { try { return JSON.stringify(last2[0].args || {}) === JSON.stringify(last2[1].args || {}); } catch { return false; } })();
181
+ if (sameCall) {
182
+ try {
183
+ const appName = detectApp(toolName, args, result);
184
+ if (appName) {
185
+ const kb = _knowledgeSearchFn(appName);
186
+ if (kb) {
187
+ hints.push(`KNOWLEDGE (${appName}): ${kb}`);
188
+ }
189
+ }
190
+ } catch (e) {
191
+ log.debug("knowledge search failed", { error: e.message });
192
+ }
193
+ }
194
+ }
195
+
196
+ if (hints.length === 0) return null;
197
+
198
+ // 4-level escalation: hint → warning → re-plan → hard stop
199
+ // Use tool name + first hint prefix as key (avoids collision between different hints with same prefix)
200
+ const hintKey = toolName + ":" + hints[0].slice(0, 40);
201
+ _hintCounts[hintKey] = (_hintCounts[hintKey] || 0) + 1;
202
+ const count = _hintCounts[hintKey];
203
+
204
+ let hint;
205
+ if (count >= 4) {
206
+ // Level 4: HARD STOP
207
+ hint = `[SUPERVISOR OVERRIDE] This hint was given ${count} times and ignored. STOP — the agent cannot recover from this pattern.\n` + hints.join("\n");
208
+ log.warn("supervisor-hard-stop", { tool: toolName, count, key: hintKey });
209
+ } else if (count >= 3) {
210
+ // Level 3: Force re-plan
211
+ hint = `[SUPERVISOR RE-PLAN] You ignored this hint ${count} times. STOP current approach. Create a NEW plan with a DIFFERENT strategy. Do NOT retry the same approach.\n` + hints.join("\n");
212
+ log.warn("supervisor-replan", { tool: toolName, count, key: hintKey });
213
+ } else if (count >= 2) {
214
+ // Level 2: Warning
215
+ hint = `[SUPERVISOR WARNING — repeated] ` + hints.join("\n");
216
+ log.info("supervisor-repeat", { tool: toolName, count, key: hintKey });
217
+ } else {
218
+ // Level 1: Hint
219
+ hint = hints.join("\n");
220
+ log.info("supervisor-hint", { tool: toolName, hints: hints.length, preview: hint.slice(0, 100) });
221
+ }
222
+ return hint;
223
+ }
224
+
225
+ /**
226
+ * Check if agent's text response is a mid-task description instead of action.
227
+ * Called with the model's text response (not tool results).
228
+ * @returns {string|null} hint to inject, or null
229
+ */
230
+ export function checkMidTaskDescription(text, hadToolCalls) {
231
+ if (!_enabled || hadToolCalls) return null;
232
+ const lower = (text || "").toLowerCase();
233
+ const descriptionPatterns = [
234
+ /\bi will (now |)(save|create|open|search|find|read|check|run|execute|navigate|download|fetch|write)\b/,
235
+ /\bnow i('ll| will| should| need to)\b/,
236
+ /\bnext i (need to|will|should)\b/,
237
+ /\blet me\b.*\b(first|start|begin)\b/,
238
+ /\bi('m going to|'ll)\b.*\b(save|create|open|search|find|read|check|run|execute)\b/,
239
+ /\bi need to\b.*\b(first|next|then)\b/,
240
+ /\bmy next step\b/, /\bthe next step\b/,
241
+ ];
242
+ const isDescription = descriptionPatterns.some(p => p.test(lower));
243
+ if (isDescription) {
244
+ log.info("mid-task-description", { text: lower.slice(0, 80) });
245
+ return "HINT: Don't describe what you'll do — just call the tools. The user sees your progress through tool activity, not text descriptions.";
246
+ }
247
+ return null;
248
+ }
249
+
250
+ /**
251
+ * Is this tool one that actually looks at the screen?
252
+ *
253
+ * The three dialog rules used to match on the result string alone, with no
254
+ * check that anything was observed. So the word "recover" appearing in a file
255
+ * being read, a test run's output, or a page fetched from the web produced
256
+ * "Document Recovery dialog detected. Press alt+d to DISCARD" — an
257
+ * instruction to press a key that throws work away, triggered by a word in
258
+ * prose. Every other rule in this file already keyed on the tool or its
259
+ * arguments, which is what makes a hint about an action rather than a guess
260
+ * about a word.
261
+ *
262
+ * Only these tools return what is on the screen, so only these can be evidence
263
+ * that a dialog is up. desktop_click and desktop_type are deliberately absent:
264
+ * they act on the screen without reporting it, so their result saying
265
+ * "Recovery" is the model talking, not a screen.
266
+ *
267
+ * The screenbox MCP tools count: they observe the same screen and are a
268
+ * legitimate source of this evidence, so the test is about capability rather
269
+ * than spelling.
270
+ */
271
+ function _isScreenObserver(name) {
272
+ if (typeof name !== "string") return false;
273
+ return (
274
+ name === "desktop_look" ||
275
+ name === "desktop_screenshot" ||
276
+ name === "desktop_screenshot_gui" ||
277
+ name === "desktop_shell" ||
278
+ name.startsWith("mcp_screenbox_")
279
+ );
280
+ }
281
+
282
+ /**
283
+ * Detect if a tool call is a write/create/modify action.
284
+ */
285
+ function _isWriteAction(name, args) {
286
+ // File write tools
287
+ if (name === "write_file" || name === "edit_file") return true;
288
+ // Shell commands that write
289
+ if (name.includes("shell") || name === "run_command") {
290
+ const cmd = args?.command || "";
291
+ if (cmd.match(/\b(echo|cat|printf|tee)\b.*[>|]/) || cmd.includes(">>")) return true;
292
+ if (cmd.match(/\b(cp|mv|mkdir|touch|curl\s.*-o)\b/)) return true;
293
+ if (cmd.includes("apt-get install") || cmd.includes("pip install")) return true;
294
+ }
295
+ // Desktop: type, key (Ctrl+S = save), clipboard paste
296
+ if (name.includes("type") && (args?.text || "").length > 20) return true;
297
+ if (name.includes("key")) {
298
+ const keys = (args?.keys || args?.key || "").toLowerCase();
299
+ if (keys === "ctrl+s" || keys === "ctrl+shift+s" || keys === "return" || keys === "enter") return true;
300
+ }
301
+ // Chrome: navigate is not write, but type into form + submit is
302
+ if (name.includes("chrome") && args?.action === "eval") return true;
303
+ return false;
304
+ }
305
+
306
+ /**
307
+ * Detect if a tool call verifies a previous write action.
308
+ */
309
+ function _isVerifyAction(name, args, prevArgs) {
310
+ // Read file = verify
311
+ if (name === "read_file" || name === "list_directory") return true;
312
+ if (name.includes("glob") || name.includes("search_in_files")) return true;
313
+ // Shell: cat, ls, head, tail, test -f
314
+ if (name.includes("shell") || name === "run_command") {
315
+ const cmd = args?.command || "";
316
+ if (cmd.match(/\b(cat|head|tail|less|wc|ls|test|stat|file|md5sum|sha256sum)\b/)) return true;
317
+ if (cmd.includes("echo $?") || cmd.includes("$?")) return true;
318
+ }
319
+ // Screenshot = visual verify
320
+ if (name.includes("screenshot") || name.includes("look")) return true;
321
+ // Chrome: page_read, page_map = verify
322
+ if (name.includes("chrome")) {
323
+ const action = args?.action || "";
324
+ if (["page_read", "page_map", "view_read", "page_info"].includes(action)) return true;
325
+ }
326
+ // Think = reasoning about result, counts as verify
327
+ if (name === "think") return true;
328
+ return false;
329
+ }
330
+
331
+ function detectApp(toolName, args, result) {
332
+ const resultStr = typeof result === "string" ? result : JSON.stringify(result);
333
+ const cmd = args?.command || args?.action || "";
334
+ // Check tool result content
335
+ if (resultStr.includes("LibreOffice") || resultStr.includes("libreoffice") || resultStr.includes("Calc")) return "libreoffice-calc";
336
+ if (resultStr.includes("Mousepad") || resultStr.includes("mousepad")) return "mousepad";
337
+ if (resultStr.includes("Chrome") || resultStr.includes("Chromium") || resultStr.includes("chrome")) return "chrome";
338
+ if (resultStr.includes("GIMP") || resultStr.includes("gimp")) return "gimp";
339
+ if (resultStr.includes("Thunar") || resultStr.includes("thunar")) return "file-manager";
340
+ if (resultStr.includes("Terminal") || resultStr.includes("terminal")) return "terminal";
341
+ // Check command content
342
+ if (cmd.includes("mousepad")) return "mousepad";
343
+ if (cmd.includes("soffice") || cmd.includes("libreoffice")) return "libreoffice-calc";
344
+ if (cmd.includes("chromium") || cmd.includes("chrome")) return "chrome";
345
+ if (cmd.includes("gimp")) return "gimp";
346
+ // Check tool name for desktop tools
347
+ if (toolName.includes("chrome")) return "chrome";
348
+ return null;
349
+ }
350
+
351
+ /**
352
+ * Conditional reflection — triggers only when context is at risk.
353
+ * Replaces always-on mandatory reflection. Fires on:
354
+ * - Large tool result (>3k chars) — model likely to lose focus
355
+ * - Error in last tool — model needs to reconsider approach
356
+ * - 5+ tool calls without reflection — periodic checkpoint
357
+ *
358
+ * @param {object} opts - { lastToolResult, lastToolName, apiCallCount, planStep }
359
+ * @returns {string|null} reflection message or null
360
+ */
361
+ let _callsSinceReflection = 0;
362
+
363
+ export function evaluateReflection({ lastToolResult, lastToolName, apiCallCount, planStep }) {
364
+ if (!_enabled) return null;
365
+ _callsSinceReflection++;
366
+
367
+ const resultSize = (lastToolResult || "").length;
368
+ const isError = (lastToolResult || "").includes("error") || (lastToolResult || "").includes("Error");
369
+ const isLargeResult = resultSize > 3000;
370
+ const needsCheckpoint = _callsSinceReflection >= 5;
371
+
372
+ // Check if agent is repeating same action (stuck).
373
+ // Bug fix 2026-04-12: previously only checked name, firing "repeated action"
374
+ // reflections on every legitimate sequential multi-file read. Now requires
375
+ // BOTH same name AND same args — real stuck behavior, not busy work.
376
+ let isStuck = false;
377
+ if (_lastTools.length >= 2) {
378
+ const [a, b] = [_lastTools[_lastTools.length - 2], _lastTools[_lastTools.length - 1]];
379
+ if (a.name === b.name) {
380
+ try {
381
+ isStuck = JSON.stringify(a.args || {}) === JSON.stringify(b.args || {});
382
+ } catch { isStuck = false; }
383
+ }
384
+ }
385
+
386
+ if (!isLargeResult && !isError && !needsCheckpoint && !isStuck) return null;
387
+
388
+ _callsSinceReflection = 0;
389
+
390
+ const trigger = isStuck ? "repeated action" : isLargeResult ? "large result" : isError ? "error" : "checkpoint";
391
+ const summary = resultSize > 200 ? lastToolResult.slice(0, 200) + "..." : lastToolResult || "none";
392
+
393
+ log.info("reflection-triggered", { trigger, tool: lastToolName, resultSize });
394
+
395
+ // Extract EXPECT from previous assistant message if present
396
+ const expectMatch = _lastExpect;
397
+ const evalLine = expectMatch
398
+ ? `Your EXPECT was: "${expectMatch}". Did the result match? If not — your action failed. Change approach.`
399
+ : "Did your last action achieve its goal? Check the result carefully.";
400
+
401
+ const question = isStuck
402
+ ? "You repeated the same action. The state did NOT change. Try a DIFFERENT tool or method."
403
+ : evalLine;
404
+
405
+ const parts = [
406
+ `[REFLECTION — ${trigger}]`,
407
+ `Last: ${lastToolName} (${resultSize} chars)`,
408
+ planStep || "",
409
+ question,
410
+ ].filter(Boolean);
411
+
412
+ return parts.join("\n");
413
+ }
414
+
415
+ /**
416
+ * Track EXPECT from assistant's text response.
417
+ * Called from agent.js when assistant returns text + tool calls.
418
+ */
419
+ export function trackExpect(assistantText) {
420
+ if (!assistantText) return;
421
+ const match = assistantText.match(/EXPECT:\s*(.+?)(?:\n|$)/i);
422
+ _lastExpect = match ? match[1].trim() : null;
423
+ }
424
+
425
+ export function resetSupervisor() {
426
+ _lastTools = [];
427
+ _hintCounts = {};
428
+ _callsSinceReflection = 0;
429
+ _lastExpect = null;
430
+ }