flint-agent 1.14.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (171) hide show
  1. package/.env.example +108 -0
  2. package/CHANGELOG.md +55 -0
  3. package/FEATURES.md +298 -0
  4. package/LICENSE +21 -0
  5. package/README.md +435 -0
  6. package/bin/flint.js +47 -0
  7. package/config/classifier-prompt.md +218 -0
  8. package/config/models-curated.json +4 -0
  9. package/config/providers.json +74 -0
  10. package/package.json +92 -0
  11. package/patches/ink+6.8.0.patch +78 -0
  12. package/profiles/desktop.md +65 -0
  13. package/profiles/generic.md +20 -0
  14. package/profiles/marketer.md +20 -0
  15. package/profiles/profiles.json +34 -0
  16. package/profiles/ux-reviewer.md +25 -0
  17. package/src/agent/agent.js +1743 -0
  18. package/src/agent/auto.js +346 -0
  19. package/src/agent/backoff.js +143 -0
  20. package/src/agent/compression.js +310 -0
  21. package/src/agent/content-resolver.js +180 -0
  22. package/src/agent/flow-controller.js +309 -0
  23. package/src/agent/intent-manifest.js +231 -0
  24. package/src/agent/intent-timeout.js +46 -0
  25. package/src/agent/intent.js +633 -0
  26. package/src/agent/knowledge.js +114 -0
  27. package/src/agent/learning.js +180 -0
  28. package/src/agent/modes.js +187 -0
  29. package/src/agent/outcome-ask.js +91 -0
  30. package/src/agent/project-context.js +76 -0
  31. package/src/agent/prompt-budget.js +117 -0
  32. package/src/agent/reflection-extractor.js +140 -0
  33. package/src/agent/steering.js +86 -0
  34. package/src/agent/supervisor.js +430 -0
  35. package/src/agent/swap.js +443 -0
  36. package/src/agent/system-prompt.js +446 -0
  37. package/src/agent/time-stamp.js +48 -0
  38. package/src/agent/tool-guard.js +201 -0
  39. package/src/agent/toolcall-text.js +162 -0
  40. package/src/agent/usage.js +297 -0
  41. package/src/agent/vision.js +94 -0
  42. package/src/agent/watchdog.js +139 -0
  43. package/src/agent/workspace-changes.js +177 -0
  44. package/src/api/address.js +14 -0
  45. package/src/api/client.js +280 -0
  46. package/src/api/server.js +535 -0
  47. package/src/api/stream-pipe.js +113 -0
  48. package/src/app-state.js +39 -0
  49. package/src/bootstrap.js +501 -0
  50. package/src/bus/drain-loop.js +497 -0
  51. package/src/bus/index.js +270 -0
  52. package/src/bus/plugins.js +65 -0
  53. package/src/child-idle.js +14 -0
  54. package/src/cli.js +118 -0
  55. package/src/commands/commands.js +1297 -0
  56. package/src/commands/registry.js +132 -0
  57. package/src/components/App.js +491 -0
  58. package/src/components/CarefulMenu.js +145 -0
  59. package/src/components/HistoryWriter.js +86 -0
  60. package/src/components/LineInput.js +69 -0
  61. package/src/components/LiveZone.js +294 -0
  62. package/src/components/OverlayMenu.js +179 -0
  63. package/src/components/SystemPanel.js +156 -0
  64. package/src/components/Table.js +54 -0
  65. package/src/config.js +249 -0
  66. package/src/free-models.js +230 -0
  67. package/src/index.js +1111 -0
  68. package/src/input-handler.js +13 -0
  69. package/src/input-text.js +123 -0
  70. package/src/launcher.js +129 -0
  71. package/src/logging/api-log.js +95 -0
  72. package/src/logging/chat-log-follower.js +113 -0
  73. package/src/logging/chat-log.js +15 -0
  74. package/src/logging/log-collector.js +182 -0
  75. package/src/logging/logger.js +112 -0
  76. package/src/logging/tool-log.js +20 -0
  77. package/src/mcp-client.js +314 -0
  78. package/src/memory/conversation-digest.js +113 -0
  79. package/src/memory/extract-facts.js +98 -0
  80. package/src/memory/facts.js +181 -0
  81. package/src/memory/inbox.js +63 -0
  82. package/src/memory/markdown.js +38 -0
  83. package/src/memory/patterns.js +185 -0
  84. package/src/memory/project.js +66 -0
  85. package/src/memory/reflections.js +74 -0
  86. package/src/memory/retrieval.js +84 -0
  87. package/src/memory/rules.js +105 -0
  88. package/src/memory/session-facts.js +125 -0
  89. package/src/memory/skills.js +191 -0
  90. package/src/memory/sqlite-store.js +653 -0
  91. package/src/memory/store.js +208 -0
  92. package/src/memory/tools.js +196 -0
  93. package/src/memory/user-model.js +86 -0
  94. package/src/message-handler.js +775 -0
  95. package/src/model-check.js +218 -0
  96. package/src/plugins/loader.js +120 -0
  97. package/src/plugins/manager.js +88 -0
  98. package/src/production-env.js +22 -0
  99. package/src/profiles.js +42 -0
  100. package/src/providers/adapters/anthropic.js +270 -0
  101. package/src/providers/adapters/openai.js +120 -0
  102. package/src/providers/keys-dpapi.js +41 -0
  103. package/src/providers/keys-fallback.js +31 -0
  104. package/src/providers/keys.js +132 -0
  105. package/src/providers/models.js +154 -0
  106. package/src/providers/registry.js +56 -0
  107. package/src/providers/state.js +56 -0
  108. package/src/registry.js +96 -0
  109. package/src/restart.js +29 -0
  110. package/src/sandbox/backend.js +130 -0
  111. package/src/security/api-auth.js +132 -0
  112. package/src/security/audit.js +98 -0
  113. package/src/security/child-policy.js +41 -0
  114. package/src/security/command-guard.js +173 -0
  115. package/src/security/content-fence.js +250 -0
  116. package/src/security/content-validator.js +132 -0
  117. package/src/security/index.js +143 -0
  118. package/src/security/network-guard.js +126 -0
  119. package/src/security/pairing.js +180 -0
  120. package/src/security/path-guard.js +140 -0
  121. package/src/security/persona-guard.js +67 -0
  122. package/src/security/policies.js +452 -0
  123. package/src/security/safety-constants.js +34 -0
  124. package/src/security/watchdog.js +107 -0
  125. package/src/sessions.js +130 -0
  126. package/src/spend.js +97 -0
  127. package/src/startup-watchdog.js +59 -0
  128. package/src/stdio/args.js +71 -0
  129. package/src/stdio/guard.js +59 -0
  130. package/src/stdio/protocol.js +167 -0
  131. package/src/stdio/run.js +106 -0
  132. package/src/stdio/session.js +180 -0
  133. package/src/store/agent-slice.js +306 -0
  134. package/src/store/dataset-slice.js +73 -0
  135. package/src/store/index.js +22 -0
  136. package/src/store/process-slice.js +135 -0
  137. package/src/store/session-slice.js +191 -0
  138. package/src/store/ui-slice.js +119 -0
  139. package/src/tasks/db.js +184 -0
  140. package/src/tasks/queries.js +589 -0
  141. package/src/tools/agent-tools.js +473 -0
  142. package/src/tools/checkpoint.js +152 -0
  143. package/src/tools/command-approvals.js +180 -0
  144. package/src/tools/dataset.js +50 -0
  145. package/src/tools/filesystem.js +682 -0
  146. package/src/tools/inbox-tools.js +48 -0
  147. package/src/tools/mesh.js +135 -0
  148. package/src/tools/own-env.js +136 -0
  149. package/src/tools/permissions.js +681 -0
  150. package/src/tools/plugin-tools.js +123 -0
  151. package/src/tools/process-tools.js +595 -0
  152. package/src/tools/registry.js +307 -0
  153. package/src/tools/swap-tools.js +72 -0
  154. package/src/tools/system.js +662 -0
  155. package/src/tools/tasks.js +532 -0
  156. package/src/tools/tool-search.js +171 -0
  157. package/src/ui/header.js +140 -0
  158. package/src/ui/input-cursor.js +23 -0
  159. package/src/ui/last-line.js +25 -0
  160. package/src/ui/line-edit.js +135 -0
  161. package/src/ui/output.js +399 -0
  162. package/src/ui/paste-tokens.js +131 -0
  163. package/src/ui/prompt-attention.js +134 -0
  164. package/src/ui/render-options.js +13 -0
  165. package/src/ui/replay.js +94 -0
  166. package/src/ui/splash.js +49 -0
  167. package/src/ui/status-level.js +36 -0
  168. package/src/ui/tool-ledger.js +203 -0
  169. package/src/ui/window-title.js +150 -0
  170. package/src/update.js +205 -0
  171. package/system.md +63 -0
@@ -0,0 +1,633 @@
1
+ // Intent classifier — one cheap LLM call BEFORE the main agent loop.
2
+ //
3
+ // The classifier reads session context (summary + recent messages + new message)
4
+ // PLUS the live list of available tools from the registry, and returns a manifest
5
+ // with:
6
+ // - intent class (creative_text, memory_read, shell_multi, ...)
7
+ // - concrete tool names picked from the registry
8
+ // - max_steps and expected output shape
9
+ //
10
+ // Tool names are NOT hardcoded in the manifest — they come from the live registry
11
+ // at call time. Adding a tool in the registry makes it automatically eligible for
12
+ // the classifier without any code changes here.
13
+
14
+ import { getSpendLevel, spendSettings } from "../spend.js";
15
+ import { createHash } from "node:crypto";
16
+ import { intentTimeoutMs } from "./intent-timeout.js";
17
+ import { appendFileSync, mkdirSync, statSync, renameSync, existsSync, readFileSync } from "node:fs";
18
+ import { homedir } from "node:os";
19
+ import { join, dirname } from "node:path";
20
+ import { fileURLToPath } from "node:url";
21
+ import { config } from "../config.js";
22
+ import { createLogger } from "../logging/logger.js";
23
+ import { INTENTS, formatIntentCatalog, resolveIntent } from "./intent-manifest.js";
24
+ import { chatCompletion } from "../api/client.js";
25
+ import { CORE_TOOLS, TOOL_SEARCH_NAME, loadedToolNames } from "../tools/tool-search.js";
26
+
27
+ // Load the classifier prompt from config/classifier-prompt.md. The file is the
28
+ // source of truth for classifier behaviour — edits there propagate on restart
29
+ // with no code change. The file has three parts separated by `---`:
30
+ // the frontmatter header, the main SYSTEM+rules body, and the SCHEMA block.
31
+ // We extract the two we need (system body + schema block) at module init.
32
+ const _thisFile = fileURLToPath(import.meta.url);
33
+ const _promptPath = join(dirname(_thisFile), "..", "..", "config", "classifier-prompt.md");
34
+ const _promptRaw = readFileSync(_promptPath, "utf8");
35
+ const _parts = _promptRaw.split(/^---\s*$/m).map(s => s.trim()).filter(Boolean);
36
+ // Expected: [header, body, schema]. Fall back to full file if split fails.
37
+ const _body = _parts[1] || _promptRaw;
38
+ const _schemaSection = _parts[2] || "";
39
+ const _schemaMatch = _schemaSection.match(/```json\s*([\s\S]*?)```/);
40
+
41
+ const log = createLogger("intent");
42
+
43
+ // ── LRU cache for classifier results ──
44
+ // Same user message + same recent messages + same tool list = same classification.
45
+ // No point paying for the LLM call twice in a short window.
46
+ const CACHE_MAX = 50;
47
+
48
+ // With the classifier off, up to this many MCP tools are handed over whole;
49
+ // more go through tool_search (fallbackManifest). Set by the spend mode
50
+ // (spend.js: economy 10, normal 30, generous 200); FLINT_MCP_INLINE_MAX wins.
51
+ export function mcpInlineMax(env = process.env, level = getSpendLevel(env)) {
52
+ const v = parseInt(env.FLINT_MCP_INLINE_MAX, 10);
53
+ return Number.isFinite(v) ? v : spendSettings(level).mcpInlineMax;
54
+ }
55
+ const CACHE_TTL_MS = 5 * 60 * 1000;
56
+ const _cache = new Map(); // key → {manifest, ts}
57
+ // The classifier per-attempt budget is read from INTENT_TIMEOUT_MS via its own
58
+ // module (src/agent/intent-timeout.js) so it can be tested without importing the
59
+ // config, the API client and the tool manifest. A hard deadline here is what made a
60
+ // test depend on machine load: classifyIntent catches its own errors and falls back to
61
+ // the all-tools manifest, so a blown deadline looks like a valid classification.
62
+ // See that file.
63
+
64
+ function cacheKey(newMessage, recentMessages, availableTools) {
65
+ const h = createHash("sha256");
66
+ h.update(newMessage || "");
67
+ h.update("|");
68
+ for (const m of (recentMessages || [])) {
69
+ h.update(m.role || "");
70
+ h.update(":");
71
+ h.update(typeof m.content === "string" ? m.content : JSON.stringify(m.content || ""));
72
+ h.update("\n");
73
+ }
74
+ h.update("|");
75
+ for (const t of (availableTools || [])) {
76
+ h.update((t.function?.name || t.name || "") + ",");
77
+ }
78
+ return h.digest("hex");
79
+ }
80
+
81
+ function cacheGet(key) {
82
+ const entry = _cache.get(key);
83
+ if (!entry) return null;
84
+ if (Date.now() - entry.ts > CACHE_TTL_MS) {
85
+ _cache.delete(key);
86
+ return null;
87
+ }
88
+ // LRU: move to end
89
+ _cache.delete(key);
90
+ _cache.set(key, entry);
91
+ return entry.manifest;
92
+ }
93
+
94
+ function cacheSet(key, manifest) {
95
+ if (_cache.size >= CACHE_MAX) {
96
+ const oldest = _cache.keys().next().value;
97
+ if (oldest !== undefined) _cache.delete(oldest);
98
+ }
99
+ _cache.set(key, { manifest, ts: Date.now() });
100
+ }
101
+
102
+ // ── Prompt injection guard ──
103
+ // A user whose message mentions classifier output keywords could try to disable
104
+ // tool access by steering classification toward text-only intents. We detect
105
+ // obvious markers and, if present, skip the classifier entirely and return the
106
+ // full-tool fallback. The attacker cannot narrow the tool surface this way.
107
+ const INJECTION_PATTERNS = [
108
+ /\bclassif(y|ied|ication)\s+as\b/i,
109
+ /\bintent\s*[:=]\s*["']?[a-z_]+/i,
110
+ /\btools\s*[:=]\s*\[/i,
111
+ /\bneedsTools\b/i,
112
+ /\bmax_?steps\s*[:=]/i,
113
+ ];
114
+
115
+ function looksLikeInjection(text) {
116
+ if (!text || typeof text !== "string") return false;
117
+ return INJECTION_PATTERNS.some(re => re.test(text));
118
+ }
119
+
120
+ // ── Shadow log of classification decisions ──
121
+ // Every classification is appended so we can later measure accuracy against
122
+ // E2E outcomes. No runtime cost beyond one async-fs append.
123
+ const DECISIONS_FILE = join(homedir(), ".flint", "intent-decisions.jsonl");
124
+ const DECISIONS_MAX_BYTES = 10 * 1024 * 1024; // 10 MB → archive + start fresh
125
+ let _logCallCount = 0;
126
+
127
+ // Rotate the decisions log if it crosses the size threshold. Archive name is
128
+ // `intent-decisions-YYYY-MM-DD.jsonl`; if an archive for today already exists
129
+ // (rare), the timestamp is appended to keep history distinct. Best-effort —
130
+ // failures are swallowed, logging continues on the (unrotated) file.
131
+ function maybeRotateDecisions() {
132
+ try {
133
+ if (!existsSync(DECISIONS_FILE)) return;
134
+ const st = statSync(DECISIONS_FILE);
135
+ if (st.size < DECISIONS_MAX_BYTES) return;
136
+ const date = new Date().toISOString().slice(0, 10);
137
+ let archive = DECISIONS_FILE.replace(/\.jsonl$/, `-${date}.jsonl`);
138
+ if (existsSync(archive)) {
139
+ const stamp = new Date().toISOString().replace(/[:.]/g, "-");
140
+ archive = DECISIONS_FILE.replace(/\.jsonl$/, `-${stamp}.jsonl`);
141
+ }
142
+ renameSync(DECISIONS_FILE, archive);
143
+ } catch {}
144
+ }
145
+
146
+ function logDecision(entry) {
147
+ try {
148
+ mkdirSync(join(homedir(), ".flint"), { recursive: true });
149
+ // Gate the size check to avoid a statSync on every classification; once
150
+ // per 100 decisions is enough (max 100 extra log lines before rotation).
151
+ if (_logCallCount++ % 100 === 0) maybeRotateDecisions();
152
+ appendFileSync(DECISIONS_FILE, JSON.stringify(entry) + "\n", "utf-8");
153
+ } catch {
154
+ // Logging is best-effort — never break classification on log failure
155
+ }
156
+ }
157
+
158
+ // Classifier system prompt — body loaded from config/classifier-prompt.md at
159
+ // module init. The markdown file is the source of truth for classifier rules.
160
+ // The live intent catalog (from intent-manifest.js) is appended here so that
161
+ // adding/removing intents in code is automatically reflected without editing
162
+ // the markdown. AVAILABLE_TOOLS is appended per-call in buildUserPrompt.
163
+ const CLASSIFIER_SYSTEM = `${_body}\n\n## INTENT_CLASSES\n\n${formatIntentCatalog()}\n`;
164
+
165
+ // Schema hint — extracted from the fenced ```json block in the SCHEMA section
166
+ // of config/classifier-prompt.md. Fallback to a minimal inline schema if the
167
+ // markdown file's SCHEMA block is missing or malformed.
168
+ const SCHEMA_HINT = (_schemaMatch && _schemaMatch[1].trim()) || `{
169
+ "intent": "<one of the class names above>",
170
+ "tools": ["<tool_name_from_the_list>", ...],
171
+ "assessment": "<normal|dangerous|overscoped|ambiguous|nonsensical|impossible>",
172
+ "requires_prior_tool_call": [],
173
+ "user_wants": "<one sentence paraphrase>",
174
+ "reason": "<why this class>"
175
+ }`;
176
+
177
+ /**
178
+ * Classify a user request in its conversation context.
179
+ *
180
+ * @param {object} ctx
181
+ * @param {string} ctx.newMessage — the new user message (required)
182
+ * @param {Array} ctx.availableTools — live tool defs from registry [{type,function:{name,description}},...]
183
+ * @param {string} [ctx.sessionSummary] — compact summary of the session so far (optional)
184
+ * @param {Array} [ctx.recentMessages] — last few {role, content} messages (optional)
185
+ * @returns {Promise<object>} manifest: {intent, tools, max_steps, expected, user_wants, reason, fallback?}
186
+ */
187
+ export async function classifyIntent(ctx) {
188
+ const { newMessage, sessionSummary, recentMessages, availableTools } = ctx;
189
+
190
+ // Headless mode bypass: when Flint runs as a headless subprocess (e.g. the
191
+ // SWE-bench runner, or any CLI `--headless --task ...` invocation) the
192
+ // caller's prompt is usually a long, complex code-editing brief that the
193
+ // classifier misreads as `file_read` and caps at 3 iterations. There is no
194
+ // interactive user to correct the classifier. Give the agent full tool
195
+ // access and let config.maxIterations govern the budget instead.
196
+ if (config.headless) {
197
+ return fallbackManifest("headless mode — classifier bypassed", availableTools);
198
+ }
199
+
200
+ // A/B switch: no classifier call at all, the built-in tools every turn, no
201
+ // assessment gate. Across 2026-09-26 the classifier's tool picks caused three
202
+ // of Flint's five losses (no web on a SWE task, every tool taken away from
203
+ // an explicit delete, three operator tools for a browser task). This measures
204
+ // what the turn does without it.
205
+ if (process.env.FLINT_NO_CLASSIFIER === "1") {
206
+ return fallbackManifest("classifier disabled (FLINT_NO_CLASSIFIER)", availableTools);
207
+ }
208
+
209
+ // FLINT_TOOL_MODE=search: an explicit choice, so it comes before the
210
+ // classifier-off fallback below. No classifier call. The core set, whatever
211
+ // tool_search has loaded so far this session, and tool_search itself.
212
+ if (process.env.FLINT_TOOL_MODE === "search") {
213
+ const present = new Set((availableTools || []).map(t => t.function?.name || t.name).filter(Boolean));
214
+ const names = [...CORE_TOOLS, ...loadedToolNames(), TOOL_SEARCH_NAME].filter(n => present.has(n));
215
+ return { ...fallbackManifest("tool search mode", availableTools), tools: [...new Set(names)], fallbackScope: "search" };
216
+ }
217
+
218
+ // No INTENT_MODEL: the classifier is off, not an error.
219
+ if (!config.intentModel) {
220
+ return fallbackManifest("classifier disabled (INTENT_MODEL not set)", availableTools);
221
+ }
222
+
223
+ if (!newMessage || typeof newMessage !== "string") {
224
+ return fallbackManifest("empty or non-text message", availableTools);
225
+ }
226
+
227
+ // Prompt injection guard — if the user message contains classifier-control
228
+ // keywords, skip classification entirely and give the agent full tool access.
229
+ // An attacker cannot use this path to narrow the tool surface.
230
+ if (looksLikeInjection(newMessage)) {
231
+ log.warn("intent: injection pattern detected, using full-tool fallback");
232
+ const fallback = fallbackManifest("injection pattern in user message", availableTools);
233
+ logDecision({
234
+ ts: new Date().toISOString(),
235
+ message: newMessage.slice(0, 200),
236
+ intent: fallback.intent,
237
+ tools: fallback.tools.length,
238
+ assessment: fallback.assessment,
239
+ fallback: true,
240
+ reason: "injection_guard",
241
+ });
242
+ return fallback;
243
+ }
244
+
245
+ // Cache: identical input in a short window reuses the previous decision.
246
+ const key = cacheKey(newMessage, recentMessages, availableTools);
247
+ const cached = cacheGet(key);
248
+ if (cached) {
249
+ log.debug("intent: cache hit", { intent: cached.intent });
250
+ return cached;
251
+ }
252
+
253
+ // Build the user-side prompt for the classifier.
254
+ //
255
+ // Classifier stays lightweight on purpose: intent catalog + available
256
+ // tools + user message + schema. Project rules (FLINT.md), memory
257
+ // layers, profile, OS hints etc. are the agent loop's concern — not
258
+ // the classifier's. Loading them here bloats the prompt (~4K chars of
259
+ // project rules for a ~300-token decision) without improving tool
260
+ // picks. The agent loop still gets FLINT.md and applies its guidance
261
+ // when forming the actual tool call arguments.
262
+ //
263
+ // ORDER MATTERS — LLMs have strong primacy + recency bias and a well-known
264
+ // "lost in the middle" problem. The three things the classifier MUST
265
+ // consider carefully are (a) the user's new message, (b) the tool catalog
266
+ // it picks from, (c) the JSON schema for the response. Those go at the
267
+ // start (primacy) or the very end (recency). Conversation history is
268
+ // deprioritised context — it goes in the middle, wrapped in a block that
269
+ // explicitly marks it as "for disambiguation only, do not let it drive
270
+ // tool choice". Previous layout had the new message 2nd-to-last and the
271
+ // tool catalog in the middle — observed 2026-04-21: when preceding turns
272
+ // in the session were about shell/ssh/docker, the classifier carried that
273
+ // bias forward and routed "sum tags in planner" → shell_command instead
274
+ // of the mcp_planner_* tools that were right there in the catalog.
275
+ const parts = [];
276
+
277
+ // 1) NEW MESSAGE first — highest attention.
278
+ parts.push("NEW MESSAGE:\n" + newMessage.slice(0, 1000));
279
+
280
+ // 2) AVAILABLE TOOLS right after — the classifier must pick from this list.
281
+ const toolList = Array.isArray(availableTools) ? availableTools : [];
282
+ if (toolList.length) {
283
+ const toolLines = toolList
284
+ .map(t => {
285
+ const name = t.function?.name || t.name;
286
+ const desc = (t.function?.description || t.description || "").replace(/\s+/g, " ").slice(0, 120);
287
+ return name ? ` ${name} — ${desc}` : null;
288
+ })
289
+ .filter(Boolean)
290
+ .join("\n");
291
+ parts.push("AVAILABLE TOOLS:\n" + toolLines);
292
+ }
293
+
294
+ // 3) Session context in the middle, framed as weak context so the model
295
+ // treats it as disambiguation help, not as a driver of tool choice.
296
+ if (sessionSummary && sessionSummary.trim()) {
297
+ parts.push(
298
+ "SESSION SUMMARY (context only — do not let this override tool choice from the new message):\n" +
299
+ sessionSummary.trim().slice(0, 1000),
300
+ );
301
+ }
302
+ if (Array.isArray(recentMessages) && recentMessages.length) {
303
+ const historyLines = [];
304
+ for (const m of recentMessages.slice(-5)) {
305
+ if (!m || !m.role) continue;
306
+ const text = typeof m.content === "string"
307
+ ? m.content
308
+ : Array.isArray(m.content)
309
+ ? m.content.map(c => c.text || "").join(" ")
310
+ : "";
311
+ if (!text.trim()) continue;
312
+ historyLines.push(`${m.role}: ${text.slice(0, 300)}`);
313
+ }
314
+ if (historyLines.length) {
315
+ parts.push(
316
+ "RECENT MESSAGES (context only — classify the new message on its own merits, do not carry bias from previous turns):\n" +
317
+ historyLines.join("\n"),
318
+ );
319
+ }
320
+ }
321
+
322
+ // 4) Schema at the very end — the last thing the model reads before
323
+ // generating its response, for maximum structural compliance.
324
+ parts.push("Classify the NEW MESSAGE above. Return JSON matching:\n" + SCHEMA_HINT);
325
+
326
+ const userPrompt = parts.join("\n\n");
327
+
328
+ // Env-gated snapshot of the classifier prompt. Writes one JSON file per
329
+ // classify call to ~/.flint/classifier-prompt-snapshots/ when
330
+ // FLINT_DUMP_CLASSIFIER_PROMPT=1. Used to debug "why did the classifier
331
+ // pick X" — read the file and see what the model actually saw.
332
+ if (process.env.FLINT_DUMP_CLASSIFIER_PROMPT === "1") {
333
+ try {
334
+ const { writeFileSync, mkdirSync, existsSync: exists } = await import("node:fs");
335
+ const dir = join(homedir(), ".flint", "classifier-prompt-snapshots");
336
+ if (!exists(dir)) mkdirSync(dir, { recursive: true });
337
+ const ts = Date.now();
338
+ writeFileSync(join(dir, `${ts}.json`), JSON.stringify({
339
+ ts,
340
+ model: config.intentModel,
341
+ new_message: newMessage,
342
+ system_prompt_chars: CLASSIFIER_SYSTEM.length,
343
+ user_prompt_chars: userPrompt.length,
344
+ messages: [
345
+ { role: "system", content: CLASSIFIER_SYSTEM },
346
+ { role: "user", content: userPrompt },
347
+ ],
348
+ }, null, 2));
349
+ } catch {}
350
+ }
351
+
352
+ // Call the classifier model with retries on transient errors (fetch failed / 5xx).
353
+ // Through the one door: the classifier is not free, so it is subject
354
+ // to the same ceilings as the main loop and its spend lands in the same
355
+ // notebook without this module doing anything about it.
356
+ let raw;
357
+ const MAX_ATTEMPTS = 3;
358
+ let lastError = null;
359
+ for (let attempt = 1; attempt <= MAX_ATTEMPTS; attempt++) {
360
+ try {
361
+ const { message } = await chatCompletion(
362
+ [
363
+ { role: "system", content: CLASSIFIER_SYSTEM },
364
+ { role: "user", content: userPrompt },
365
+ ],
366
+ [],
367
+ null,
368
+ {
369
+ source: "classifier",
370
+ model: config.intentModel,
371
+ maxTokens: 400,
372
+ temperature: 0,
373
+ // Pin sampling seed — Gemini Flash is non-deterministic even at
374
+ // temperature=0 on boundary prompts. A fixed seed
375
+ // gives reproducible outputs for identical inputs, which kills
376
+ // the L1-007/L1-008 flake where the same prompt was classified
377
+ // "normal" in one run and "impossible" in another. Constant is
378
+ // arbitrary; what matters is that it stays the same across calls.
379
+ seed: 42,
380
+ responseFormat: { type: "json_object" },
381
+ stream: false,
382
+ timeoutMs: intentTimeoutMs(),
383
+ },
384
+ );
385
+ raw = message?.content || "";
386
+ break; // success
387
+ } catch (err) {
388
+ // No budget left is an answer, not a hiccup. Retrying it would only burn
389
+ // three seconds to be refused three times.
390
+ if (err.isBudgetError) {
391
+ log.info("classifier skipped, budget spent", { scope: err.scope });
392
+ return fallbackManifest(`budget exhausted (${err.scope})`, toolList);
393
+ }
394
+ // 4xx are client errors — don't retry. 5xx are transient — retry.
395
+ const status = err.statusCode;
396
+ if (status && status < 500) {
397
+ log.warn("classifier http error", { status, attempt });
398
+ return fallbackManifest(`classifier http ${status}`, toolList);
399
+ }
400
+ lastError = err.message;
401
+ if (attempt < MAX_ATTEMPTS) {
402
+ log.info("classifier retry", { attempt, error: err.message });
403
+ await new Promise(r => setTimeout(r, 500 * attempt));
404
+ continue;
405
+ }
406
+ log.warn("classifier call failed", { error: err.message, attempts: attempt });
407
+ return fallbackManifest(`classifier error: ${err.message}`, toolList);
408
+ }
409
+ }
410
+
411
+ if (!raw) {
412
+ return fallbackManifest(`classifier empty after retries: ${lastError}`, toolList);
413
+ }
414
+
415
+ // Parse classifier JSON output
416
+ let parsed;
417
+ try {
418
+ const clean = raw.replace(/^```(?:json)?\s*/i, "").replace(/\s*```\s*$/i, "").trim();
419
+ parsed = JSON.parse(clean);
420
+ } catch (err) {
421
+ log.warn("classifier returned invalid json", { raw: raw.slice(0, 200) });
422
+ return fallbackManifest("invalid json from classifier", toolList);
423
+ }
424
+
425
+ // Lenient intent validation. Smart models (Sonnet, Opus, Gemini 2.5+)
426
+ // frequently invent descriptive intent names ("mesh_write", "tool_simple",
427
+ // "lookup_read") that match the request semantically but are not in our
428
+ // catalog. Their tool picks are usually correct. Previously we threw the
429
+ // tools away and fell back to complex_multi-with-ALL-tools — which
430
+ // defeats the classifier. Now: if the intent name is unknown but the
431
+ // picked tools are valid, keep the tools and normalise the intent to
432
+ // `complex_multi` (a catch-all that needsTools=true with no pattern).
433
+ let intentName = parsed.intent;
434
+ let spec = INTENTS[intentName];
435
+ const liveToolNames = new Set(toolList.map(t => t.function?.name || t.name).filter(Boolean));
436
+ const pickedRaw = Array.isArray(parsed.tools) ? parsed.tools : [];
437
+ const validPicks = pickedRaw.filter(n => typeof n === "string" && liveToolNames.has(n));
438
+
439
+ if (!spec) {
440
+ if (validPicks.length > 0) {
441
+ log.info("classifier picked unknown intent name, normalising to complex_multi", {
442
+ original_intent: intentName,
443
+ valid_tools: validPicks.length,
444
+ });
445
+ intentName = "complex_multi";
446
+ spec = INTENTS.complex_multi;
447
+ } else {
448
+ log.warn("classifier picked unknown intent AND no valid tools", { intent: parsed.intent });
449
+ return fallbackManifest(`unknown intent ${parsed.intent}`, toolList);
450
+ }
451
+ }
452
+
453
+ // Resolve tools:
454
+ // - text-only intents: always empty array (override anything the classifier returned)
455
+ // - tool-backed: keep only names that actually exist in the live registry
456
+ let tools;
457
+ if (!spec.needsTools) {
458
+ tools = [];
459
+ } else {
460
+ tools = validPicks;
461
+ if (tools.length === 0) {
462
+ // Classifier picked no valid tools for a tool-backed intent — degrade to complex_multi
463
+ log.warn("classifier picked no valid tools for tool-backed intent, falling back", { intent: intentName, picked: pickedRaw });
464
+ return fallbackManifest(`${intentName}: no valid tools picked`, toolList);
465
+ }
466
+ }
467
+
468
+ // Validate assessment: classifier returns one of 6 classes, default to 'normal'
469
+ // if missing or unrecognized. Bug fix 2026-04-19: previously this field was
470
+ // dropped on the floor, making the entire assessment gate in agent.js a no-op.
471
+ const validAssessments = new Set(["normal", "dangerous", "overscoped", "ambiguous", "nonsensical", "impossible"]);
472
+ const assessment = validAssessments.has(parsed.assessment) ? parsed.assessment : "normal";
473
+
474
+ // Parse requires_prior_tool_call — list of tool names that must be called
475
+ // before the agent's final text answer. For factual codebase lookups, the
476
+ // agent must verify via search/read instead of guessing from memory.
477
+ // Only keep names that actually exist in the live registry.
478
+ const requiresPrior = Array.isArray(parsed.requires_prior_tool_call)
479
+ ? parsed.requires_prior_tool_call.filter(n => typeof n === "string" && liveToolNames.has(n))
480
+ : [];
481
+
482
+ const manifest = {
483
+ intent: intentName,
484
+ tools, // concrete tool names from live registry, or [] for text-only
485
+ // tool_pattern (if any) widens the effective tool surface at filter time —
486
+ // see filterToolsByManifest. Optional; only tool-backed intents set it.
487
+ tool_pattern: spec.tool_pattern || null,
488
+ max_steps: spec.max_steps,
489
+ expected: spec.expected,
490
+ // Whether a request of this class is supposed to end with something
491
+ // different on disk. Carried from the catalog, not decided here.
492
+ changes: spec.changes,
493
+ user_wants: (parsed.user_wants || "").slice(0, 300),
494
+ reason: (parsed.reason || "").slice(0, 300),
495
+ assessment,
496
+ requires_prior_tool_call: requiresPrior,
497
+ fallback: false,
498
+ };
499
+
500
+ log.info("intent classified", {
501
+ intent: manifest.intent,
502
+ tools: manifest.tools.length,
503
+ max_steps: manifest.max_steps,
504
+ wants: manifest.user_wants,
505
+ });
506
+
507
+ cacheSet(key, manifest);
508
+ logDecision({
509
+ ts: new Date().toISOString(),
510
+ message: newMessage.slice(0, 200),
511
+ intent: manifest.intent,
512
+ tools: manifest.tools,
513
+ max_steps: manifest.max_steps,
514
+ expected: manifest.expected,
515
+ user_wants: manifest.user_wants,
516
+ reason: manifest.reason,
517
+ assessment: manifest.assessment,
518
+ requires_prior_tool_call: manifest.requires_prior_tool_call,
519
+ raw_requires_prior: parsed.requires_prior_tool_call, // debug — what classifier actually returned
520
+ fallback: false,
521
+ });
522
+
523
+ return manifest;
524
+ }
525
+
526
+ /**
527
+ * Build a safe fallback manifest: complex_multi with the built-in tool surface.
528
+ * Used when classification fails for any reason.
529
+ *
530
+ * It used to hand over EVERY live tool, MCP surfaces included. Measured 2026-09-20 on
531
+ * one task: 191 schemas is 33k tokens per call against 6k for the built-ins alone, the
532
+ * turn costs about $0.022 instead of $0.003, and the agent wanders 5 to 16 steps where
533
+ * a narrowed turn takes 2. Every expensive run that day came through this path. MCP
534
+ * servers are optional integrations; the built-ins are the agent's own hands, so a
535
+ * failed classification now costs the user a retry on an MCP task instead of costing
536
+ * money and steps on every task. FLINT_FALLBACK_ALL_TOOLS=1 restores the old shape.
537
+ *
538
+ * 2026-10-02: hidden was too far. An operator who connected Screenbox got an agent
539
+ * that said it had no Screenbox tools and could not do the task. Now a few MCP tools
540
+ * (MCP_INLINE_MAX) are handed over whole, and past that tool_search lists every
541
+ * connected server in its description and loads what the turn asks for.
542
+ */
543
+ function fallbackManifest(reason, availableTools) {
544
+ const spec = resolveIntent("complex_multi");
545
+ const all = (availableTools || []).map(t => t.function?.name || t.name).filter(Boolean);
546
+ const builtin = all.filter(n => !n.startsWith("mcp_"));
547
+ const mcp = all.filter(n => n.startsWith("mcp_"));
548
+ // A few MCP tools go in whole; their schemas cost little and a search would
549
+ // be a wasted step. Past MCP_INLINE_MAX the built-ins go in with
550
+ // tool_search, whose description lists what it can load
551
+ // (tool-search.js mcpCatalog), plus whatever a search already loaded.
552
+ const loaded = new Set(loadedToolNames());
553
+ const toolNames = config.fallbackAllTools || builtin.length === 0 || mcp.length <= mcpInlineMax()
554
+ ? all
555
+ : [...builtin, ...mcp.filter(n => loaded.has(n))];
556
+ return {
557
+ intent: "complex_multi",
558
+ tools: toolNames,
559
+ max_steps: spec.max_steps,
560
+ expected: spec.expected,
561
+ changes: spec.changes,
562
+ user_wants: "",
563
+ reason,
564
+ assessment: "normal", // fallback always proceeds normally — gate only fires when classifier explicitly flags
565
+ fallback: true,
566
+ fallbackScope: toolNames.length === all.length ? "all" : "builtin",
567
+ };
568
+ }
569
+
570
+ /**
571
+ * Apply an intent manifest to a list of tool definitions.
572
+ * Returns the filtered subset the agent loop should expose to the model.
573
+ *
574
+ * @param {Array} allTools — full tool definitions from registry
575
+ * @param {object} manifest — from classifyIntent()
576
+ * @returns {Array} filtered tool definitions
577
+ */
578
+ export function filterToolsByManifest(allTools, manifest) {
579
+ if (!manifest) return allTools;
580
+ const wanted = manifest.tools;
581
+ if (!Array.isArray(wanted)) return allTools;
582
+ // Tree-mode widening: when the resolved intent spec carries a tool_pattern
583
+ // regex, union the classifier's picks with all tools whose name matches
584
+ // the pattern. This is the pragmatic "tool family" fix — the
585
+ // classifier's exact picks are advisory, the intent-wide family is the
586
+ // real selection. Text-only intents (wanted=[] AND no tool_pattern) still
587
+ // get no tools.
588
+ const pattern = manifest.tool_pattern;
589
+ if (wanted.length === 0 && !pattern) return [];
590
+ const allowed = new Set(wanted);
591
+ return allTools.filter(t => {
592
+ const name = t.function?.name || t.name;
593
+ if (!name) return false;
594
+ if (allowed.has(name)) return true;
595
+ if (pattern && pattern.test(name)) return true;
596
+ return false;
597
+ });
598
+ }
599
+
600
+ /** Build a short hint block to inject into the system message. */
601
+ export function formatIntentHint(manifest) {
602
+ if (!manifest || manifest.fallback) return "";
603
+ const isTextOnly = Array.isArray(manifest.tools) && manifest.tools.length === 0;
604
+ const toolDesc = isTextOnly
605
+ ? "none (respond with text only)"
606
+ : manifest.tools.join(", ");
607
+
608
+ const parts = [
609
+ `Classified as: ${manifest.intent}`,
610
+ `User wants: ${manifest.user_wants || "(see message)"}`,
611
+ `Available tools: ${toolDesc}`,
612
+ `Max steps: ${manifest.max_steps}`,
613
+ `Expected output: ${manifest.expected}`,
614
+ ];
615
+
616
+ if (isTextOnly) {
617
+ // Override the default Execution Flow (Intent/Action/Evaluate cycle with EXPECT markers):
618
+ // for text-only intents there is no tool call to evaluate — the model must answer directly.
619
+ parts.push("");
620
+ parts.push("IMPORTANT: This is a text-only request. Do NOT write 'EXPECT: ...' or describe what you plan to do.");
621
+ parts.push("Answer the user directly with the final result in one message.");
622
+ } else {
623
+ // For tool-backed intents the model has historically sometimes shortcircuited
624
+ // to a text response without ever calling a tool (e.g. classifier picks
625
+ // file_write for "translate and save", but Gemini Flash just outputs the
626
+ // translation in chat and forgets the file). Force the model to use a tool.
627
+ parts.push("");
628
+ parts.push("IMPORTANT: This task requires you to call at least one of the available tools above.");
629
+ parts.push("Do NOT respond with text only and consider the task done — you MUST produce a tool_call to fulfil the request (e.g. write_file to save a result, run_command to execute, etc).");
630
+ }
631
+
632
+ return `\n\n<intent>\n${parts.join("\n")}\n</intent>`;
633
+ }