flint-agent 1.14.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.example +108 -0
- package/CHANGELOG.md +55 -0
- package/FEATURES.md +298 -0
- package/LICENSE +21 -0
- package/README.md +435 -0
- package/bin/flint.js +47 -0
- package/config/classifier-prompt.md +218 -0
- package/config/models-curated.json +4 -0
- package/config/providers.json +74 -0
- package/package.json +92 -0
- package/patches/ink+6.8.0.patch +78 -0
- package/profiles/desktop.md +65 -0
- package/profiles/generic.md +20 -0
- package/profiles/marketer.md +20 -0
- package/profiles/profiles.json +34 -0
- package/profiles/ux-reviewer.md +25 -0
- package/src/agent/agent.js +1743 -0
- package/src/agent/auto.js +346 -0
- package/src/agent/backoff.js +143 -0
- package/src/agent/compression.js +310 -0
- package/src/agent/content-resolver.js +180 -0
- package/src/agent/flow-controller.js +309 -0
- package/src/agent/intent-manifest.js +231 -0
- package/src/agent/intent-timeout.js +46 -0
- package/src/agent/intent.js +633 -0
- package/src/agent/knowledge.js +114 -0
- package/src/agent/learning.js +180 -0
- package/src/agent/modes.js +187 -0
- package/src/agent/outcome-ask.js +91 -0
- package/src/agent/project-context.js +76 -0
- package/src/agent/prompt-budget.js +117 -0
- package/src/agent/reflection-extractor.js +140 -0
- package/src/agent/steering.js +86 -0
- package/src/agent/supervisor.js +430 -0
- package/src/agent/swap.js +443 -0
- package/src/agent/system-prompt.js +446 -0
- package/src/agent/time-stamp.js +48 -0
- package/src/agent/tool-guard.js +201 -0
- package/src/agent/toolcall-text.js +162 -0
- package/src/agent/usage.js +297 -0
- package/src/agent/vision.js +94 -0
- package/src/agent/watchdog.js +139 -0
- package/src/agent/workspace-changes.js +177 -0
- package/src/api/address.js +14 -0
- package/src/api/client.js +280 -0
- package/src/api/server.js +535 -0
- package/src/api/stream-pipe.js +113 -0
- package/src/app-state.js +39 -0
- package/src/bootstrap.js +501 -0
- package/src/bus/drain-loop.js +497 -0
- package/src/bus/index.js +270 -0
- package/src/bus/plugins.js +65 -0
- package/src/child-idle.js +14 -0
- package/src/cli.js +118 -0
- package/src/commands/commands.js +1297 -0
- package/src/commands/registry.js +132 -0
- package/src/components/App.js +491 -0
- package/src/components/CarefulMenu.js +145 -0
- package/src/components/HistoryWriter.js +86 -0
- package/src/components/LineInput.js +69 -0
- package/src/components/LiveZone.js +294 -0
- package/src/components/OverlayMenu.js +179 -0
- package/src/components/SystemPanel.js +156 -0
- package/src/components/Table.js +54 -0
- package/src/config.js +249 -0
- package/src/free-models.js +230 -0
- package/src/index.js +1111 -0
- package/src/input-handler.js +13 -0
- package/src/input-text.js +123 -0
- package/src/launcher.js +129 -0
- package/src/logging/api-log.js +95 -0
- package/src/logging/chat-log-follower.js +113 -0
- package/src/logging/chat-log.js +15 -0
- package/src/logging/log-collector.js +182 -0
- package/src/logging/logger.js +112 -0
- package/src/logging/tool-log.js +20 -0
- package/src/mcp-client.js +314 -0
- package/src/memory/conversation-digest.js +113 -0
- package/src/memory/extract-facts.js +98 -0
- package/src/memory/facts.js +181 -0
- package/src/memory/inbox.js +63 -0
- package/src/memory/markdown.js +38 -0
- package/src/memory/patterns.js +185 -0
- package/src/memory/project.js +66 -0
- package/src/memory/reflections.js +74 -0
- package/src/memory/retrieval.js +84 -0
- package/src/memory/rules.js +105 -0
- package/src/memory/session-facts.js +125 -0
- package/src/memory/skills.js +191 -0
- package/src/memory/sqlite-store.js +653 -0
- package/src/memory/store.js +208 -0
- package/src/memory/tools.js +196 -0
- package/src/memory/user-model.js +86 -0
- package/src/message-handler.js +775 -0
- package/src/model-check.js +218 -0
- package/src/plugins/loader.js +120 -0
- package/src/plugins/manager.js +88 -0
- package/src/production-env.js +22 -0
- package/src/profiles.js +42 -0
- package/src/providers/adapters/anthropic.js +270 -0
- package/src/providers/adapters/openai.js +120 -0
- package/src/providers/keys-dpapi.js +41 -0
- package/src/providers/keys-fallback.js +31 -0
- package/src/providers/keys.js +132 -0
- package/src/providers/models.js +154 -0
- package/src/providers/registry.js +56 -0
- package/src/providers/state.js +56 -0
- package/src/registry.js +96 -0
- package/src/restart.js +29 -0
- package/src/sandbox/backend.js +130 -0
- package/src/security/api-auth.js +132 -0
- package/src/security/audit.js +98 -0
- package/src/security/child-policy.js +41 -0
- package/src/security/command-guard.js +173 -0
- package/src/security/content-fence.js +250 -0
- package/src/security/content-validator.js +132 -0
- package/src/security/index.js +143 -0
- package/src/security/network-guard.js +126 -0
- package/src/security/pairing.js +180 -0
- package/src/security/path-guard.js +140 -0
- package/src/security/persona-guard.js +67 -0
- package/src/security/policies.js +452 -0
- package/src/security/safety-constants.js +34 -0
- package/src/security/watchdog.js +107 -0
- package/src/sessions.js +130 -0
- package/src/spend.js +97 -0
- package/src/startup-watchdog.js +59 -0
- package/src/stdio/args.js +71 -0
- package/src/stdio/guard.js +59 -0
- package/src/stdio/protocol.js +167 -0
- package/src/stdio/run.js +106 -0
- package/src/stdio/session.js +180 -0
- package/src/store/agent-slice.js +306 -0
- package/src/store/dataset-slice.js +73 -0
- package/src/store/index.js +22 -0
- package/src/store/process-slice.js +135 -0
- package/src/store/session-slice.js +191 -0
- package/src/store/ui-slice.js +119 -0
- package/src/tasks/db.js +184 -0
- package/src/tasks/queries.js +589 -0
- package/src/tools/agent-tools.js +473 -0
- package/src/tools/checkpoint.js +152 -0
- package/src/tools/command-approvals.js +180 -0
- package/src/tools/dataset.js +50 -0
- package/src/tools/filesystem.js +682 -0
- package/src/tools/inbox-tools.js +48 -0
- package/src/tools/mesh.js +135 -0
- package/src/tools/own-env.js +136 -0
- package/src/tools/permissions.js +681 -0
- package/src/tools/plugin-tools.js +123 -0
- package/src/tools/process-tools.js +595 -0
- package/src/tools/registry.js +307 -0
- package/src/tools/swap-tools.js +72 -0
- package/src/tools/system.js +662 -0
- package/src/tools/tasks.js +532 -0
- package/src/tools/tool-search.js +171 -0
- package/src/ui/header.js +140 -0
- package/src/ui/input-cursor.js +23 -0
- package/src/ui/last-line.js +25 -0
- package/src/ui/line-edit.js +135 -0
- package/src/ui/output.js +399 -0
- package/src/ui/paste-tokens.js +131 -0
- package/src/ui/prompt-attention.js +134 -0
- package/src/ui/render-options.js +13 -0
- package/src/ui/replay.js +94 -0
- package/src/ui/splash.js +49 -0
- package/src/ui/status-level.js +36 -0
- package/src/ui/tool-ledger.js +203 -0
- package/src/ui/window-title.js +150 -0
- package/src/update.js +205 -0
- package/system.md +63 -0
|
@@ -0,0 +1,633 @@
|
|
|
1
|
+
// Intent classifier — one cheap LLM call BEFORE the main agent loop.
|
|
2
|
+
//
|
|
3
|
+
// The classifier reads session context (summary + recent messages + new message)
|
|
4
|
+
// PLUS the live list of available tools from the registry, and returns a manifest
|
|
5
|
+
// with:
|
|
6
|
+
// - intent class (creative_text, memory_read, shell_multi, ...)
|
|
7
|
+
// - concrete tool names picked from the registry
|
|
8
|
+
// - max_steps and expected output shape
|
|
9
|
+
//
|
|
10
|
+
// Tool names are NOT hardcoded in the manifest — they come from the live registry
|
|
11
|
+
// at call time. Adding a tool in the registry makes it automatically eligible for
|
|
12
|
+
// the classifier without any code changes here.
|
|
13
|
+
|
|
14
|
+
import { getSpendLevel, spendSettings } from "../spend.js";
|
|
15
|
+
import { createHash } from "node:crypto";
|
|
16
|
+
import { intentTimeoutMs } from "./intent-timeout.js";
|
|
17
|
+
import { appendFileSync, mkdirSync, statSync, renameSync, existsSync, readFileSync } from "node:fs";
|
|
18
|
+
import { homedir } from "node:os";
|
|
19
|
+
import { join, dirname } from "node:path";
|
|
20
|
+
import { fileURLToPath } from "node:url";
|
|
21
|
+
import { config } from "../config.js";
|
|
22
|
+
import { createLogger } from "../logging/logger.js";
|
|
23
|
+
import { INTENTS, formatIntentCatalog, resolveIntent } from "./intent-manifest.js";
|
|
24
|
+
import { chatCompletion } from "../api/client.js";
|
|
25
|
+
import { CORE_TOOLS, TOOL_SEARCH_NAME, loadedToolNames } from "../tools/tool-search.js";
|
|
26
|
+
|
|
27
|
+
// Load the classifier prompt from config/classifier-prompt.md. The file is the
|
|
28
|
+
// source of truth for classifier behaviour — edits there propagate on restart
|
|
29
|
+
// with no code change. The file has three parts separated by `---`:
|
|
30
|
+
// the frontmatter header, the main SYSTEM+rules body, and the SCHEMA block.
|
|
31
|
+
// We extract the two we need (system body + schema block) at module init.
|
|
32
|
+
const _thisFile = fileURLToPath(import.meta.url);
|
|
33
|
+
const _promptPath = join(dirname(_thisFile), "..", "..", "config", "classifier-prompt.md");
|
|
34
|
+
const _promptRaw = readFileSync(_promptPath, "utf8");
|
|
35
|
+
const _parts = _promptRaw.split(/^---\s*$/m).map(s => s.trim()).filter(Boolean);
|
|
36
|
+
// Expected: [header, body, schema]. Fall back to full file if split fails.
|
|
37
|
+
const _body = _parts[1] || _promptRaw;
|
|
38
|
+
const _schemaSection = _parts[2] || "";
|
|
39
|
+
const _schemaMatch = _schemaSection.match(/```json\s*([\s\S]*?)```/);
|
|
40
|
+
|
|
41
|
+
const log = createLogger("intent");
|
|
42
|
+
|
|
43
|
+
// ── LRU cache for classifier results ──
|
|
44
|
+
// Same user message + same recent messages + same tool list = same classification.
|
|
45
|
+
// No point paying for the LLM call twice in a short window.
|
|
46
|
+
const CACHE_MAX = 50;
|
|
47
|
+
|
|
48
|
+
// With the classifier off, up to this many MCP tools are handed over whole;
|
|
49
|
+
// more go through tool_search (fallbackManifest). Set by the spend mode
|
|
50
|
+
// (spend.js: economy 10, normal 30, generous 200); FLINT_MCP_INLINE_MAX wins.
|
|
51
|
+
export function mcpInlineMax(env = process.env, level = getSpendLevel(env)) {
|
|
52
|
+
const v = parseInt(env.FLINT_MCP_INLINE_MAX, 10);
|
|
53
|
+
return Number.isFinite(v) ? v : spendSettings(level).mcpInlineMax;
|
|
54
|
+
}
|
|
55
|
+
const CACHE_TTL_MS = 5 * 60 * 1000;
|
|
56
|
+
const _cache = new Map(); // key → {manifest, ts}
|
|
57
|
+
// The classifier per-attempt budget is read from INTENT_TIMEOUT_MS via its own
|
|
58
|
+
// module (src/agent/intent-timeout.js) so it can be tested without importing the
|
|
59
|
+
// config, the API client and the tool manifest. A hard deadline here is what made a
|
|
60
|
+
// test depend on machine load: classifyIntent catches its own errors and falls back to
|
|
61
|
+
// the all-tools manifest, so a blown deadline looks like a valid classification.
|
|
62
|
+
// See that file.
|
|
63
|
+
|
|
64
|
+
function cacheKey(newMessage, recentMessages, availableTools) {
|
|
65
|
+
const h = createHash("sha256");
|
|
66
|
+
h.update(newMessage || "");
|
|
67
|
+
h.update("|");
|
|
68
|
+
for (const m of (recentMessages || [])) {
|
|
69
|
+
h.update(m.role || "");
|
|
70
|
+
h.update(":");
|
|
71
|
+
h.update(typeof m.content === "string" ? m.content : JSON.stringify(m.content || ""));
|
|
72
|
+
h.update("\n");
|
|
73
|
+
}
|
|
74
|
+
h.update("|");
|
|
75
|
+
for (const t of (availableTools || [])) {
|
|
76
|
+
h.update((t.function?.name || t.name || "") + ",");
|
|
77
|
+
}
|
|
78
|
+
return h.digest("hex");
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
function cacheGet(key) {
|
|
82
|
+
const entry = _cache.get(key);
|
|
83
|
+
if (!entry) return null;
|
|
84
|
+
if (Date.now() - entry.ts > CACHE_TTL_MS) {
|
|
85
|
+
_cache.delete(key);
|
|
86
|
+
return null;
|
|
87
|
+
}
|
|
88
|
+
// LRU: move to end
|
|
89
|
+
_cache.delete(key);
|
|
90
|
+
_cache.set(key, entry);
|
|
91
|
+
return entry.manifest;
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
function cacheSet(key, manifest) {
|
|
95
|
+
if (_cache.size >= CACHE_MAX) {
|
|
96
|
+
const oldest = _cache.keys().next().value;
|
|
97
|
+
if (oldest !== undefined) _cache.delete(oldest);
|
|
98
|
+
}
|
|
99
|
+
_cache.set(key, { manifest, ts: Date.now() });
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
// ── Prompt injection guard ──
|
|
103
|
+
// A user whose message mentions classifier output keywords could try to disable
|
|
104
|
+
// tool access by steering classification toward text-only intents. We detect
|
|
105
|
+
// obvious markers and, if present, skip the classifier entirely and return the
|
|
106
|
+
// full-tool fallback. The attacker cannot narrow the tool surface this way.
|
|
107
|
+
const INJECTION_PATTERNS = [
|
|
108
|
+
/\bclassif(y|ied|ication)\s+as\b/i,
|
|
109
|
+
/\bintent\s*[:=]\s*["']?[a-z_]+/i,
|
|
110
|
+
/\btools\s*[:=]\s*\[/i,
|
|
111
|
+
/\bneedsTools\b/i,
|
|
112
|
+
/\bmax_?steps\s*[:=]/i,
|
|
113
|
+
];
|
|
114
|
+
|
|
115
|
+
function looksLikeInjection(text) {
|
|
116
|
+
if (!text || typeof text !== "string") return false;
|
|
117
|
+
return INJECTION_PATTERNS.some(re => re.test(text));
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
// ── Shadow log of classification decisions ──
|
|
121
|
+
// Every classification is appended so we can later measure accuracy against
|
|
122
|
+
// E2E outcomes. No runtime cost beyond one async-fs append.
|
|
123
|
+
const DECISIONS_FILE = join(homedir(), ".flint", "intent-decisions.jsonl");
|
|
124
|
+
const DECISIONS_MAX_BYTES = 10 * 1024 * 1024; // 10 MB → archive + start fresh
|
|
125
|
+
let _logCallCount = 0;
|
|
126
|
+
|
|
127
|
+
// Rotate the decisions log if it crosses the size threshold. Archive name is
|
|
128
|
+
// `intent-decisions-YYYY-MM-DD.jsonl`; if an archive for today already exists
|
|
129
|
+
// (rare), the timestamp is appended to keep history distinct. Best-effort —
|
|
130
|
+
// failures are swallowed, logging continues on the (unrotated) file.
|
|
131
|
+
function maybeRotateDecisions() {
|
|
132
|
+
try {
|
|
133
|
+
if (!existsSync(DECISIONS_FILE)) return;
|
|
134
|
+
const st = statSync(DECISIONS_FILE);
|
|
135
|
+
if (st.size < DECISIONS_MAX_BYTES) return;
|
|
136
|
+
const date = new Date().toISOString().slice(0, 10);
|
|
137
|
+
let archive = DECISIONS_FILE.replace(/\.jsonl$/, `-${date}.jsonl`);
|
|
138
|
+
if (existsSync(archive)) {
|
|
139
|
+
const stamp = new Date().toISOString().replace(/[:.]/g, "-");
|
|
140
|
+
archive = DECISIONS_FILE.replace(/\.jsonl$/, `-${stamp}.jsonl`);
|
|
141
|
+
}
|
|
142
|
+
renameSync(DECISIONS_FILE, archive);
|
|
143
|
+
} catch {}
|
|
144
|
+
}
|
|
145
|
+
|
|
146
|
+
function logDecision(entry) {
|
|
147
|
+
try {
|
|
148
|
+
mkdirSync(join(homedir(), ".flint"), { recursive: true });
|
|
149
|
+
// Gate the size check to avoid a statSync on every classification; once
|
|
150
|
+
// per 100 decisions is enough (max 100 extra log lines before rotation).
|
|
151
|
+
if (_logCallCount++ % 100 === 0) maybeRotateDecisions();
|
|
152
|
+
appendFileSync(DECISIONS_FILE, JSON.stringify(entry) + "\n", "utf-8");
|
|
153
|
+
} catch {
|
|
154
|
+
// Logging is best-effort — never break classification on log failure
|
|
155
|
+
}
|
|
156
|
+
}
|
|
157
|
+
|
|
158
|
+
// Classifier system prompt — body loaded from config/classifier-prompt.md at
|
|
159
|
+
// module init. The markdown file is the source of truth for classifier rules.
|
|
160
|
+
// The live intent catalog (from intent-manifest.js) is appended here so that
|
|
161
|
+
// adding/removing intents in code is automatically reflected without editing
|
|
162
|
+
// the markdown. AVAILABLE_TOOLS is appended per-call in buildUserPrompt.
|
|
163
|
+
const CLASSIFIER_SYSTEM = `${_body}\n\n## INTENT_CLASSES\n\n${formatIntentCatalog()}\n`;
|
|
164
|
+
|
|
165
|
+
// Schema hint — extracted from the fenced ```json block in the SCHEMA section
|
|
166
|
+
// of config/classifier-prompt.md. Fallback to a minimal inline schema if the
|
|
167
|
+
// markdown file's SCHEMA block is missing or malformed.
|
|
168
|
+
const SCHEMA_HINT = (_schemaMatch && _schemaMatch[1].trim()) || `{
|
|
169
|
+
"intent": "<one of the class names above>",
|
|
170
|
+
"tools": ["<tool_name_from_the_list>", ...],
|
|
171
|
+
"assessment": "<normal|dangerous|overscoped|ambiguous|nonsensical|impossible>",
|
|
172
|
+
"requires_prior_tool_call": [],
|
|
173
|
+
"user_wants": "<one sentence paraphrase>",
|
|
174
|
+
"reason": "<why this class>"
|
|
175
|
+
}`;
|
|
176
|
+
|
|
177
|
+
/**
|
|
178
|
+
* Classify a user request in its conversation context.
|
|
179
|
+
*
|
|
180
|
+
* @param {object} ctx
|
|
181
|
+
* @param {string} ctx.newMessage — the new user message (required)
|
|
182
|
+
* @param {Array} ctx.availableTools — live tool defs from registry [{type,function:{name,description}},...]
|
|
183
|
+
* @param {string} [ctx.sessionSummary] — compact summary of the session so far (optional)
|
|
184
|
+
* @param {Array} [ctx.recentMessages] — last few {role, content} messages (optional)
|
|
185
|
+
* @returns {Promise<object>} manifest: {intent, tools, max_steps, expected, user_wants, reason, fallback?}
|
|
186
|
+
*/
|
|
187
|
+
export async function classifyIntent(ctx) {
|
|
188
|
+
const { newMessage, sessionSummary, recentMessages, availableTools } = ctx;
|
|
189
|
+
|
|
190
|
+
// Headless mode bypass: when Flint runs as a headless subprocess (e.g. the
|
|
191
|
+
// SWE-bench runner, or any CLI `--headless --task ...` invocation) the
|
|
192
|
+
// caller's prompt is usually a long, complex code-editing brief that the
|
|
193
|
+
// classifier misreads as `file_read` and caps at 3 iterations. There is no
|
|
194
|
+
// interactive user to correct the classifier. Give the agent full tool
|
|
195
|
+
// access and let config.maxIterations govern the budget instead.
|
|
196
|
+
if (config.headless) {
|
|
197
|
+
return fallbackManifest("headless mode — classifier bypassed", availableTools);
|
|
198
|
+
}
|
|
199
|
+
|
|
200
|
+
// A/B switch: no classifier call at all, the built-in tools every turn, no
|
|
201
|
+
// assessment gate. Across 2026-09-26 the classifier's tool picks caused three
|
|
202
|
+
// of Flint's five losses (no web on a SWE task, every tool taken away from
|
|
203
|
+
// an explicit delete, three operator tools for a browser task). This measures
|
|
204
|
+
// what the turn does without it.
|
|
205
|
+
if (process.env.FLINT_NO_CLASSIFIER === "1") {
|
|
206
|
+
return fallbackManifest("classifier disabled (FLINT_NO_CLASSIFIER)", availableTools);
|
|
207
|
+
}
|
|
208
|
+
|
|
209
|
+
// FLINT_TOOL_MODE=search: an explicit choice, so it comes before the
|
|
210
|
+
// classifier-off fallback below. No classifier call. The core set, whatever
|
|
211
|
+
// tool_search has loaded so far this session, and tool_search itself.
|
|
212
|
+
if (process.env.FLINT_TOOL_MODE === "search") {
|
|
213
|
+
const present = new Set((availableTools || []).map(t => t.function?.name || t.name).filter(Boolean));
|
|
214
|
+
const names = [...CORE_TOOLS, ...loadedToolNames(), TOOL_SEARCH_NAME].filter(n => present.has(n));
|
|
215
|
+
return { ...fallbackManifest("tool search mode", availableTools), tools: [...new Set(names)], fallbackScope: "search" };
|
|
216
|
+
}
|
|
217
|
+
|
|
218
|
+
// No INTENT_MODEL: the classifier is off, not an error.
|
|
219
|
+
if (!config.intentModel) {
|
|
220
|
+
return fallbackManifest("classifier disabled (INTENT_MODEL not set)", availableTools);
|
|
221
|
+
}
|
|
222
|
+
|
|
223
|
+
if (!newMessage || typeof newMessage !== "string") {
|
|
224
|
+
return fallbackManifest("empty or non-text message", availableTools);
|
|
225
|
+
}
|
|
226
|
+
|
|
227
|
+
// Prompt injection guard — if the user message contains classifier-control
|
|
228
|
+
// keywords, skip classification entirely and give the agent full tool access.
|
|
229
|
+
// An attacker cannot use this path to narrow the tool surface.
|
|
230
|
+
if (looksLikeInjection(newMessage)) {
|
|
231
|
+
log.warn("intent: injection pattern detected, using full-tool fallback");
|
|
232
|
+
const fallback = fallbackManifest("injection pattern in user message", availableTools);
|
|
233
|
+
logDecision({
|
|
234
|
+
ts: new Date().toISOString(),
|
|
235
|
+
message: newMessage.slice(0, 200),
|
|
236
|
+
intent: fallback.intent,
|
|
237
|
+
tools: fallback.tools.length,
|
|
238
|
+
assessment: fallback.assessment,
|
|
239
|
+
fallback: true,
|
|
240
|
+
reason: "injection_guard",
|
|
241
|
+
});
|
|
242
|
+
return fallback;
|
|
243
|
+
}
|
|
244
|
+
|
|
245
|
+
// Cache: identical input in a short window reuses the previous decision.
|
|
246
|
+
const key = cacheKey(newMessage, recentMessages, availableTools);
|
|
247
|
+
const cached = cacheGet(key);
|
|
248
|
+
if (cached) {
|
|
249
|
+
log.debug("intent: cache hit", { intent: cached.intent });
|
|
250
|
+
return cached;
|
|
251
|
+
}
|
|
252
|
+
|
|
253
|
+
// Build the user-side prompt for the classifier.
|
|
254
|
+
//
|
|
255
|
+
// Classifier stays lightweight on purpose: intent catalog + available
|
|
256
|
+
// tools + user message + schema. Project rules (FLINT.md), memory
|
|
257
|
+
// layers, profile, OS hints etc. are the agent loop's concern — not
|
|
258
|
+
// the classifier's. Loading them here bloats the prompt (~4K chars of
|
|
259
|
+
// project rules for a ~300-token decision) without improving tool
|
|
260
|
+
// picks. The agent loop still gets FLINT.md and applies its guidance
|
|
261
|
+
// when forming the actual tool call arguments.
|
|
262
|
+
//
|
|
263
|
+
// ORDER MATTERS — LLMs have strong primacy + recency bias and a well-known
|
|
264
|
+
// "lost in the middle" problem. The three things the classifier MUST
|
|
265
|
+
// consider carefully are (a) the user's new message, (b) the tool catalog
|
|
266
|
+
// it picks from, (c) the JSON schema for the response. Those go at the
|
|
267
|
+
// start (primacy) or the very end (recency). Conversation history is
|
|
268
|
+
// deprioritised context — it goes in the middle, wrapped in a block that
|
|
269
|
+
// explicitly marks it as "for disambiguation only, do not let it drive
|
|
270
|
+
// tool choice". Previous layout had the new message 2nd-to-last and the
|
|
271
|
+
// tool catalog in the middle — observed 2026-04-21: when preceding turns
|
|
272
|
+
// in the session were about shell/ssh/docker, the classifier carried that
|
|
273
|
+
// bias forward and routed "sum tags in planner" → shell_command instead
|
|
274
|
+
// of the mcp_planner_* tools that were right there in the catalog.
|
|
275
|
+
const parts = [];
|
|
276
|
+
|
|
277
|
+
// 1) NEW MESSAGE first — highest attention.
|
|
278
|
+
parts.push("NEW MESSAGE:\n" + newMessage.slice(0, 1000));
|
|
279
|
+
|
|
280
|
+
// 2) AVAILABLE TOOLS right after — the classifier must pick from this list.
|
|
281
|
+
const toolList = Array.isArray(availableTools) ? availableTools : [];
|
|
282
|
+
if (toolList.length) {
|
|
283
|
+
const toolLines = toolList
|
|
284
|
+
.map(t => {
|
|
285
|
+
const name = t.function?.name || t.name;
|
|
286
|
+
const desc = (t.function?.description || t.description || "").replace(/\s+/g, " ").slice(0, 120);
|
|
287
|
+
return name ? ` ${name} — ${desc}` : null;
|
|
288
|
+
})
|
|
289
|
+
.filter(Boolean)
|
|
290
|
+
.join("\n");
|
|
291
|
+
parts.push("AVAILABLE TOOLS:\n" + toolLines);
|
|
292
|
+
}
|
|
293
|
+
|
|
294
|
+
// 3) Session context in the middle, framed as weak context so the model
|
|
295
|
+
// treats it as disambiguation help, not as a driver of tool choice.
|
|
296
|
+
if (sessionSummary && sessionSummary.trim()) {
|
|
297
|
+
parts.push(
|
|
298
|
+
"SESSION SUMMARY (context only — do not let this override tool choice from the new message):\n" +
|
|
299
|
+
sessionSummary.trim().slice(0, 1000),
|
|
300
|
+
);
|
|
301
|
+
}
|
|
302
|
+
if (Array.isArray(recentMessages) && recentMessages.length) {
|
|
303
|
+
const historyLines = [];
|
|
304
|
+
for (const m of recentMessages.slice(-5)) {
|
|
305
|
+
if (!m || !m.role) continue;
|
|
306
|
+
const text = typeof m.content === "string"
|
|
307
|
+
? m.content
|
|
308
|
+
: Array.isArray(m.content)
|
|
309
|
+
? m.content.map(c => c.text || "").join(" ")
|
|
310
|
+
: "";
|
|
311
|
+
if (!text.trim()) continue;
|
|
312
|
+
historyLines.push(`${m.role}: ${text.slice(0, 300)}`);
|
|
313
|
+
}
|
|
314
|
+
if (historyLines.length) {
|
|
315
|
+
parts.push(
|
|
316
|
+
"RECENT MESSAGES (context only — classify the new message on its own merits, do not carry bias from previous turns):\n" +
|
|
317
|
+
historyLines.join("\n"),
|
|
318
|
+
);
|
|
319
|
+
}
|
|
320
|
+
}
|
|
321
|
+
|
|
322
|
+
// 4) Schema at the very end — the last thing the model reads before
|
|
323
|
+
// generating its response, for maximum structural compliance.
|
|
324
|
+
parts.push("Classify the NEW MESSAGE above. Return JSON matching:\n" + SCHEMA_HINT);
|
|
325
|
+
|
|
326
|
+
const userPrompt = parts.join("\n\n");
|
|
327
|
+
|
|
328
|
+
// Env-gated snapshot of the classifier prompt. Writes one JSON file per
|
|
329
|
+
// classify call to ~/.flint/classifier-prompt-snapshots/ when
|
|
330
|
+
// FLINT_DUMP_CLASSIFIER_PROMPT=1. Used to debug "why did the classifier
|
|
331
|
+
// pick X" — read the file and see what the model actually saw.
|
|
332
|
+
if (process.env.FLINT_DUMP_CLASSIFIER_PROMPT === "1") {
|
|
333
|
+
try {
|
|
334
|
+
const { writeFileSync, mkdirSync, existsSync: exists } = await import("node:fs");
|
|
335
|
+
const dir = join(homedir(), ".flint", "classifier-prompt-snapshots");
|
|
336
|
+
if (!exists(dir)) mkdirSync(dir, { recursive: true });
|
|
337
|
+
const ts = Date.now();
|
|
338
|
+
writeFileSync(join(dir, `${ts}.json`), JSON.stringify({
|
|
339
|
+
ts,
|
|
340
|
+
model: config.intentModel,
|
|
341
|
+
new_message: newMessage,
|
|
342
|
+
system_prompt_chars: CLASSIFIER_SYSTEM.length,
|
|
343
|
+
user_prompt_chars: userPrompt.length,
|
|
344
|
+
messages: [
|
|
345
|
+
{ role: "system", content: CLASSIFIER_SYSTEM },
|
|
346
|
+
{ role: "user", content: userPrompt },
|
|
347
|
+
],
|
|
348
|
+
}, null, 2));
|
|
349
|
+
} catch {}
|
|
350
|
+
}
|
|
351
|
+
|
|
352
|
+
// Call the classifier model with retries on transient errors (fetch failed / 5xx).
|
|
353
|
+
// Through the one door: the classifier is not free, so it is subject
|
|
354
|
+
// to the same ceilings as the main loop and its spend lands in the same
|
|
355
|
+
// notebook without this module doing anything about it.
|
|
356
|
+
let raw;
|
|
357
|
+
const MAX_ATTEMPTS = 3;
|
|
358
|
+
let lastError = null;
|
|
359
|
+
for (let attempt = 1; attempt <= MAX_ATTEMPTS; attempt++) {
|
|
360
|
+
try {
|
|
361
|
+
const { message } = await chatCompletion(
|
|
362
|
+
[
|
|
363
|
+
{ role: "system", content: CLASSIFIER_SYSTEM },
|
|
364
|
+
{ role: "user", content: userPrompt },
|
|
365
|
+
],
|
|
366
|
+
[],
|
|
367
|
+
null,
|
|
368
|
+
{
|
|
369
|
+
source: "classifier",
|
|
370
|
+
model: config.intentModel,
|
|
371
|
+
maxTokens: 400,
|
|
372
|
+
temperature: 0,
|
|
373
|
+
// Pin sampling seed — Gemini Flash is non-deterministic even at
|
|
374
|
+
// temperature=0 on boundary prompts. A fixed seed
|
|
375
|
+
// gives reproducible outputs for identical inputs, which kills
|
|
376
|
+
// the L1-007/L1-008 flake where the same prompt was classified
|
|
377
|
+
// "normal" in one run and "impossible" in another. Constant is
|
|
378
|
+
// arbitrary; what matters is that it stays the same across calls.
|
|
379
|
+
seed: 42,
|
|
380
|
+
responseFormat: { type: "json_object" },
|
|
381
|
+
stream: false,
|
|
382
|
+
timeoutMs: intentTimeoutMs(),
|
|
383
|
+
},
|
|
384
|
+
);
|
|
385
|
+
raw = message?.content || "";
|
|
386
|
+
break; // success
|
|
387
|
+
} catch (err) {
|
|
388
|
+
// No budget left is an answer, not a hiccup. Retrying it would only burn
|
|
389
|
+
// three seconds to be refused three times.
|
|
390
|
+
if (err.isBudgetError) {
|
|
391
|
+
log.info("classifier skipped, budget spent", { scope: err.scope });
|
|
392
|
+
return fallbackManifest(`budget exhausted (${err.scope})`, toolList);
|
|
393
|
+
}
|
|
394
|
+
// 4xx are client errors — don't retry. 5xx are transient — retry.
|
|
395
|
+
const status = err.statusCode;
|
|
396
|
+
if (status && status < 500) {
|
|
397
|
+
log.warn("classifier http error", { status, attempt });
|
|
398
|
+
return fallbackManifest(`classifier http ${status}`, toolList);
|
|
399
|
+
}
|
|
400
|
+
lastError = err.message;
|
|
401
|
+
if (attempt < MAX_ATTEMPTS) {
|
|
402
|
+
log.info("classifier retry", { attempt, error: err.message });
|
|
403
|
+
await new Promise(r => setTimeout(r, 500 * attempt));
|
|
404
|
+
continue;
|
|
405
|
+
}
|
|
406
|
+
log.warn("classifier call failed", { error: err.message, attempts: attempt });
|
|
407
|
+
return fallbackManifest(`classifier error: ${err.message}`, toolList);
|
|
408
|
+
}
|
|
409
|
+
}
|
|
410
|
+
|
|
411
|
+
if (!raw) {
|
|
412
|
+
return fallbackManifest(`classifier empty after retries: ${lastError}`, toolList);
|
|
413
|
+
}
|
|
414
|
+
|
|
415
|
+
// Parse classifier JSON output
|
|
416
|
+
let parsed;
|
|
417
|
+
try {
|
|
418
|
+
const clean = raw.replace(/^```(?:json)?\s*/i, "").replace(/\s*```\s*$/i, "").trim();
|
|
419
|
+
parsed = JSON.parse(clean);
|
|
420
|
+
} catch (err) {
|
|
421
|
+
log.warn("classifier returned invalid json", { raw: raw.slice(0, 200) });
|
|
422
|
+
return fallbackManifest("invalid json from classifier", toolList);
|
|
423
|
+
}
|
|
424
|
+
|
|
425
|
+
// Lenient intent validation. Smart models (Sonnet, Opus, Gemini 2.5+)
|
|
426
|
+
// frequently invent descriptive intent names ("mesh_write", "tool_simple",
|
|
427
|
+
// "lookup_read") that match the request semantically but are not in our
|
|
428
|
+
// catalog. Their tool picks are usually correct. Previously we threw the
|
|
429
|
+
// tools away and fell back to complex_multi-with-ALL-tools — which
|
|
430
|
+
// defeats the classifier. Now: if the intent name is unknown but the
|
|
431
|
+
// picked tools are valid, keep the tools and normalise the intent to
|
|
432
|
+
// `complex_multi` (a catch-all that needsTools=true with no pattern).
|
|
433
|
+
let intentName = parsed.intent;
|
|
434
|
+
let spec = INTENTS[intentName];
|
|
435
|
+
const liveToolNames = new Set(toolList.map(t => t.function?.name || t.name).filter(Boolean));
|
|
436
|
+
const pickedRaw = Array.isArray(parsed.tools) ? parsed.tools : [];
|
|
437
|
+
const validPicks = pickedRaw.filter(n => typeof n === "string" && liveToolNames.has(n));
|
|
438
|
+
|
|
439
|
+
if (!spec) {
|
|
440
|
+
if (validPicks.length > 0) {
|
|
441
|
+
log.info("classifier picked unknown intent name, normalising to complex_multi", {
|
|
442
|
+
original_intent: intentName,
|
|
443
|
+
valid_tools: validPicks.length,
|
|
444
|
+
});
|
|
445
|
+
intentName = "complex_multi";
|
|
446
|
+
spec = INTENTS.complex_multi;
|
|
447
|
+
} else {
|
|
448
|
+
log.warn("classifier picked unknown intent AND no valid tools", { intent: parsed.intent });
|
|
449
|
+
return fallbackManifest(`unknown intent ${parsed.intent}`, toolList);
|
|
450
|
+
}
|
|
451
|
+
}
|
|
452
|
+
|
|
453
|
+
// Resolve tools:
|
|
454
|
+
// - text-only intents: always empty array (override anything the classifier returned)
|
|
455
|
+
// - tool-backed: keep only names that actually exist in the live registry
|
|
456
|
+
let tools;
|
|
457
|
+
if (!spec.needsTools) {
|
|
458
|
+
tools = [];
|
|
459
|
+
} else {
|
|
460
|
+
tools = validPicks;
|
|
461
|
+
if (tools.length === 0) {
|
|
462
|
+
// Classifier picked no valid tools for a tool-backed intent — degrade to complex_multi
|
|
463
|
+
log.warn("classifier picked no valid tools for tool-backed intent, falling back", { intent: intentName, picked: pickedRaw });
|
|
464
|
+
return fallbackManifest(`${intentName}: no valid tools picked`, toolList);
|
|
465
|
+
}
|
|
466
|
+
}
|
|
467
|
+
|
|
468
|
+
// Validate assessment: classifier returns one of 6 classes, default to 'normal'
|
|
469
|
+
// if missing or unrecognized. Bug fix 2026-04-19: previously this field was
|
|
470
|
+
// dropped on the floor, making the entire assessment gate in agent.js a no-op.
|
|
471
|
+
const validAssessments = new Set(["normal", "dangerous", "overscoped", "ambiguous", "nonsensical", "impossible"]);
|
|
472
|
+
const assessment = validAssessments.has(parsed.assessment) ? parsed.assessment : "normal";
|
|
473
|
+
|
|
474
|
+
// Parse requires_prior_tool_call — list of tool names that must be called
|
|
475
|
+
// before the agent's final text answer. For factual codebase lookups, the
|
|
476
|
+
// agent must verify via search/read instead of guessing from memory.
|
|
477
|
+
// Only keep names that actually exist in the live registry.
|
|
478
|
+
const requiresPrior = Array.isArray(parsed.requires_prior_tool_call)
|
|
479
|
+
? parsed.requires_prior_tool_call.filter(n => typeof n === "string" && liveToolNames.has(n))
|
|
480
|
+
: [];
|
|
481
|
+
|
|
482
|
+
const manifest = {
|
|
483
|
+
intent: intentName,
|
|
484
|
+
tools, // concrete tool names from live registry, or [] for text-only
|
|
485
|
+
// tool_pattern (if any) widens the effective tool surface at filter time —
|
|
486
|
+
// see filterToolsByManifest. Optional; only tool-backed intents set it.
|
|
487
|
+
tool_pattern: spec.tool_pattern || null,
|
|
488
|
+
max_steps: spec.max_steps,
|
|
489
|
+
expected: spec.expected,
|
|
490
|
+
// Whether a request of this class is supposed to end with something
|
|
491
|
+
// different on disk. Carried from the catalog, not decided here.
|
|
492
|
+
changes: spec.changes,
|
|
493
|
+
user_wants: (parsed.user_wants || "").slice(0, 300),
|
|
494
|
+
reason: (parsed.reason || "").slice(0, 300),
|
|
495
|
+
assessment,
|
|
496
|
+
requires_prior_tool_call: requiresPrior,
|
|
497
|
+
fallback: false,
|
|
498
|
+
};
|
|
499
|
+
|
|
500
|
+
log.info("intent classified", {
|
|
501
|
+
intent: manifest.intent,
|
|
502
|
+
tools: manifest.tools.length,
|
|
503
|
+
max_steps: manifest.max_steps,
|
|
504
|
+
wants: manifest.user_wants,
|
|
505
|
+
});
|
|
506
|
+
|
|
507
|
+
cacheSet(key, manifest);
|
|
508
|
+
logDecision({
|
|
509
|
+
ts: new Date().toISOString(),
|
|
510
|
+
message: newMessage.slice(0, 200),
|
|
511
|
+
intent: manifest.intent,
|
|
512
|
+
tools: manifest.tools,
|
|
513
|
+
max_steps: manifest.max_steps,
|
|
514
|
+
expected: manifest.expected,
|
|
515
|
+
user_wants: manifest.user_wants,
|
|
516
|
+
reason: manifest.reason,
|
|
517
|
+
assessment: manifest.assessment,
|
|
518
|
+
requires_prior_tool_call: manifest.requires_prior_tool_call,
|
|
519
|
+
raw_requires_prior: parsed.requires_prior_tool_call, // debug — what classifier actually returned
|
|
520
|
+
fallback: false,
|
|
521
|
+
});
|
|
522
|
+
|
|
523
|
+
return manifest;
|
|
524
|
+
}
|
|
525
|
+
|
|
526
|
+
/**
|
|
527
|
+
* Build a safe fallback manifest: complex_multi with the built-in tool surface.
|
|
528
|
+
* Used when classification fails for any reason.
|
|
529
|
+
*
|
|
530
|
+
* It used to hand over EVERY live tool, MCP surfaces included. Measured 2026-09-20 on
|
|
531
|
+
* one task: 191 schemas is 33k tokens per call against 6k for the built-ins alone, the
|
|
532
|
+
* turn costs about $0.022 instead of $0.003, and the agent wanders 5 to 16 steps where
|
|
533
|
+
* a narrowed turn takes 2. Every expensive run that day came through this path. MCP
|
|
534
|
+
* servers are optional integrations; the built-ins are the agent's own hands, so a
|
|
535
|
+
* failed classification now costs the user a retry on an MCP task instead of costing
|
|
536
|
+
* money and steps on every task. FLINT_FALLBACK_ALL_TOOLS=1 restores the old shape.
|
|
537
|
+
*
|
|
538
|
+
* 2026-10-02: hidden was too far. An operator who connected Screenbox got an agent
|
|
539
|
+
* that said it had no Screenbox tools and could not do the task. Now a few MCP tools
|
|
540
|
+
* (MCP_INLINE_MAX) are handed over whole, and past that tool_search lists every
|
|
541
|
+
* connected server in its description and loads what the turn asks for.
|
|
542
|
+
*/
|
|
543
|
+
function fallbackManifest(reason, availableTools) {
|
|
544
|
+
const spec = resolveIntent("complex_multi");
|
|
545
|
+
const all = (availableTools || []).map(t => t.function?.name || t.name).filter(Boolean);
|
|
546
|
+
const builtin = all.filter(n => !n.startsWith("mcp_"));
|
|
547
|
+
const mcp = all.filter(n => n.startsWith("mcp_"));
|
|
548
|
+
// A few MCP tools go in whole; their schemas cost little and a search would
|
|
549
|
+
// be a wasted step. Past MCP_INLINE_MAX the built-ins go in with
|
|
550
|
+
// tool_search, whose description lists what it can load
|
|
551
|
+
// (tool-search.js mcpCatalog), plus whatever a search already loaded.
|
|
552
|
+
const loaded = new Set(loadedToolNames());
|
|
553
|
+
const toolNames = config.fallbackAllTools || builtin.length === 0 || mcp.length <= mcpInlineMax()
|
|
554
|
+
? all
|
|
555
|
+
: [...builtin, ...mcp.filter(n => loaded.has(n))];
|
|
556
|
+
return {
|
|
557
|
+
intent: "complex_multi",
|
|
558
|
+
tools: toolNames,
|
|
559
|
+
max_steps: spec.max_steps,
|
|
560
|
+
expected: spec.expected,
|
|
561
|
+
changes: spec.changes,
|
|
562
|
+
user_wants: "",
|
|
563
|
+
reason,
|
|
564
|
+
assessment: "normal", // fallback always proceeds normally — gate only fires when classifier explicitly flags
|
|
565
|
+
fallback: true,
|
|
566
|
+
fallbackScope: toolNames.length === all.length ? "all" : "builtin",
|
|
567
|
+
};
|
|
568
|
+
}
|
|
569
|
+
|
|
570
|
+
/**
|
|
571
|
+
* Apply an intent manifest to a list of tool definitions.
|
|
572
|
+
* Returns the filtered subset the agent loop should expose to the model.
|
|
573
|
+
*
|
|
574
|
+
* @param {Array} allTools — full tool definitions from registry
|
|
575
|
+
* @param {object} manifest — from classifyIntent()
|
|
576
|
+
* @returns {Array} filtered tool definitions
|
|
577
|
+
*/
|
|
578
|
+
export function filterToolsByManifest(allTools, manifest) {
|
|
579
|
+
if (!manifest) return allTools;
|
|
580
|
+
const wanted = manifest.tools;
|
|
581
|
+
if (!Array.isArray(wanted)) return allTools;
|
|
582
|
+
// Tree-mode widening: when the resolved intent spec carries a tool_pattern
|
|
583
|
+
// regex, union the classifier's picks with all tools whose name matches
|
|
584
|
+
// the pattern. This is the pragmatic "tool family" fix — the
|
|
585
|
+
// classifier's exact picks are advisory, the intent-wide family is the
|
|
586
|
+
// real selection. Text-only intents (wanted=[] AND no tool_pattern) still
|
|
587
|
+
// get no tools.
|
|
588
|
+
const pattern = manifest.tool_pattern;
|
|
589
|
+
if (wanted.length === 0 && !pattern) return [];
|
|
590
|
+
const allowed = new Set(wanted);
|
|
591
|
+
return allTools.filter(t => {
|
|
592
|
+
const name = t.function?.name || t.name;
|
|
593
|
+
if (!name) return false;
|
|
594
|
+
if (allowed.has(name)) return true;
|
|
595
|
+
if (pattern && pattern.test(name)) return true;
|
|
596
|
+
return false;
|
|
597
|
+
});
|
|
598
|
+
}
|
|
599
|
+
|
|
600
|
+
/** Build a short hint block to inject into the system message. */
|
|
601
|
+
export function formatIntentHint(manifest) {
|
|
602
|
+
if (!manifest || manifest.fallback) return "";
|
|
603
|
+
const isTextOnly = Array.isArray(manifest.tools) && manifest.tools.length === 0;
|
|
604
|
+
const toolDesc = isTextOnly
|
|
605
|
+
? "none (respond with text only)"
|
|
606
|
+
: manifest.tools.join(", ");
|
|
607
|
+
|
|
608
|
+
const parts = [
|
|
609
|
+
`Classified as: ${manifest.intent}`,
|
|
610
|
+
`User wants: ${manifest.user_wants || "(see message)"}`,
|
|
611
|
+
`Available tools: ${toolDesc}`,
|
|
612
|
+
`Max steps: ${manifest.max_steps}`,
|
|
613
|
+
`Expected output: ${manifest.expected}`,
|
|
614
|
+
];
|
|
615
|
+
|
|
616
|
+
if (isTextOnly) {
|
|
617
|
+
// Override the default Execution Flow (Intent/Action/Evaluate cycle with EXPECT markers):
|
|
618
|
+
// for text-only intents there is no tool call to evaluate — the model must answer directly.
|
|
619
|
+
parts.push("");
|
|
620
|
+
parts.push("IMPORTANT: This is a text-only request. Do NOT write 'EXPECT: ...' or describe what you plan to do.");
|
|
621
|
+
parts.push("Answer the user directly with the final result in one message.");
|
|
622
|
+
} else {
|
|
623
|
+
// For tool-backed intents the model has historically sometimes shortcircuited
|
|
624
|
+
// to a text response without ever calling a tool (e.g. classifier picks
|
|
625
|
+
// file_write for "translate and save", but Gemini Flash just outputs the
|
|
626
|
+
// translation in chat and forgets the file). Force the model to use a tool.
|
|
627
|
+
parts.push("");
|
|
628
|
+
parts.push("IMPORTANT: This task requires you to call at least one of the available tools above.");
|
|
629
|
+
parts.push("Do NOT respond with text only and consider the task done — you MUST produce a tool_call to fulfil the request (e.g. write_file to save a result, run_command to execute, etc).");
|
|
630
|
+
}
|
|
631
|
+
|
|
632
|
+
return `\n\n<intent>\n${parts.join("\n")}\n</intent>`;
|
|
633
|
+
}
|