flint-agent 1.14.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.example +108 -0
- package/CHANGELOG.md +55 -0
- package/FEATURES.md +298 -0
- package/LICENSE +21 -0
- package/README.md +435 -0
- package/bin/flint.js +47 -0
- package/config/classifier-prompt.md +218 -0
- package/config/models-curated.json +4 -0
- package/config/providers.json +74 -0
- package/package.json +92 -0
- package/patches/ink+6.8.0.patch +78 -0
- package/profiles/desktop.md +65 -0
- package/profiles/generic.md +20 -0
- package/profiles/marketer.md +20 -0
- package/profiles/profiles.json +34 -0
- package/profiles/ux-reviewer.md +25 -0
- package/src/agent/agent.js +1743 -0
- package/src/agent/auto.js +346 -0
- package/src/agent/backoff.js +143 -0
- package/src/agent/compression.js +310 -0
- package/src/agent/content-resolver.js +180 -0
- package/src/agent/flow-controller.js +309 -0
- package/src/agent/intent-manifest.js +231 -0
- package/src/agent/intent-timeout.js +46 -0
- package/src/agent/intent.js +633 -0
- package/src/agent/knowledge.js +114 -0
- package/src/agent/learning.js +180 -0
- package/src/agent/modes.js +187 -0
- package/src/agent/outcome-ask.js +91 -0
- package/src/agent/project-context.js +76 -0
- package/src/agent/prompt-budget.js +117 -0
- package/src/agent/reflection-extractor.js +140 -0
- package/src/agent/steering.js +86 -0
- package/src/agent/supervisor.js +430 -0
- package/src/agent/swap.js +443 -0
- package/src/agent/system-prompt.js +446 -0
- package/src/agent/time-stamp.js +48 -0
- package/src/agent/tool-guard.js +201 -0
- package/src/agent/toolcall-text.js +162 -0
- package/src/agent/usage.js +297 -0
- package/src/agent/vision.js +94 -0
- package/src/agent/watchdog.js +139 -0
- package/src/agent/workspace-changes.js +177 -0
- package/src/api/address.js +14 -0
- package/src/api/client.js +280 -0
- package/src/api/server.js +535 -0
- package/src/api/stream-pipe.js +113 -0
- package/src/app-state.js +39 -0
- package/src/bootstrap.js +501 -0
- package/src/bus/drain-loop.js +497 -0
- package/src/bus/index.js +270 -0
- package/src/bus/plugins.js +65 -0
- package/src/child-idle.js +14 -0
- package/src/cli.js +118 -0
- package/src/commands/commands.js +1297 -0
- package/src/commands/registry.js +132 -0
- package/src/components/App.js +491 -0
- package/src/components/CarefulMenu.js +145 -0
- package/src/components/HistoryWriter.js +86 -0
- package/src/components/LineInput.js +69 -0
- package/src/components/LiveZone.js +294 -0
- package/src/components/OverlayMenu.js +179 -0
- package/src/components/SystemPanel.js +156 -0
- package/src/components/Table.js +54 -0
- package/src/config.js +249 -0
- package/src/free-models.js +230 -0
- package/src/index.js +1111 -0
- package/src/input-handler.js +13 -0
- package/src/input-text.js +123 -0
- package/src/launcher.js +129 -0
- package/src/logging/api-log.js +95 -0
- package/src/logging/chat-log-follower.js +113 -0
- package/src/logging/chat-log.js +15 -0
- package/src/logging/log-collector.js +182 -0
- package/src/logging/logger.js +112 -0
- package/src/logging/tool-log.js +20 -0
- package/src/mcp-client.js +314 -0
- package/src/memory/conversation-digest.js +113 -0
- package/src/memory/extract-facts.js +98 -0
- package/src/memory/facts.js +181 -0
- package/src/memory/inbox.js +63 -0
- package/src/memory/markdown.js +38 -0
- package/src/memory/patterns.js +185 -0
- package/src/memory/project.js +66 -0
- package/src/memory/reflections.js +74 -0
- package/src/memory/retrieval.js +84 -0
- package/src/memory/rules.js +105 -0
- package/src/memory/session-facts.js +125 -0
- package/src/memory/skills.js +191 -0
- package/src/memory/sqlite-store.js +653 -0
- package/src/memory/store.js +208 -0
- package/src/memory/tools.js +196 -0
- package/src/memory/user-model.js +86 -0
- package/src/message-handler.js +775 -0
- package/src/model-check.js +218 -0
- package/src/plugins/loader.js +120 -0
- package/src/plugins/manager.js +88 -0
- package/src/production-env.js +22 -0
- package/src/profiles.js +42 -0
- package/src/providers/adapters/anthropic.js +270 -0
- package/src/providers/adapters/openai.js +120 -0
- package/src/providers/keys-dpapi.js +41 -0
- package/src/providers/keys-fallback.js +31 -0
- package/src/providers/keys.js +132 -0
- package/src/providers/models.js +154 -0
- package/src/providers/registry.js +56 -0
- package/src/providers/state.js +56 -0
- package/src/registry.js +96 -0
- package/src/restart.js +29 -0
- package/src/sandbox/backend.js +130 -0
- package/src/security/api-auth.js +132 -0
- package/src/security/audit.js +98 -0
- package/src/security/child-policy.js +41 -0
- package/src/security/command-guard.js +173 -0
- package/src/security/content-fence.js +250 -0
- package/src/security/content-validator.js +132 -0
- package/src/security/index.js +143 -0
- package/src/security/network-guard.js +126 -0
- package/src/security/pairing.js +180 -0
- package/src/security/path-guard.js +140 -0
- package/src/security/persona-guard.js +67 -0
- package/src/security/policies.js +452 -0
- package/src/security/safety-constants.js +34 -0
- package/src/security/watchdog.js +107 -0
- package/src/sessions.js +130 -0
- package/src/spend.js +97 -0
- package/src/startup-watchdog.js +59 -0
- package/src/stdio/args.js +71 -0
- package/src/stdio/guard.js +59 -0
- package/src/stdio/protocol.js +167 -0
- package/src/stdio/run.js +106 -0
- package/src/stdio/session.js +180 -0
- package/src/store/agent-slice.js +306 -0
- package/src/store/dataset-slice.js +73 -0
- package/src/store/index.js +22 -0
- package/src/store/process-slice.js +135 -0
- package/src/store/session-slice.js +191 -0
- package/src/store/ui-slice.js +119 -0
- package/src/tasks/db.js +184 -0
- package/src/tasks/queries.js +589 -0
- package/src/tools/agent-tools.js +473 -0
- package/src/tools/checkpoint.js +152 -0
- package/src/tools/command-approvals.js +180 -0
- package/src/tools/dataset.js +50 -0
- package/src/tools/filesystem.js +682 -0
- package/src/tools/inbox-tools.js +48 -0
- package/src/tools/mesh.js +135 -0
- package/src/tools/own-env.js +136 -0
- package/src/tools/permissions.js +681 -0
- package/src/tools/plugin-tools.js +123 -0
- package/src/tools/process-tools.js +595 -0
- package/src/tools/registry.js +307 -0
- package/src/tools/swap-tools.js +72 -0
- package/src/tools/system.js +662 -0
- package/src/tools/tasks.js +532 -0
- package/src/tools/tool-search.js +171 -0
- package/src/ui/header.js +140 -0
- package/src/ui/input-cursor.js +23 -0
- package/src/ui/last-line.js +25 -0
- package/src/ui/line-edit.js +135 -0
- package/src/ui/output.js +399 -0
- package/src/ui/paste-tokens.js +131 -0
- package/src/ui/prompt-attention.js +134 -0
- package/src/ui/render-options.js +13 -0
- package/src/ui/replay.js +94 -0
- package/src/ui/splash.js +49 -0
- package/src/ui/status-level.js +36 -0
- package/src/ui/tool-ledger.js +203 -0
- package/src/ui/window-title.js +150 -0
- package/src/update.js +205 -0
- package/system.md +63 -0
|
@@ -0,0 +1,201 @@
|
|
|
1
|
+
// Who decides which tools a turn gets, and when that decision is not trusted.
|
|
2
|
+
//
|
|
3
|
+
// The classifier runs BEFORE the agent has read the task, and its picks are a
|
|
4
|
+
// guess. On 2026-09-29 the guess took the tools away from the task three times
|
|
5
|
+
// in one evening and nobody was told:
|
|
6
|
+
//
|
|
7
|
+
// "Read <brief> and do what it says" -> file_read (1 tool, 3 steps)
|
|
8
|
+
// "so why did you stop" -> chat (0 tools)
|
|
9
|
+
// "Continue the task: write the fix and the
|
|
10
|
+
// tests as files ..., run them with npx
|
|
11
|
+
// vitest, commit." -> chat (0 tools)
|
|
12
|
+
//
|
|
13
|
+
// The third one is the worst: a turn that arrives in the middle of work, or
|
|
14
|
+
// that points at a file holding the instructions, has by definition not
|
|
15
|
+
// described all the work. A class list and a one-sentence paraphrase are not
|
|
16
|
+
// enough to decide that read_file is all it needs, and no one was shown the
|
|
17
|
+
// narrowing.
|
|
18
|
+
//
|
|
19
|
+
// Two rules, both about trust rather than about wording:
|
|
20
|
+
//
|
|
21
|
+
// 1. A guess is never the whole reason a turn loses its tools. When the
|
|
22
|
+
// message is mid-task (tools already ran in this session) or points at a
|
|
23
|
+
// file with instructions, the turn gets the full built-in surface and the
|
|
24
|
+
// classifier's class is kept only for display.
|
|
25
|
+
// 2. Whatever narrowing does happen is visible: the operator is told which
|
|
26
|
+
// class it was and how many tools survived. A silent narrowing is what
|
|
27
|
+
// makes a turn that could not be done look like a turn that gave up.
|
|
28
|
+
//
|
|
29
|
+
// Widening costs tokens — the fallback is every built-in tool, not the core
|
|
30
|
+
// set. That is the direction to pay in: a narrow turn that cannot do the task
|
|
31
|
+
// is worth more than the tokens it saved, and the price is only paid on the
|
|
32
|
+
// messages where the guess is known to be unsafe.
|
|
33
|
+
|
|
34
|
+
// What a message pointing at instructions usually looks like. Extensions only:
|
|
35
|
+
// a path with one of these is a thing someone wrote down, which is the whole
|
|
36
|
+
// reason the request cannot be read off its first line.
|
|
37
|
+
const INSTRUCTION_FILE = /[\w./\\-]*\.(?:md|markdown|txt|rst|adoc|brief|task|todo|instructions)\b/i;
|
|
38
|
+
|
|
39
|
+
// Phrases that hand the work to something outside the message itself.
|
|
40
|
+
const INSTRUCTION_PHRASES = [
|
|
41
|
+
/\bdo (?:what|as) (?:it|this|the file|the brief) says\b/i,
|
|
42
|
+
/\bas (?:it|the file|the brief) says\b/i,
|
|
43
|
+
/\bfollow (?:the )?(?:instructions?|steps?|directions?) in\b/i,
|
|
44
|
+
/\b(?:according to|per) the (?:instructions?|brief|file)\b/i,
|
|
45
|
+
/\bsee (?:the )?file\b.*\bfor\b/i,
|
|
46
|
+
/\bin (?:the|this) file\b/i,
|
|
47
|
+
];
|
|
48
|
+
|
|
49
|
+
/**
|
|
50
|
+
* Does this message hand the work to something else — a file, a brief, a
|
|
51
|
+
* list of instructions somewhere on disk?
|
|
52
|
+
*
|
|
53
|
+
* @param {string} text
|
|
54
|
+
* @returns {null | {kind: "file"|"phrase", path?: string}} null when it does not
|
|
55
|
+
*/
|
|
56
|
+
export function pointsAtInstructions(text) {
|
|
57
|
+
const s = String(text || "");
|
|
58
|
+
if (!s.trim()) return null;
|
|
59
|
+
const fileMatch = s.match(INSTRUCTION_FILE);
|
|
60
|
+
if (fileMatch) return { kind: "file", path: fileMatch[0] };
|
|
61
|
+
if (INSTRUCTION_PHRASES.some((re) => re.test(s))) return { kind: "phrase" };
|
|
62
|
+
return null;
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
/**
|
|
66
|
+
* Is this message arriving in the middle of work?
|
|
67
|
+
*
|
|
68
|
+
* `priorToolCalls` is the fact that matters, not the number of messages: if
|
|
69
|
+
* this session already ran a tool, the operator is in the middle of a task,
|
|
70
|
+
* and a bare message in the middle of a task is a continuation, not a new
|
|
71
|
+
* subject. `priorTurns` is kept for the softer case — a second message with no
|
|
72
|
+
* tool yet is usually a follow-up of the first ("why?").
|
|
73
|
+
*
|
|
74
|
+
* @param {object} ctx - { priorToolCalls?: boolean, priorTurns?: number }
|
|
75
|
+
* @returns {boolean}
|
|
76
|
+
*/
|
|
77
|
+
export function isMidTask(ctx = {}) {
|
|
78
|
+
if (ctx.priorToolCalls) return true;
|
|
79
|
+
return (ctx.priorTurns || 0) > 0;
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
// Below this, a tool set cannot do a task that involves writing, running and
|
|
83
|
+
// checking anything — which is what a mid-task message almost always is.
|
|
84
|
+
const MIN_TRUSTWORTHY_TOOLS = 6;
|
|
85
|
+
|
|
86
|
+
/**
|
|
87
|
+
* Is the classifier's tool set good enough to act on for this message?
|
|
88
|
+
*
|
|
89
|
+
* @param {object} args
|
|
90
|
+
* @param {object} args.manifest — the classifier's manifest
|
|
91
|
+
* @param {string} args.message — the user's message
|
|
92
|
+
* @param {object} [args.ctx] — { priorToolCalls, priorTurns }
|
|
93
|
+
* @returns {null | {untrusted: true, reason: string}} null when the set is trusted
|
|
94
|
+
*/
|
|
95
|
+
export function untrustedToolSet({ manifest, message, ctx = {} }) {
|
|
96
|
+
const picked = Array.isArray(manifest?.tools) ? manifest.tools.length : 0;
|
|
97
|
+
const midTask = isMidTask(ctx);
|
|
98
|
+
const pointer = pointsAtInstructions(message);
|
|
99
|
+
|
|
100
|
+
if (pointer) {
|
|
101
|
+
return {
|
|
102
|
+
untrusted: true,
|
|
103
|
+
reason: `the message points at instructions (${pointer.path || pointer.kind}), so the request is not in the message`,
|
|
104
|
+
};
|
|
105
|
+
}
|
|
106
|
+
if (picked === 0 && midTask) {
|
|
107
|
+
return { untrusted: true, reason: "a text-only class for a mid-task message" };
|
|
108
|
+
}
|
|
109
|
+
if (picked === 0 && pointer) {
|
|
110
|
+
return { untrusted: true, reason: "a text-only class for a message that points at instructions" };
|
|
111
|
+
}
|
|
112
|
+
if (midTask && picked < MIN_TRUSTWORTHY_TOOLS) {
|
|
113
|
+
return {
|
|
114
|
+
untrusted: true,
|
|
115
|
+
reason: `only ${picked} tool${picked === 1 ? "" : "s"} for a message that arrives mid-task`,
|
|
116
|
+
};
|
|
117
|
+
}
|
|
118
|
+
return null;
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
/**
|
|
122
|
+
* One line for the operator about what this turn can do.
|
|
123
|
+
*
|
|
124
|
+
* Returns "" when nothing was narrowed: a turn with every tool is the normal
|
|
125
|
+
* case and a line about it would be noise on every message. When something
|
|
126
|
+
* was narrowed, the class and the count are both stated, because the count is
|
|
127
|
+
* the part that decides whether the task is possible.
|
|
128
|
+
*
|
|
129
|
+
* @param {object} args - { manifest, allDefs, tools }
|
|
130
|
+
* @returns {string}
|
|
131
|
+
*/
|
|
132
|
+
export function describeToolScope({ manifest, allDefs = [], tools = [] } = {}) {
|
|
133
|
+
if (!manifest) return "";
|
|
134
|
+
const total = allDefs.length;
|
|
135
|
+
const names = tools.map((t) => t.function?.name || t.name).filter(Boolean);
|
|
136
|
+
if (!total) return "";
|
|
137
|
+
// Full surface is the normal case and a line about it would be noise on every
|
|
138
|
+
// message; anything else, including zero tools, is stated.
|
|
139
|
+
if (names.length >= total) return "";
|
|
140
|
+
const shown = names.slice(0, 8).join(", ");
|
|
141
|
+
const more = names.length > 8 ? `, +${names.length - 8} more` : "";
|
|
142
|
+
const list = names.length ? `(${shown}${more})` : "(none)";
|
|
143
|
+
return `mode ${manifest.intent}: ${names.length} of ${total} tools ${list}`;
|
|
144
|
+
}
|
|
145
|
+
|
|
146
|
+
/**
|
|
147
|
+
* The tool list this turn actually gets, and one line about it for the
|
|
148
|
+
* operator.
|
|
149
|
+
*
|
|
150
|
+
* The assessment gate wins over everything: a `dangerous` classification is a
|
|
151
|
+
* deliberate refusal to hand over the tools, and this module is not entitled
|
|
152
|
+
* to overrule it. That is the difference between a guess that may be wrong
|
|
153
|
+
* (the classifier's tool picks) and a decision that was made on purpose.
|
|
154
|
+
*
|
|
155
|
+
* The fallback shape is the built-in surface rather than every live tool, for
|
|
156
|
+
* the same reason fallbackManifest uses it: MCP schemas cost 33k tokens a call
|
|
157
|
+
* and the MCP tools can be fetched by tool_search when the turn finds it needs
|
|
158
|
+
* them. Measured in src/agent/intent.js.
|
|
159
|
+
*
|
|
160
|
+
* @param {object} args
|
|
161
|
+
* @param {object} args.manifest — the classifier's manifest
|
|
162
|
+
* @param {Array} args.allDefs — every live tool definition
|
|
163
|
+
* @param {Array} [args.narrowedTools] — what filterToolsByManifest returned
|
|
164
|
+
* @param {boolean} [args.blockTools] — the assessment gate took the tools
|
|
165
|
+
* @param {string} [args.message] — the user's message
|
|
166
|
+
* @param {object} [args.ctx] — { priorToolCalls, priorTurns }
|
|
167
|
+
* @returns {{tools: Array, note: string, widened: boolean}}
|
|
168
|
+
*/
|
|
169
|
+
export function decideToolScope({
|
|
170
|
+
manifest,
|
|
171
|
+
allDefs = [],
|
|
172
|
+
narrowedTools = [],
|
|
173
|
+
blockTools = false,
|
|
174
|
+
message = "",
|
|
175
|
+
ctx = {},
|
|
176
|
+
} = {}) {
|
|
177
|
+
const defs = Array.isArray(allDefs) ? allDefs : [];
|
|
178
|
+
const nameOf = (t) => t.function?.name || t.name;
|
|
179
|
+
|
|
180
|
+
if (blockTools) {
|
|
181
|
+
return { tools: [], note: describeToolScope({ manifest, allDefs: defs, tools: [] }), widened: false };
|
|
182
|
+
}
|
|
183
|
+
|
|
184
|
+
const trust = untrustedToolSet({ manifest, message, ctx });
|
|
185
|
+
if (!trust) {
|
|
186
|
+
return { tools: narrowedTools, note: describeToolScope({ manifest, allDefs: defs, tools: narrowedTools }), widened: false };
|
|
187
|
+
}
|
|
188
|
+
|
|
189
|
+
const builtin = defs.filter((t) => !String(nameOf(t) || "").startsWith("mcp_"));
|
|
190
|
+
const widened = builtin.length > 0 ? builtin : defs;
|
|
191
|
+
// The count is always stated, including when the widened set is every tool:
|
|
192
|
+
// "all 13 tools" and "1 of 13 tools" are different facts to the operator, and
|
|
193
|
+
// the whole point of this line is that the difference is no longer invisible.
|
|
194
|
+
const names = widened.map(nameOf).filter(Boolean);
|
|
195
|
+
const shown = names.slice(0, 8).join(", ");
|
|
196
|
+
const more = names.length > 8 ? `, +${names.length - 8} more` : "";
|
|
197
|
+
const scopeText = `mode ${manifest?.intent || "unknown"}: ${names.length} of ${defs.length} tools`;
|
|
198
|
+
const note = `[${manifest?.intent || "unknown"}] ${scopeText}${shown ? ` (${shown}${more})` : ""}` +
|
|
199
|
+
` — the tools were widened: ${trust.reason}. Nothing is lost; if this turns out to be the wrong read, say so.`;
|
|
200
|
+
return { tools: widened, note, widened: true };
|
|
201
|
+
}
|
|
@@ -0,0 +1,162 @@
|
|
|
1
|
+
// Tool calls the model wrote as text.
|
|
2
|
+
//
|
|
3
|
+
// On 2026-09-29 a turn classified `chat` (0 tools, 1 step) had the model write
|
|
4
|
+
// its tool calls into the answer:
|
|
5
|
+
//
|
|
6
|
+
// <tool_call><function=edit_file>... 9 KB of it ...
|
|
7
|
+
//
|
|
8
|
+
// Nothing ran it. The text streamed to the console as if it were progress, the
|
|
9
|
+
// operator saw nine kilobytes of a diff scroll past, and the only thing that
|
|
10
|
+
// recovered the session was /new.
|
|
11
|
+
//
|
|
12
|
+
// Two things were wrong with that, and they are separate:
|
|
13
|
+
//
|
|
14
|
+
// 1. The text was printed as progress. Whatever the model meant, a turn that
|
|
15
|
+
// cannot run what it wrote must not LOOK like it ran it.
|
|
16
|
+
// 2. Nothing said so. The answer came back as `done`, and a task that
|
|
17
|
+
// changed nothing reads exactly like a task that finished.
|
|
18
|
+
//
|
|
19
|
+
// So: find the calls, and either Flint can run them or the operator is told
|
|
20
|
+
// that the model answered in a form Flint could not run. Never neither.
|
|
21
|
+
|
|
22
|
+
// One model puts a zero-width space inside the tag so that markdown renderers
|
|
23
|
+
// leave it alone. It is written as an escape here rather than as the character,
|
|
24
|
+
// because a literal invisible character inside a regex is unreadable and gets
|
|
25
|
+
// eaten by every editor that reformats a file.
|
|
26
|
+
const ZW = "\\u200b";
|
|
27
|
+
|
|
28
|
+
// No `g` on the two detectors. With it, `.test()` advances `lastIndex` and the
|
|
29
|
+
// next call starts mid-string, so the same answer is detected on one call and
|
|
30
|
+
// missed on the next -- a detector that alternates is worse than none. The
|
|
31
|
+
// scanning regexes keep `g` because they are always used with matchAll, which
|
|
32
|
+
// does not share that state.
|
|
33
|
+
const TOOL_CALL_TAG = new RegExp(`<\\s*${ZW}?\\s*tool_call\\s*>`, "i");
|
|
34
|
+
// Two shapes need the same pattern matched two ways, and the difference is
|
|
35
|
+
// not cosmetic: String.match() with a `g` flag returns the full matches and
|
|
36
|
+
// throws the capture groups away, so `fn[1]` — the tool name — comes back
|
|
37
|
+
// undefined. A scan regex and a single-match regex, both built from one source.
|
|
38
|
+
const FUNCTION_SRC = `<\\s*${ZW}?\\s*function\\s*=\\s*([A-Za-z0-9_.-]+)\\s*>`;
|
|
39
|
+
const FUNCTION_TAG = new RegExp(FUNCTION_SRC, "gi"); // matchAll
|
|
40
|
+
const FUNCTION_ONE = new RegExp(FUNCTION_SRC, "i"); // .match(), keeps groups
|
|
41
|
+
const FUNCTION_OPEN = new RegExp(`<\\s*${ZW}?\\s*function\\s*=`, "i");
|
|
42
|
+
// The closing tag carries the same zero-width character as the opening one, so
|
|
43
|
+
// it has to be built the same way. Without this the body ran to end-of-answer
|
|
44
|
+
// and every JSON-body call came back truncated — which the "unreadable" branch
|
|
45
|
+
// reported honestly, but the tool name was lost with it.
|
|
46
|
+
const CLOSE_TAG = new RegExp(`<\\s*${ZW}?\\s*\\/\\s*(?:tool_call|function)\\s*>`, "i");
|
|
47
|
+
|
|
48
|
+
/**
|
|
49
|
+
* Does this text contain a tool call written as text?
|
|
50
|
+
*
|
|
51
|
+
* Cheap and prefix-friendly, so it can be run on a growing stream buffer: it
|
|
52
|
+
* answers "has this started?", not "is it complete?".
|
|
53
|
+
*
|
|
54
|
+
* @param {string} text
|
|
55
|
+
* @returns {boolean}
|
|
56
|
+
*/
|
|
57
|
+
export function looksLikeTextToolCall(text) {
|
|
58
|
+
const s = String(text || "");
|
|
59
|
+
if (!s) return false;
|
|
60
|
+
return TOOL_CALL_TAG.test(s) || FUNCTION_OPEN.test(s);
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
/**
|
|
64
|
+
* Pull the tool calls out of an answer that wrote them as text.
|
|
65
|
+
*
|
|
66
|
+
* Both shapes seen in the wild:
|
|
67
|
+
* <tool_call><function=edit_file>{"path":"a.js"}</tool_call> (function tag)
|
|
68
|
+
* <tool_call>{"name":"edit_file","arguments":{...}}</tool_call> (JSON body)
|
|
69
|
+
*
|
|
70
|
+
* Arguments are reported as the raw text that followed the name: Flint does not
|
|
71
|
+
* run them (they may be truncated mid-JSON by the model's own output limits),
|
|
72
|
+
* and a half-written argument string is exactly what the operator needs to see
|
|
73
|
+
* named as unrunnable rather than silently dropped.
|
|
74
|
+
*
|
|
75
|
+
* @param {string} text
|
|
76
|
+
* @returns {Array<{name: string, args: string}>} possibly empty
|
|
77
|
+
*/
|
|
78
|
+
export function findTextToolCalls(text) {
|
|
79
|
+
const s = String(text || "");
|
|
80
|
+
if (!looksLikeTextToolCall(s)) return [];
|
|
81
|
+
|
|
82
|
+
const calls = [];
|
|
83
|
+
// One pass over the opening tags, not one pass per shape.
|
|
84
|
+
//
|
|
85
|
+
// The two shapes nest: `<tool_call><function=edit_file>{...}</tool_call>`
|
|
86
|
+
// is a tool_call whose body is a function tag. Scanning for both
|
|
87
|
+
// independently counts it twice, and a greedy body scan then swallows the
|
|
88
|
+
// NEXT call, so the second one comes back as "unreadable". Both shapes were
|
|
89
|
+
// counted wrong before this: one call reported as two, then two reported as
|
|
90
|
+
// one good and one unreadable.
|
|
91
|
+
// matchAll requires the g flag, so the scan gets its own copy rather than
|
|
92
|
+
// the detector being reused with state attached to it.
|
|
93
|
+
const OPEN_SCAN = new RegExp(TOOL_CALL_TAG.source, "gi");
|
|
94
|
+
OPEN_SCAN.lastIndex = 0;
|
|
95
|
+
for (const open of s.matchAll(OPEN_SCAN)) {
|
|
96
|
+
const rest = s.slice(open.index + open[0].length);
|
|
97
|
+
const close = rest.search(CLOSE_TAG);
|
|
98
|
+
// No closing tag means the model was cut off mid-call. The rest of the
|
|
99
|
+
// answer IS the body, which is the 9 KB case.
|
|
100
|
+
const body = (close === -1 ? rest : rest.slice(0, close)).trim();
|
|
101
|
+
|
|
102
|
+
// Shape A: a function tag, with the name in the tag.
|
|
103
|
+
const fn = body.match(FUNCTION_ONE);
|
|
104
|
+
if (fn) {
|
|
105
|
+
const after = body.slice(body.indexOf(fn[0]) + fn[0].length);
|
|
106
|
+
const inner = after.trim();
|
|
107
|
+
calls.push({ name: fn[1], args: inner || "" });
|
|
108
|
+
continue;
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
// Shape B: a JSON body carrying its own name.
|
|
112
|
+
if (!body) {
|
|
113
|
+
// The tag opened and the model stopped before writing anything.
|
|
114
|
+
calls.push({ name: "(unreadable)", args: "" });
|
|
115
|
+
continue;
|
|
116
|
+
}
|
|
117
|
+
const brace = body.indexOf("{");
|
|
118
|
+
const json = brace === -1 ? body : body.slice(brace);
|
|
119
|
+
try {
|
|
120
|
+
const parsed = JSON.parse(json);
|
|
121
|
+
const name = parsed.name || parsed.function?.name || parsed.tool;
|
|
122
|
+
calls.push({
|
|
123
|
+
name: name ? String(name) : "(unreadable)",
|
|
124
|
+
args: JSON.stringify(parsed.arguments ?? parsed.args ?? {}),
|
|
125
|
+
});
|
|
126
|
+
} catch {
|
|
127
|
+
calls.push({ name: "(unreadable)", args: json.slice(0, 200) });
|
|
128
|
+
}
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
// A function tag with no enclosing tool_call, which is the shape seen in the
|
|
132
|
+
// logs with the outer tag missing.
|
|
133
|
+
if (calls.length === 0) {
|
|
134
|
+
FUNCTION_TAG.lastIndex = 0;
|
|
135
|
+
for (const m of s.matchAll(FUNCTION_TAG)) {
|
|
136
|
+
const rest = s.slice(m.index + m[0].length);
|
|
137
|
+
const close = rest.search(CLOSE_TAG);
|
|
138
|
+
calls.push({ name: m[1], args: (close === -1 ? rest : rest.slice(0, close)).trim() });
|
|
139
|
+
}
|
|
140
|
+
}
|
|
141
|
+
|
|
142
|
+
return calls;
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
/**
|
|
146
|
+
* What the operator is told when the model wrote tool calls as text.
|
|
147
|
+
*
|
|
148
|
+
* Says what Flint could not run and what to do, and never says the work was
|
|
149
|
+
* done. Short on purpose: this line replaces nine kilobytes of text nobody can
|
|
150
|
+
* act on.
|
|
151
|
+
*
|
|
152
|
+
* @param {Array<{name: string, args: string}>} calls
|
|
153
|
+
* @param {object} [opts] - { attempts?: number }
|
|
154
|
+
* @returns {string}
|
|
155
|
+
*/
|
|
156
|
+
export function toolCallTextNote(calls, { attempts = 0 } = {}) {
|
|
157
|
+
const names = [...new Set((calls || []).map((c) => c.name))];
|
|
158
|
+
const list = names.length ? names.join(", ") : "(no name readable)";
|
|
159
|
+
const again = attempts > 0 ? ` It wrote it as text ${attempts + 1} times in a row.` : "";
|
|
160
|
+
return `Stopped: the model wrote its tool call as text instead of calling the tool, so nothing ran. ` +
|
|
161
|
+
`Tool(s) written as text: ${list}. This answer did no work: read it, then tell me what to carry out.${again}`;
|
|
162
|
+
}
|
|
@@ -0,0 +1,297 @@
|
|
|
1
|
+
// Usage accounting: the one place that decides what an API call cost, and the
|
|
2
|
+
// one place that refuses the next one when the money is gone.
|
|
3
|
+
//
|
|
4
|
+
// WHY the refusal lives HERE and not at each ceiling: there were three
|
|
5
|
+
// guards reading three different numbers. agent.js priced the turn itself with
|
|
6
|
+
// the local formula below, and checked it AFTER the turn; the session total in
|
|
7
|
+
// the store was only topped up once the turn ended, so a ceiling consulted mid
|
|
8
|
+
// turn could not see what the classifier and the fact extractor had already
|
|
9
|
+
// spent; auto mode read a lagging delta of that same total. Three notebooks,
|
|
10
|
+
// and four modules spending outside all of them. It leaked twice in two days,
|
|
11
|
+
// in opposite directions.
|
|
12
|
+
//
|
|
13
|
+
// Now every call to the provider goes through src/api/client.js, and that door
|
|
14
|
+
// asks this module two questions in order: may I, and then here is what it
|
|
15
|
+
// cost. Nobody adds anything up, so nobody can forget to.
|
|
16
|
+
//
|
|
17
|
+
// WHY this exists: the cost was computed locally as
|
|
18
|
+
// promptTokens * pricing.prompt + completionTokens * pricing.completion.
|
|
19
|
+
// That is wrong by more than 2x on any model with prompt caching, because a
|
|
20
|
+
// cache read costs ~120x less than a miss ($0.0036/M vs $0.435/M on
|
|
21
|
+
// mimo-v2.5-pro), and the local table knows nothing about which tokens were
|
|
22
|
+
// read from cache. The number is what the "is this expensive?" decision is
|
|
23
|
+
// made on, so a wrong instrument is worse than none: on 2026-09-19 it led to
|
|
24
|
+
// the conclusion that caching was broken when it was working.
|
|
25
|
+
//
|
|
26
|
+
// OpenRouter already returns the exact charge in `usage.cost` on every call,
|
|
27
|
+
// streaming included, WITHOUT any request-body flag. So: take the provider's
|
|
28
|
+
// number when it is there, estimate only when it is not, and never let an
|
|
29
|
+
// estimate pass itself off as a fact.
|
|
30
|
+
|
|
31
|
+
import { createLogger } from "../logging/logger.js";
|
|
32
|
+
|
|
33
|
+
const log = createLogger("usage");
|
|
34
|
+
|
|
35
|
+
// Where the fallback rate card comes from when the provider reports no cost.
|
|
36
|
+
// Injected at startup (bootstrap.js) rather than imported, so this module stays
|
|
37
|
+
// free of the store and can be exercised on its own.
|
|
38
|
+
//
|
|
39
|
+
// It is the MAIN model's card, and a side call may run on a cheaper model, so
|
|
40
|
+
// what it produces is a ceiling, not a price. That is why it is reached only
|
|
41
|
+
// when `usage.cost` is absent, and why everything it touches comes back
|
|
42
|
+
// flagged `estimated`.
|
|
43
|
+
let pricingSource = () => null;
|
|
44
|
+
|
|
45
|
+
/** The model's context window in tokens, when the rate card knows it. */
|
|
46
|
+
export function contextWindow() {
|
|
47
|
+
return pricingSource()?.contextLength || null;
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
export function setPricingSource(fn) {
|
|
51
|
+
pricingSource = typeof fn === "function" ? fn : () => null;
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
// `facts` is not on wip/judge-4758: the branch instrumented the classifier and
|
|
55
|
+
// the extractor and missed memory/extract-facts.js, which agent.js fires on
|
|
56
|
+
// every user message. Keep it when merging the rest of that branch's work.
|
|
57
|
+
// `outcome` is the one question asked out of band: which of the three
|
|
58
|
+
// things happened on a turn that changed nothing. It is a model call like any
|
|
59
|
+
// other, so it has to show up in the bill, or the fix is invisible money.
|
|
60
|
+
export const USAGE_SOURCES = ["agent", "classifier", "judge", "extractor", "facts", "outcome", "swap"];
|
|
61
|
+
|
|
62
|
+
/**
|
|
63
|
+
* Normalize a provider `usage` object into the fields we account on.
|
|
64
|
+
* Understands the OpenAI/OpenRouter shape; the Anthropic adapter maps its
|
|
65
|
+
* own field names into the same shape before returning.
|
|
66
|
+
*
|
|
67
|
+
* @returns {null | {promptTokens, completionTokens, cachedTokens, cacheWriteTokens, cost}}
|
|
68
|
+
* `cost` is null when the provider did not report one.
|
|
69
|
+
*/
|
|
70
|
+
export function readUsage(usage) {
|
|
71
|
+
if (!usage) return null;
|
|
72
|
+
const details = usage.prompt_tokens_details || {};
|
|
73
|
+
return {
|
|
74
|
+
promptTokens: usage.prompt_tokens || 0,
|
|
75
|
+
completionTokens: usage.completion_tokens || 0,
|
|
76
|
+
cachedTokens: details.cached_tokens || 0,
|
|
77
|
+
cacheWriteTokens: details.cache_write_tokens || 0,
|
|
78
|
+
cost: typeof usage.cost === "number" ? usage.cost : null,
|
|
79
|
+
};
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
/**
|
|
83
|
+
* What did this call cost, and do we actually know?
|
|
84
|
+
*
|
|
85
|
+
* Provider number wins. The fallback estimate still prices cached tokens at
|
|
86
|
+
* the cache-read rate when the model publishes one, so it is at least the
|
|
87
|
+
* right shape, but it is returned flagged, and every surface shows the flag.
|
|
88
|
+
*
|
|
89
|
+
* @returns {{cost: number, estimated: boolean}}
|
|
90
|
+
*/
|
|
91
|
+
export function priceUsage(u, pricing) {
|
|
92
|
+
if (!u) return { cost: 0, estimated: false };
|
|
93
|
+
if (u.cost != null) return { cost: u.cost, estimated: false };
|
|
94
|
+
if (!pricing) return { cost: 0, estimated: true };
|
|
95
|
+
|
|
96
|
+
// Cached tokens are included in prompt_tokens, so bill them once, cheaply.
|
|
97
|
+
const uncached = Math.max(0, u.promptTokens - u.cachedTokens);
|
|
98
|
+
const cacheRate = pricing.cacheRead != null ? pricing.cacheRead : pricing.prompt;
|
|
99
|
+
const cost =
|
|
100
|
+
uncached * pricing.prompt +
|
|
101
|
+
u.cachedTokens * cacheRate +
|
|
102
|
+
u.completionTokens * pricing.completion;
|
|
103
|
+
return { cost, estimated: true };
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
function emptyEntry() {
|
|
107
|
+
return {
|
|
108
|
+
calls: 0,
|
|
109
|
+
promptTokens: 0,
|
|
110
|
+
completionTokens: 0,
|
|
111
|
+
cachedTokens: 0,
|
|
112
|
+
cacheWriteTokens: 0,
|
|
113
|
+
cost: 0,
|
|
114
|
+
estimated: false,
|
|
115
|
+
};
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
function emptyLedger() {
|
|
119
|
+
// Built FROM the list, not repeated by hand. Repeating it cost eleven model
|
|
120
|
+
// calls: `outcome` was added to USAGE_SOURCES, the ledger was not touched,
|
|
121
|
+
// and recordUsage dropped every one of them on the floor because the entry
|
|
122
|
+
// did not exist. The list is the one place a source is declared.
|
|
123
|
+
const ledger = {};
|
|
124
|
+
for (const source of USAGE_SOURCES) {
|
|
125
|
+
ledger[source] = emptyEntry();
|
|
126
|
+
}
|
|
127
|
+
return ledger;
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
// What has been spent since the last drain, by source. `agent` has a row like
|
|
131
|
+
// everything else: the main loop is not a special kind of money. Before, it
|
|
132
|
+
// was priced separately by agent.js with the local formula, which is how the
|
|
133
|
+
// per-turn ceiling ended up guarding a number that was wrong by more than 2x on
|
|
134
|
+
// a cached model.
|
|
135
|
+
let ledger = emptyLedger();
|
|
136
|
+
|
|
137
|
+
// The three running totals, in dollars. They are three WINDOWS onto the same
|
|
138
|
+
// stream of calls, not three tallies: every recorded call moves all three.
|
|
139
|
+
//
|
|
140
|
+
// action one turn. Reset by beginAction().
|
|
141
|
+
// session everything since the session was opened or resumed.
|
|
142
|
+
// run one autonomous run. Counted only while a run is open, because a
|
|
143
|
+
// run's ceiling is its own budget, not the session's history.
|
|
144
|
+
//
|
|
145
|
+
// Draining the by-source ledger for display does NOT touch them. A total that
|
|
146
|
+
// resets when somebody looks at it is not a total.
|
|
147
|
+
let spend = { action: 0, session: 0, run: 0 };
|
|
148
|
+
let runOpen = false;
|
|
149
|
+
let runLimit = 0;
|
|
150
|
+
|
|
151
|
+
/** Thrown by the door instead of making the call. */
|
|
152
|
+
export class BudgetExceededError extends Error {
|
|
153
|
+
constructor({ scope, spent, limit }) {
|
|
154
|
+
super(`Budget exhausted (${scope}): $${spent.toFixed(4)} of $${limit.toFixed(2)}`);
|
|
155
|
+
this.name = "BudgetExceededError";
|
|
156
|
+
this.isBudgetError = true;
|
|
157
|
+
this.scope = scope;
|
|
158
|
+
this.spent = spent;
|
|
159
|
+
this.limit = limit;
|
|
160
|
+
}
|
|
161
|
+
}
|
|
162
|
+
|
|
163
|
+
/** @returns {{action: number, session: number, run: number}} dollars spent */
|
|
164
|
+
export function getSpend() {
|
|
165
|
+
return { ...spend };
|
|
166
|
+
}
|
|
167
|
+
|
|
168
|
+
/** A new turn starts. The per-action ceiling counts from here. */
|
|
169
|
+
export function beginAction() {
|
|
170
|
+
spend.action = 0;
|
|
171
|
+
}
|
|
172
|
+
|
|
173
|
+
/**
|
|
174
|
+
* An autonomous run starts, with its own ceiling.
|
|
175
|
+
* @param {number} limit dollars; 0 or less means no ceiling of its own
|
|
176
|
+
*/
|
|
177
|
+
export function beginRun(limit = 0) {
|
|
178
|
+
spend.run = 0;
|
|
179
|
+
runLimit = limit > 0 ? limit : 0;
|
|
180
|
+
runOpen = true;
|
|
181
|
+
}
|
|
182
|
+
|
|
183
|
+
export function endRun() {
|
|
184
|
+
runOpen = false;
|
|
185
|
+
runLimit = 0;
|
|
186
|
+
}
|
|
187
|
+
|
|
188
|
+
/** Restore the session total when a saved session is reopened. */
|
|
189
|
+
export function seedSessionSpend(cost) {
|
|
190
|
+
spend.session = typeof cost === "number" && cost > 0 ? cost : 0;
|
|
191
|
+
}
|
|
192
|
+
|
|
193
|
+
/** Start a fresh session: the session window goes back to zero, the run closes. */
|
|
194
|
+
export function resetSessionSpend() {
|
|
195
|
+
spend = { action: 0, session: 0, run: 0 };
|
|
196
|
+
endRun();
|
|
197
|
+
ledger = emptyLedger();
|
|
198
|
+
}
|
|
199
|
+
|
|
200
|
+
/**
|
|
201
|
+
* Which ceiling, if any, has already been reached. The per-action and session
|
|
202
|
+
* ceilings are read from config by the caller (they can change mid-session via
|
|
203
|
+
* /budget), the run ceiling belongs to the open run.
|
|
204
|
+
*
|
|
205
|
+
* A ceiling of 0 means unlimited, which is the default for both config values.
|
|
206
|
+
*
|
|
207
|
+
* @returns {null | {scope: "action"|"session"|"run", spent: number, limit: number}}
|
|
208
|
+
*/
|
|
209
|
+
export function overBudget({ perAction = 0, session = 0 } = {}) {
|
|
210
|
+
if (perAction > 0 && spend.action >= perAction) {
|
|
211
|
+
return { scope: "action", spent: spend.action, limit: perAction };
|
|
212
|
+
}
|
|
213
|
+
if (session > 0 && spend.session >= session) {
|
|
214
|
+
return { scope: "session", spent: spend.session, limit: session };
|
|
215
|
+
}
|
|
216
|
+
if (runOpen && runLimit > 0 && spend.run >= runLimit) {
|
|
217
|
+
return { scope: "run", spent: spend.run, limit: runLimit };
|
|
218
|
+
}
|
|
219
|
+
return null;
|
|
220
|
+
}
|
|
221
|
+
|
|
222
|
+
/** overBudget, but it refuses instead of reporting. Throws BudgetExceededError. */
|
|
223
|
+
export function assertWithinBudget(limits) {
|
|
224
|
+
const over = overBudget(limits);
|
|
225
|
+
if (over) {
|
|
226
|
+
log.warn("budget refusal", over);
|
|
227
|
+
throw new BudgetExceededError(over);
|
|
228
|
+
}
|
|
229
|
+
}
|
|
230
|
+
|
|
231
|
+
/**
|
|
232
|
+
* Record one call to the provider. Called by the door, by nobody else: a
|
|
233
|
+
* caller that has to remember this is a caller that will forget.
|
|
234
|
+
* Safe with a missing/!ok usage object.
|
|
235
|
+
* @param {"agent"|"classifier"|"judge"|"extractor"|"facts"|"outcome"} source
|
|
236
|
+
*/
|
|
237
|
+
export function recordUsage(source, usage, pricing = null) {
|
|
238
|
+
const entry = ledger[source];
|
|
239
|
+
if (!entry) {
|
|
240
|
+
// A source that is spent but not declared is money leaving with no row to
|
|
241
|
+
// show it. Say so loudly instead of returning quietly, which is how the
|
|
242
|
+
// `outcome` calls went missing in the first place.
|
|
243
|
+
log.error("usage for an unknown source", { source, known: Object.keys(ledger) });
|
|
244
|
+
return null;
|
|
245
|
+
}
|
|
246
|
+
const u = readUsage(usage);
|
|
247
|
+
if (!u) {
|
|
248
|
+
// The call happened and we cannot say what it cost. Saying nothing would
|
|
249
|
+
// silently understate the session, so mark the whole source as estimated.
|
|
250
|
+
entry.calls += 1;
|
|
251
|
+
entry.estimated = true;
|
|
252
|
+
return null;
|
|
253
|
+
}
|
|
254
|
+
const { cost, estimated } = priceUsage(u, pricing || pricingSource());
|
|
255
|
+
entry.calls += 1;
|
|
256
|
+
entry.promptTokens += u.promptTokens;
|
|
257
|
+
entry.completionTokens += u.completionTokens;
|
|
258
|
+
entry.cachedTokens += u.cachedTokens;
|
|
259
|
+
entry.cacheWriteTokens += u.cacheWriteTokens;
|
|
260
|
+
entry.cost += cost;
|
|
261
|
+
if (estimated) entry.estimated = true;
|
|
262
|
+
|
|
263
|
+
spend.action += cost;
|
|
264
|
+
spend.session += cost;
|
|
265
|
+
if (runOpen) spend.run += cost;
|
|
266
|
+
|
|
267
|
+
log.debug("usage", { source, promptTokens: u.promptTokens, cached: u.cachedTokens, cost, estimated });
|
|
268
|
+
return { ...u, cost, estimated };
|
|
269
|
+
}
|
|
270
|
+
|
|
271
|
+
/**
|
|
272
|
+
* Take everything recorded since the last drain, for the surfaces that display
|
|
273
|
+
* it. The running totals are deliberately left alone.
|
|
274
|
+
*/
|
|
275
|
+
export function drainUsage() {
|
|
276
|
+
const drained = ledger;
|
|
277
|
+
ledger = emptyLedger();
|
|
278
|
+
return drained;
|
|
279
|
+
}
|
|
280
|
+
|
|
281
|
+
/** Sum of a by-source map, for the totals line. */
|
|
282
|
+
export function sumUsage(bySource) {
|
|
283
|
+
const total = emptyEntry();
|
|
284
|
+
for (const entry of Object.values(bySource || {})) {
|
|
285
|
+
if (!entry) continue;
|
|
286
|
+
total.calls += entry.calls || 0;
|
|
287
|
+
total.promptTokens += entry.promptTokens || 0;
|
|
288
|
+
total.completionTokens += entry.completionTokens || 0;
|
|
289
|
+
total.cachedTokens += entry.cachedTokens || 0;
|
|
290
|
+
total.cacheWriteTokens += entry.cacheWriteTokens || 0;
|
|
291
|
+
total.cost += entry.cost || 0;
|
|
292
|
+
if (entry.estimated) total.estimated = true;
|
|
293
|
+
}
|
|
294
|
+
return total;
|
|
295
|
+
}
|
|
296
|
+
|
|
297
|
+
export { emptyEntry };
|