flint-agent 1.14.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (171) hide show
  1. package/.env.example +108 -0
  2. package/CHANGELOG.md +55 -0
  3. package/FEATURES.md +298 -0
  4. package/LICENSE +21 -0
  5. package/README.md +435 -0
  6. package/bin/flint.js +47 -0
  7. package/config/classifier-prompt.md +218 -0
  8. package/config/models-curated.json +4 -0
  9. package/config/providers.json +74 -0
  10. package/package.json +92 -0
  11. package/patches/ink+6.8.0.patch +78 -0
  12. package/profiles/desktop.md +65 -0
  13. package/profiles/generic.md +20 -0
  14. package/profiles/marketer.md +20 -0
  15. package/profiles/profiles.json +34 -0
  16. package/profiles/ux-reviewer.md +25 -0
  17. package/src/agent/agent.js +1743 -0
  18. package/src/agent/auto.js +346 -0
  19. package/src/agent/backoff.js +143 -0
  20. package/src/agent/compression.js +310 -0
  21. package/src/agent/content-resolver.js +180 -0
  22. package/src/agent/flow-controller.js +309 -0
  23. package/src/agent/intent-manifest.js +231 -0
  24. package/src/agent/intent-timeout.js +46 -0
  25. package/src/agent/intent.js +633 -0
  26. package/src/agent/knowledge.js +114 -0
  27. package/src/agent/learning.js +180 -0
  28. package/src/agent/modes.js +187 -0
  29. package/src/agent/outcome-ask.js +91 -0
  30. package/src/agent/project-context.js +76 -0
  31. package/src/agent/prompt-budget.js +117 -0
  32. package/src/agent/reflection-extractor.js +140 -0
  33. package/src/agent/steering.js +86 -0
  34. package/src/agent/supervisor.js +430 -0
  35. package/src/agent/swap.js +443 -0
  36. package/src/agent/system-prompt.js +446 -0
  37. package/src/agent/time-stamp.js +48 -0
  38. package/src/agent/tool-guard.js +201 -0
  39. package/src/agent/toolcall-text.js +162 -0
  40. package/src/agent/usage.js +297 -0
  41. package/src/agent/vision.js +94 -0
  42. package/src/agent/watchdog.js +139 -0
  43. package/src/agent/workspace-changes.js +177 -0
  44. package/src/api/address.js +14 -0
  45. package/src/api/client.js +280 -0
  46. package/src/api/server.js +535 -0
  47. package/src/api/stream-pipe.js +113 -0
  48. package/src/app-state.js +39 -0
  49. package/src/bootstrap.js +501 -0
  50. package/src/bus/drain-loop.js +497 -0
  51. package/src/bus/index.js +270 -0
  52. package/src/bus/plugins.js +65 -0
  53. package/src/child-idle.js +14 -0
  54. package/src/cli.js +118 -0
  55. package/src/commands/commands.js +1297 -0
  56. package/src/commands/registry.js +132 -0
  57. package/src/components/App.js +491 -0
  58. package/src/components/CarefulMenu.js +145 -0
  59. package/src/components/HistoryWriter.js +86 -0
  60. package/src/components/LineInput.js +69 -0
  61. package/src/components/LiveZone.js +294 -0
  62. package/src/components/OverlayMenu.js +179 -0
  63. package/src/components/SystemPanel.js +156 -0
  64. package/src/components/Table.js +54 -0
  65. package/src/config.js +249 -0
  66. package/src/free-models.js +230 -0
  67. package/src/index.js +1111 -0
  68. package/src/input-handler.js +13 -0
  69. package/src/input-text.js +123 -0
  70. package/src/launcher.js +129 -0
  71. package/src/logging/api-log.js +95 -0
  72. package/src/logging/chat-log-follower.js +113 -0
  73. package/src/logging/chat-log.js +15 -0
  74. package/src/logging/log-collector.js +182 -0
  75. package/src/logging/logger.js +112 -0
  76. package/src/logging/tool-log.js +20 -0
  77. package/src/mcp-client.js +314 -0
  78. package/src/memory/conversation-digest.js +113 -0
  79. package/src/memory/extract-facts.js +98 -0
  80. package/src/memory/facts.js +181 -0
  81. package/src/memory/inbox.js +63 -0
  82. package/src/memory/markdown.js +38 -0
  83. package/src/memory/patterns.js +185 -0
  84. package/src/memory/project.js +66 -0
  85. package/src/memory/reflections.js +74 -0
  86. package/src/memory/retrieval.js +84 -0
  87. package/src/memory/rules.js +105 -0
  88. package/src/memory/session-facts.js +125 -0
  89. package/src/memory/skills.js +191 -0
  90. package/src/memory/sqlite-store.js +653 -0
  91. package/src/memory/store.js +208 -0
  92. package/src/memory/tools.js +196 -0
  93. package/src/memory/user-model.js +86 -0
  94. package/src/message-handler.js +775 -0
  95. package/src/model-check.js +218 -0
  96. package/src/plugins/loader.js +120 -0
  97. package/src/plugins/manager.js +88 -0
  98. package/src/production-env.js +22 -0
  99. package/src/profiles.js +42 -0
  100. package/src/providers/adapters/anthropic.js +270 -0
  101. package/src/providers/adapters/openai.js +120 -0
  102. package/src/providers/keys-dpapi.js +41 -0
  103. package/src/providers/keys-fallback.js +31 -0
  104. package/src/providers/keys.js +132 -0
  105. package/src/providers/models.js +154 -0
  106. package/src/providers/registry.js +56 -0
  107. package/src/providers/state.js +56 -0
  108. package/src/registry.js +96 -0
  109. package/src/restart.js +29 -0
  110. package/src/sandbox/backend.js +130 -0
  111. package/src/security/api-auth.js +132 -0
  112. package/src/security/audit.js +98 -0
  113. package/src/security/child-policy.js +41 -0
  114. package/src/security/command-guard.js +173 -0
  115. package/src/security/content-fence.js +250 -0
  116. package/src/security/content-validator.js +132 -0
  117. package/src/security/index.js +143 -0
  118. package/src/security/network-guard.js +126 -0
  119. package/src/security/pairing.js +180 -0
  120. package/src/security/path-guard.js +140 -0
  121. package/src/security/persona-guard.js +67 -0
  122. package/src/security/policies.js +452 -0
  123. package/src/security/safety-constants.js +34 -0
  124. package/src/security/watchdog.js +107 -0
  125. package/src/sessions.js +130 -0
  126. package/src/spend.js +97 -0
  127. package/src/startup-watchdog.js +59 -0
  128. package/src/stdio/args.js +71 -0
  129. package/src/stdio/guard.js +59 -0
  130. package/src/stdio/protocol.js +167 -0
  131. package/src/stdio/run.js +106 -0
  132. package/src/stdio/session.js +180 -0
  133. package/src/store/agent-slice.js +306 -0
  134. package/src/store/dataset-slice.js +73 -0
  135. package/src/store/index.js +22 -0
  136. package/src/store/process-slice.js +135 -0
  137. package/src/store/session-slice.js +191 -0
  138. package/src/store/ui-slice.js +119 -0
  139. package/src/tasks/db.js +184 -0
  140. package/src/tasks/queries.js +589 -0
  141. package/src/tools/agent-tools.js +473 -0
  142. package/src/tools/checkpoint.js +152 -0
  143. package/src/tools/command-approvals.js +180 -0
  144. package/src/tools/dataset.js +50 -0
  145. package/src/tools/filesystem.js +682 -0
  146. package/src/tools/inbox-tools.js +48 -0
  147. package/src/tools/mesh.js +135 -0
  148. package/src/tools/own-env.js +136 -0
  149. package/src/tools/permissions.js +681 -0
  150. package/src/tools/plugin-tools.js +123 -0
  151. package/src/tools/process-tools.js +595 -0
  152. package/src/tools/registry.js +307 -0
  153. package/src/tools/swap-tools.js +72 -0
  154. package/src/tools/system.js +662 -0
  155. package/src/tools/tasks.js +532 -0
  156. package/src/tools/tool-search.js +171 -0
  157. package/src/ui/header.js +140 -0
  158. package/src/ui/input-cursor.js +23 -0
  159. package/src/ui/last-line.js +25 -0
  160. package/src/ui/line-edit.js +135 -0
  161. package/src/ui/output.js +399 -0
  162. package/src/ui/paste-tokens.js +131 -0
  163. package/src/ui/prompt-attention.js +134 -0
  164. package/src/ui/render-options.js +13 -0
  165. package/src/ui/replay.js +94 -0
  166. package/src/ui/splash.js +49 -0
  167. package/src/ui/status-level.js +36 -0
  168. package/src/ui/tool-ledger.js +203 -0
  169. package/src/ui/window-title.js +150 -0
  170. package/src/update.js +205 -0
  171. package/system.md +63 -0
@@ -0,0 +1,681 @@
1
+ // Permissions + Hooks system — wraps executeTool with permission checks and hooks
2
+
3
+ import { existsSync, readFileSync, writeFileSync, appendFileSync, mkdirSync } from "node:fs";
4
+ import path from "node:path";
5
+ import { executeTool, getDefinitions } from "./registry.js";
6
+
7
+ // Tool-name auto-repair at the entry point so permission checks use the corrected name.
8
+ // Levenshtein distance <= 2, unambiguous closest match.
9
+ function _levDistance(a, b) {
10
+ const m = a.length, n = b.length;
11
+ if (a === b) return 0;
12
+ if (!m) return n;
13
+ if (!n) return m;
14
+ const dp = new Array(n + 1);
15
+ for (let j = 0; j <= n; j++) dp[j] = j;
16
+ for (let i = 1; i <= m; i++) {
17
+ let prev = dp[0]; dp[0] = i;
18
+ for (let j = 1; j <= n; j++) {
19
+ const tmp = dp[j];
20
+ dp[j] = a[i - 1] === b[j - 1] ? prev : Math.min(prev, dp[j], dp[j - 1]) + 1;
21
+ prev = tmp;
22
+ }
23
+ }
24
+ return dp[n];
25
+ }
26
+ function _repairToolName(name) {
27
+ try {
28
+ const names = getDefinitions().map(t => t.function?.name).filter(Boolean);
29
+ if (names.includes(name)) return name;
30
+ const scored = names.map(n => ({ n, d: _levDistance(name, n) })).filter(x => x.d <= 2).sort((a, b) => a.d - b.d);
31
+ if (!scored.length) return null;
32
+ if (scored.length === 1 || scored[0].d < scored[1].d) return scored[0].n;
33
+ return null;
34
+ } catch { return null; }
35
+ }
36
+ import { config } from "../config.js";
37
+ import { grantCommandApproval } from "./command-approvals.js";
38
+ import { LEVELS, levelOptions, parseLevelAnswer, SECRET_FILE_PATTERNS, toolPermissionAtLevel, DEFAULT_ONBOARDING_LEVEL } from "../security/policies.js";
39
+
40
+ // ── Security level, from the one onboarding question ──
41
+
42
+ /**
43
+ * Which tools are reads, and so do not prompt.
44
+ *
45
+ * Reading is what the agent is for. A turn that reads twenty files asked twenty
46
+ * questions, and every one of them was "may I read the file you just asked me
47
+ * to read" — an operator who answers yes to that once has learned that yes is
48
+ * the only answer, and a prompt that has stopped being a decision is worse than
49
+ * no prompt at all, because it still costs the time to read it.
50
+ *
51
+ * The exception is in `isSecretFile` below, and it is the whole reason the
52
+ * exception is safe to have.
53
+ */
54
+ export function isReadTool(name) {
55
+ return READ_TOOLS.has(name);
56
+ }
57
+
58
+ const READ_TOOLS = new Set([
59
+ "read_file", "list_directory", "glob", "search_in_files",
60
+ "web_fetch", "web_search", "view_image",
61
+ "think", "check_balance", "check_inbox", "list_models",
62
+ "memory_search", "memory_get",
63
+ // The plan-navigation and MCP-status tools, which are reads that arrived
64
+ // after the list above was written: a tool with no entry in
65
+ // DEFAULT_PERMISSIONS falls through getPermission to "confirm", so
66
+ // `task_stats` — a dashboard query over the operator's own task database —
67
+ // was stopping every turn to ask a question nobody could have an opinion
68
+ // about.
69
+ //
70
+ // Here rather than only in the table below, because these are reads and a
71
+ // read does not prompt whatever the table says. They take no file argument,
72
+ // so the secret carve-out has nothing to apply to.
73
+ //
74
+ // `today` is deliberately NOT here: it reads when called with no arguments
75
+ // and writes (marks tasks for today) when called with task_ids. It is
76
+ // allowed in the table beside its sibling task tools instead, where a
77
+ // name-only set cannot express the difference.
78
+ "task_stats", "list_goals", "focus_goal", "wait_tasks", "list_mcp_servers",
79
+ // A read of the skill/memory store, same as memory_get above it. It took no
80
+ // part in that pair until this commit, and asked on every call.
81
+ "memory_expand",
82
+ ]);
83
+
84
+ /**
85
+ * Is this a file whose contents are not the agent's business without asking?
86
+ *
87
+ * A read exception is only safe because of this. `.env`, an SSH key, a
88
+ * credentials file — these are how an agent walks off with the operator's
89
+ * tokens, and a rule that never asks about them is not a safer rule, it is a
90
+ * rule that has stopped looking.
91
+ *
92
+ * Matched on what the file *is*, not where it lives: a secret is a secret
93
+ * wherever it is kept, and anchoring on the project root would exempt a copy in
94
+ * a subdirectory, which is the copy somebody actually pasted into a bug report.
95
+ */
96
+ export function isSecretFile(filePath) {
97
+ if (!filePath) return false;
98
+ const p = String(filePath);
99
+ return SECRET_FILE_PATTERNS.some((re) => re.test(p));
100
+ }
101
+
102
+ const ONBOARDING_KEY = "onboardingAnswer";
103
+
104
+ /**
105
+ * Has the operator been asked, and what did they answer?
106
+ *
107
+ * Two pieces of state rather than one, because "answered" and "answered
108
+ * something" are different. If a cancelled question counted as an answer, Esc
109
+ * on the first launch would skip the question forever and leave the user at a
110
+ * posture they never chose and cannot see.
111
+ */
112
+ export function getOnboardingState() {
113
+ return { asked: Object.prototype.hasOwnProperty.call(sessionOverrides, ONBOARDING_KEY), answer: getOnboardingAnswer() };
114
+ }
115
+
116
+ /**
117
+ * Ask the one onboarding question, if it has not been answered yet.
118
+ *
119
+ * Without this the question in the spec does not exist: `saveOnboardingAnswer`
120
+ * had no caller outside tests, so a fresh Flint home was never asked, and
121
+ * nothing was there to skip on the second start either.
122
+ *
123
+ * @param {object} opts
124
+ * @param {(question: string) => Promise<string|null>} opts.confirm — the prompt
125
+ * @returns {Promise<boolean>} true if the question was asked now
126
+ */
127
+ export async function askOnboardingIfNeeded({ confirm } = {}) {
128
+ if (getOnboardingState().asked) return false;
129
+ if (typeof confirm !== "function") {
130
+ throw new Error("askOnboardingIfNeeded needs a confirm function; there is nothing to ask with");
131
+ }
132
+ // The prompt and the accepted answers are built from one list, so they cannot
133
+ // disagree. They did: the prompt said "pick s, n or p" and only "safe"/"normal"/
134
+ // "permissive" were recorded, so an operator who typed `n` — exactly as told —
135
+ // was silently not recorded and the question came back on every start, which is
136
+ // the opposite of "asked once". Every test before that fed full words,
137
+ // so the suite agreed with itself while the operator got nothing.
138
+ const question = levelOptions()
139
+ .map((o) => `[${o.key}]${o.level.slice(1)} (${o.description})`)
140
+ .join(" ");
141
+ const answer = await confirm(
142
+ `How careful should Flint be?\n${question}\n — pick ${levelOptions().map((o) => o.key).join(", ")}`,
143
+ );
144
+ const level = parseLevelAnswer(answer);
145
+ // A cancelled or unreadable answer is not an answer. Recording it would
146
+ // permanently skip the question, which is how a user ends up at a posture
147
+ // nobody chose.
148
+ if (!level) return true;
149
+ saveOnboardingAnswer(level);
150
+ return true;
151
+ }
152
+
153
+ /**
154
+ * The level the operator chose, or null if nobody has been asked yet.
155
+ *
156
+ * @returns {"safe"|"normal"|"permissive"|null}
157
+ */
158
+ export function getOnboardingAnswer() {
159
+ return sessionOverrides[ONBOARDING_KEY] ?? null;
160
+ }
161
+
162
+ /**
163
+ * Record the answer to the one onboarding question, and persist it.
164
+ *
165
+ * Persisted, because a question that comes back on every launch is a question
166
+ * nobody learns to answer well. A value that is not one of the three levels is
167
+ * refused rather than stored: a saved answer nobody chose is worse than no
168
+ * answer, because it looks like a choice was made.
169
+ */
170
+ export function saveOnboardingAnswer(level) {
171
+ if (!LEVELS.includes(level)) {
172
+ throw new Error(
173
+ `Unknown security level: ${level}. Expected one of: ${LEVELS.join(", ")}`,
174
+ );
175
+ }
176
+ sessionOverrides[ONBOARDING_KEY] = level;
177
+ saveOverridesToDisk();
178
+ }
179
+
180
+ /**
181
+ * Clear everything a test set up, so one test's permissions cannot decide
182
+ * another's. Not exported for the app — only for tests.
183
+ */
184
+ export function resetPermissionState() {
185
+ for (const k of Object.keys(sessionOverrides)) delete sessionOverrides[k];
186
+ Object.assign(sessionOverrides, _loaded.levels);
187
+ for (const key of Object.keys(approvedPaths)) delete approvedPaths[key];
188
+ _globalPermission = null;
189
+ _unattended = false;
190
+ }
191
+
192
+
193
+ // ── Default permission levels ──
194
+
195
+ const DEFAULT_PERMISSIONS = {
196
+ // allow — execute without confirmation
197
+ read_file: "allow",
198
+ list_directory: "allow",
199
+ glob: "allow",
200
+ search_in_files: "allow",
201
+ view_image: "allow",
202
+ think: "allow",
203
+ check_balance: "allow",
204
+ check_inbox: "allow",
205
+ list_models: "allow",
206
+ add_task: "allow",
207
+ list_processes: "allow",
208
+ peek_process: "allow",
209
+ web_fetch: "allow",
210
+ web_search: "allow",
211
+ memory_write: "allow",
212
+ memory_search: "allow",
213
+ memory_get: "allow",
214
+ memory_delete: "allow",
215
+
216
+ // Skills — the agent's own persistent procedures, stored as markdown under
217
+ // ~/.flint/memory/skills/. Same class as the memory tools above and given
218
+ // the same answer; they had no entry at all, so skill_add — a tool the
219
+ // agent calls whenever it learns a repeatable procedure — stopped the turn
220
+ // to ask the operator to approve remembering something.
221
+ skill_add: "allow",
222
+ skill_update: "allow",
223
+ skill_remove: "allow",
224
+ memory_expand: "allow",
225
+
226
+ list_agents: "allow",
227
+
228
+ // screenbox — screenshot is safe, interactive tools need confirmation
229
+ desktop_screenshot: "allow",
230
+ desktop_look: "allow",
231
+ desktop_resume: "allow",
232
+ desktop_click: "confirm",
233
+ desktop_type: "confirm",
234
+ desktop_key: "confirm",
235
+ desktop_scroll: "confirm",
236
+ desktop_chrome: "confirm",
237
+
238
+ // mesh memory — all safe
239
+ mesh_search: "allow",
240
+ mesh_add: "allow",
241
+ mesh_recent: "allow",
242
+
243
+ // Google Workspace (MCP) — read=allow, write=confirm
244
+ search_gmail_messages: "allow",
245
+ get_gmail_message_content: "allow",
246
+ get_gmail_messages_content_batch: "allow",
247
+ get_gmail_thread_content: "allow",
248
+ get_gmail_threads_content_batch: "allow",
249
+ list_gmail_labels: "allow",
250
+ send_gmail_message: "confirm",
251
+ draft_gmail_message: "confirm",
252
+ modify_gmail_message_labels: "confirm",
253
+ batch_modify_gmail_message_labels: "confirm",
254
+ manage_gmail_label: "confirm",
255
+
256
+ // datasets — safe (read-only navigation)
257
+ show_dataset: "allow",
258
+
259
+ // task planning — all safe
260
+ create_plan: "allow",
261
+ update_task: "allow",
262
+ list_tasks: "allow",
263
+ add_task_note: "allow",
264
+ link_task_file: "allow",
265
+
266
+ // Plan navigation. These arrived after the list above and had no entry, so
267
+ // getPermission fell through to "confirm" and the agent stopped every turn
268
+ // to ask about reading the operator's own task database. See READ_TOOLS for
269
+ // why `today` is in the table but not in the set.
270
+ task_stats: "allow",
271
+ list_goals: "allow",
272
+ focus_goal: "allow",
273
+ wait_tasks: "allow",
274
+ today: "allow",
275
+ create_subtask: "allow",
276
+
277
+ // MCP plumbing. reconnect_mcp asks, deliberately: it rebuilds a session on a
278
+ // server that is the operator's own machine, and a tool that has been failing
279
+ // with "fetch failed" is a tool whose name has just come out of an error
280
+ // message. It stays a question.
281
+ list_mcp_servers: "allow",
282
+ // Loads tool definitions into the turn; runs nothing itself.
283
+ tool_search: "allow",
284
+ // Read the session's own swap; nothing leaves the machine.
285
+ swap_list: "allow",
286
+ swap_read: "allow",
287
+ reconnect_mcp: "confirm",
288
+
289
+
290
+ // provider switching — safe (doesn't cost money or change data)
291
+ switch_model: "allow",
292
+ switch_provider: "allow",
293
+ list_providers: "allow",
294
+
295
+ // confirm — ask user before executing
296
+ spawn_agent: "confirm",
297
+ ask_agent: "confirm",
298
+ write_file: "confirm",
299
+ edit_file: "confirm",
300
+ delete_file: "confirm",
301
+ create_directory: "confirm",
302
+ copy_file: "confirm",
303
+ move_file: "confirm",
304
+ run_command: "confirm",
305
+ run_background_command: "confirm",
306
+ kill_process: "confirm",
307
+ restart_agent: "confirm",
308
+ clear_context: "confirm",
309
+ // A plugin is code that runs with the agent's rights.
310
+ install_plugin: "confirm",
311
+ reload_plugins: "confirm",
312
+ };
313
+
314
+ // ── Persistence ──
315
+
316
+ // config.permissionsFile, not projectRoot + the name. Under a test run the
317
+ // whole point is that this is somewhere else, and rebuilding the path from
318
+ // projectRoot here would put it back in the developer's checkout. Reads
319
+ // defensively: several integration tests replace the config object wholesale
320
+ // with only the keys their subject reads, so a missing key must fall back to
321
+ // the historical path rather than be undefined.
322
+ const PERMISSIONS_FILE = config.permissionsFile
323
+ || path.join(config.projectRoot || process.cwd(), ".permissions.json");
324
+
325
+ // Reserved key inside .permissions.json. Everything else in that file is a
326
+ // tool name -> level; this one holds per-file approvals, "<tool>:<abs path>".
327
+ // One file, because the operator should have one place to look at what they
328
+ // have granted.
329
+ const APPROVED_PATHS_KEY = "_approvedPaths";
330
+
331
+ function loadPermissionsFile() {
332
+ try {
333
+ if (existsSync(PERMISSIONS_FILE)) {
334
+ const raw = JSON.parse(readFileSync(PERMISSIONS_FILE, "utf-8"));
335
+ const { [APPROVED_PATHS_KEY]: paths, ...levels } = raw;
336
+ return { levels, paths: paths && typeof paths === "object" ? paths : {} };
337
+ }
338
+ } catch {}
339
+ return { levels: {}, paths: {} };
340
+ }
341
+
342
+ function saveOverridesToDisk() {
343
+ try {
344
+ const out = { ...sessionOverrides };
345
+ if (Object.keys(approvedPaths).length) out[APPROVED_PATHS_KEY] = approvedPaths;
346
+ writeFileSync(PERMISSIONS_FILE, JSON.stringify(out, null, 2) + "\n");
347
+ } catch {}
348
+ }
349
+
350
+ // Sub-second windows exist only in tests, but rounding them to "0s" makes the
351
+ // refusal read like a bug in the tool rather than an unanswered prompt.
352
+ function formatWindow(ms) {
353
+ return ms < 1000 ? `${ms}ms` : `${Math.round(ms / 1000)}s`;
354
+ }
355
+
356
+ // ── Security log ──
357
+
358
+ function logSecurity(action, toolName, args, reason) {
359
+ try {
360
+ const dir = config.sessionsDir;
361
+ if (!existsSync(dir)) mkdirSync(dir, { recursive: true });
362
+ const file = path.join(dir, "security.log");
363
+ const ts = new Date().toISOString();
364
+ const argsStr = args && typeof args === "object"
365
+ ? Object.entries(args).map(([k, v]) => {
366
+ const s = typeof v === "string" && v.length > 100 ? v.slice(0, 100) + "..." : String(v);
367
+ return `${k}=${s}`;
368
+ }).join(" ")
369
+ : "";
370
+ // pid, because several instances share this file: the interactive one the
371
+ // owner keeps open, plus whatever a benchmark or a test stand starts. Without
372
+ // it a denial cannot be attributed to a run, and on 2026-09-20 that led to a
373
+ // readiness timeout being blamed on the wrong instance.
374
+ appendFileSync(file, `[${ts}] [pid ${process.pid}] ${action} ${toolName}(${argsStr})${reason ? " | " + reason : ""}\n`);
375
+ } catch {}
376
+ }
377
+
378
+ // ── State ──
379
+
380
+ const _loaded = loadPermissionsFile();
381
+ const sessionOverrides = _loaded.levels;
382
+ // "<tool>:<resolved path>" -> true. A hook that forces confirmation can offer
383
+ // a key; answering "always" records THAT key rather than opening the tool up
384
+ // everywhere. Without this there was nowhere to put the operator's decision,
385
+ // so it was written as a tool-wide "allow" that the forced-confirm branch
386
+ // never consulted, and the same prompt came back forever.
387
+ const approvedPaths = _loaded.paths;
388
+ const beforeHooks = []; // (name, args) → { allow } | { deny, reason } | { confirm, reason?, key? } | null
389
+ const afterHooks = []; // (name, args, result) → transformedResult | null
390
+ let confirmFn = null; // injected via initPermissions
391
+ // Auto-deny window when the user does not respond to an approval prompt.
392
+ // Was 30s and caused a retry-storm: operator glances at another terminal
393
+ // for half a minute, the prompt times out, the agent retries, more prompts
394
+ // pile up, eventually everything fails. 600s gives a realistic "human is
395
+ // not at keyboard" window before giving up. Paired with a system.md rule
396
+ // that forbids retrying after a denial/timeout.
397
+ let confirmTimeoutMs = 600000; // 10 minutes
398
+ let _globalPermission = null; // set by bulkSetPermission — overrides everything
399
+ // True while the turn came in over the bus (api/agent/autonomous) and there is
400
+ // nobody at the keyboard. Waiting out confirmTimeoutMs then would buy nothing:
401
+ // the answer can only ever be "timeout", and the whole time the bus is serial,
402
+ // so every queued message sits behind the wait. Raising the window from 30s to
403
+ // 600s for the operator's sake made that ten times worse for unattended runs.
404
+ let _unattended = false;
405
+
406
+ // ── API ──
407
+
408
+ export function initPermissions({ confirm, timeout }) {
409
+ confirmFn = confirm;
410
+ if (timeout != null) confirmTimeoutMs = timeout;
411
+ }
412
+
413
+ export function getPermission(name) {
414
+ if (_globalPermission) return _globalPermission;
415
+ if (sessionOverrides[name]) return sessionOverrides[name];
416
+ // The chosen care level relaxes a default of "confirm" (see policies.js).
417
+ const level = getOnboardingAnswer() || DEFAULT_ONBOARDING_LEVEL;
418
+ return toolPermissionAtLevel(level, name, DEFAULT_PERMISSIONS[name] || "confirm");
419
+ }
420
+
421
+ export function setPermission(name, level) {
422
+ sessionOverrides[name] = level;
423
+ saveOverridesToDisk();
424
+ }
425
+
426
+ export function getPermissionMap() {
427
+ const map = {};
428
+ const allNames = new Set([
429
+ ...Object.keys(DEFAULT_PERMISSIONS),
430
+ ...Object.keys(sessionOverrides),
431
+ ]);
432
+ for (const name of allNames) {
433
+ map[name] = getPermission(name);
434
+ }
435
+ return map;
436
+ }
437
+
438
+ export function addBeforeHook(fn) {
439
+ beforeHooks.push(fn);
440
+ }
441
+
442
+ export function addAfterHook(fn) {
443
+ afterHooks.push(fn);
444
+ }
445
+
446
+ export function resetSessionOverrides() {
447
+ // The care level lives in the same map but is not a tool override: resetting
448
+ // permissions used to erase it, and the onboarding question came back.
449
+ for (const key of Object.keys(sessionOverrides)) {
450
+ if (key === ONBOARDING_KEY) continue;
451
+ delete sessionOverrides[key];
452
+ }
453
+ for (const key of Object.keys(approvedPaths)) {
454
+ delete approvedPaths[key];
455
+ }
456
+ saveOverridesToDisk();
457
+ }
458
+
459
+ /** Per-file approvals, for /permissions and for tests. */
460
+ export function getApprovedPaths() {
461
+ return { ...approvedPaths };
462
+ }
463
+
464
+ export function revokeApprovedPath(key) {
465
+ if (!(key in approvedPaths)) return false;
466
+ delete approvedPaths[key];
467
+ saveOverridesToDisk();
468
+ return true;
469
+ }
470
+
471
+ /** The window the operator actually gets, so the UI can stop guessing. */
472
+ export function getConfirmTimeoutMs() {
473
+ return confirmTimeoutMs;
474
+ }
475
+
476
+ /** Mark the current turn as having no operator behind it. Set by the drain loop. */
477
+ export function setUnattended(v) {
478
+ _unattended = !!v;
479
+ }
480
+
481
+ export function isUnattended() {
482
+ return _unattended;
483
+ }
484
+
485
+ export function bulkSetPermission(level) {
486
+ // SEC-02: YOLO mode is session-only — never persist to disk.
487
+ // A prompt-injected agent must not be able to self-escalate permanently.
488
+ _globalPermission = level;
489
+ }
490
+
491
+ const PLUGIN_LOAD_TOOLS = new Set(["install_plugin", "reload_plugins"]);
492
+
493
+ // ── Main wrapper ──
494
+
495
+ export async function executeToolWithPermissions(name, args) {
496
+ // 0. Auto-repair: if tool name is slightly off (typo, variant), map to closest real name.
497
+ const repaired = _repairToolName(name);
498
+ if (repaired && repaired !== name) {
499
+ console.error(`[tool-repair] '${name}' -> '${repaired}'`);
500
+ name = repaired;
501
+ }
502
+ // 1. Run before hooks — first non-null verdict wins
503
+ let hookAllowed = false;
504
+ let forceConfirm = false;
505
+ let confirmReason = null; // why this particular call needs an answer
506
+ let confirmKey = null; // what "always" would remember, if anything
507
+ // A hook already spoke for this call. Its key, its reason and its answer are
508
+ // better informed than anything the read fallback below can produce, and its
509
+ // "already approved" verdict deliberately sets no forceConfirm — so without
510
+ // this flag the fallback would ask a second time about a call a hook had
511
+ // already settled, which is how granting "[a]lways" appeared to do nothing.
512
+ let hookDecided = false;
513
+ for (const hook of beforeHooks) {
514
+ const verdict = await hook(name, args);
515
+ if (verdict) {
516
+ if (verdict.deny) {
517
+ logSecurity("DENIED_HOOK", name, args, verdict.reason);
518
+ return { result: `Denied by hook: ${verdict.reason || "no reason"}`, denied: true, denyKey: verdict.denyKey || null };
519
+ }
520
+ if (verdict.allow) { hookAllowed = true; break; }
521
+ if (verdict.confirm) {
522
+ // An answer already given for this exact file is an answer. Asking
523
+ // again is how you train an operator to stop reading the prompt.
524
+ if (verdict.key && approvedPaths[verdict.key]) {
525
+ logSecurity("ALLOWED_PATH", name, args, `previously approved: ${verdict.key}`);
526
+ hookDecided = true;
527
+ break;
528
+ }
529
+ forceConfirm = true; // force confirm even when permission is "allow"
530
+ confirmReason = verdict.reason || null;
531
+ confirmKey = verdict.key || null;
532
+ hookDecided = true;
533
+ break;
534
+ }
535
+ }
536
+ }
537
+
538
+ // 1b. Installing or loading a plugin runs code nobody reviewed with the
539
+ // agent's rights. Unless config says pluginInstall "allow", it is asked
540
+ // every time, over a hook's allow and over the API's auto-approve alike,
541
+ // and a run with no operator is refused. "Always" is remembered under its
542
+ // own key, or the prompt would come back forever.
543
+ if (PLUGIN_LOAD_TOOLS.has(name) && config.pluginInstall !== "allow") {
544
+ const key = `${name}:plugin`;
545
+ if (!approvedPaths[key]) {
546
+ forceConfirm = true;
547
+ confirmReason = "a plugin is code that runs with the agent's rights (pluginInstall is \"ask\"; FLINT_PLUGIN_INSTALL=allow skips this)";
548
+ confirmKey = key;
549
+ }
550
+ }
551
+
552
+ // 2. Check permission level (skip if hook already allowed)
553
+ //
554
+ // A read is not a question, with one exception.
555
+ //
556
+ // Reading is what the agent is for. A turn that reads twenty files asked
557
+ // twenty questions, every one of them "may I read the file you just asked me
558
+ // to read" — and an operator who answers yes to that once has learned that
559
+ // yes is the only answer. A prompt that has stopped being a decision costs
560
+ // the time to read it and buys nothing.
561
+ //
562
+ // The exception is what makes the rule safe rather than reckless: .env, an
563
+ // SSH key, a credentials file still ask, at every level including
564
+ // permissive, because that is how an agent walks off with the operator's
565
+ // tokens. A hook that already allowed this call wins — an explicit answer
566
+ // beats a default.
567
+ let level = hookAllowed ? "allow" : getPermission(name);
568
+ if (!hookAllowed && isReadTool(name) && level !== "deny") {
569
+ const fileish = args?.path || args?.file_path || args?.file;
570
+ const secret = fileish ? isSecretFile(fileish) : false;
571
+ if (!secret) {
572
+ logSecurity("ALLOWED_READ", name, args, "reads do not prompt");
573
+ level = "allow";
574
+ } else if (!forceConfirm && !hookDecided) {
575
+ // Named in the prompt, because "may I read this file" and "may I read
576
+ // .env" are different questions and the operator is entitled to know
577
+ // which one they are answering.
578
+ //
579
+ // Guarded on `!forceConfirm` because a hook may already have forced a
580
+ // confirm for this very call, with a better key and a better reason than
581
+ // anything written here. Forcing a second one asked the operator the
582
+ // same question twice, and the `[a]lways` answer was then stored against
583
+ // the hook's key while this branch went on asking under a different one —
584
+ // so granting "always" appeared to do nothing.
585
+ //
586
+ // This branch is the fallback for when nothing upstream cared, which is
587
+ // the ordinary case: no hook, and the agent is about to read a private
588
+ // key on its own recognisance.
589
+ forceConfirm = true;
590
+ confirmReason = "this looks like a secret file (.env, a private key, credentials)";
591
+ confirmKey = `${name}:secret:${fileish || ""}`;
592
+ }
593
+ }
594
+
595
+ if (level === "deny") {
596
+ logSecurity("DENIED_RULE", name, args, "tool denied by rule");
597
+ return { result: `Tool "${name}" is denied.`, denied: true };
598
+ }
599
+
600
+ if ((level === "confirm" || forceConfirm) && confirmFn) {
601
+ // What "always" will mean, in the operator's words, so the prompt does
602
+ // not promise something wider than it grants.
603
+ const scope = confirmKey
604
+ ? `always allow ${name} for this path only`
605
+ : `always allow ${name}`;
606
+
607
+ // No operator, no point asking. Deny now instead of holding the bus for the
608
+ // full window on a question that cannot be answered.
609
+ if (_unattended) {
610
+ logSecurity("DENIED_UNATTENDED", name, args, confirmReason || "approval required, no operator attached");
611
+ return {
612
+ result: `Tool "${name}" was not run: it needs the operator's approval and this run has no operator attached.` +
613
+ (confirmReason ? ` Approval was required because ${confirmReason}.` : "") +
614
+ ` Do not retry this call and do not look for another tool that does the same thing.` +
615
+ ` Finish the turn and say plainly what you need approved.`,
616
+ denied: true,
617
+ };
618
+ }
619
+
620
+ // Wrap confirm with timeout
621
+ const answer = await Promise.race([
622
+ confirmFn(name, args, { reason: confirmReason, scope, key: confirmKey }),
623
+ new Promise((resolve) => setTimeout(() => resolve("timeout"), confirmTimeoutMs)),
624
+ ]);
625
+
626
+ if (answer === "always") {
627
+ if (confirmKey) {
628
+ // Narrow on purpose: "always read this .env" is a decision a person
629
+ // can mean; "always read every secret file" is not.
630
+ approvedPaths[confirmKey] = true;
631
+ } else if ((name === "run_command" || name === "run_background_command") && args?.command) {
632
+ // A command prompt forced by the guard arrives with no key of its own,
633
+ // so this branch used to fall through to `sessionOverrides[name] =
634
+ // "allow"` — a tool-wide grant, in every project, for the rest of the
635
+ // install, out of an answer about one command in one checkout. The
636
+ // command guard asks anyway on its own patterns, so the operator was
637
+ // prompted again on the next `git push` despite having answered
638
+ // "always" — and the grant sat in the file widening the tool the whole
639
+ // time. Recorded per project instead, which is the question that was
640
+ // actually asked, and the guard reads it.
641
+ grantCommandApproval({ cwd: args.cwd, command: args.command });
642
+ } else {
643
+ sessionOverrides[name] = "allow";
644
+ }
645
+ saveOverridesToDisk();
646
+ } else if (answer === "timeout") {
647
+ logSecurity("DENIED_TIMEOUT", name, args, `no response within ${confirmTimeoutMs / 1000}s`);
648
+ // The model has to be told WHY, or it invents a reason and burns the
649
+ // turn working around the wrong one: on 2026-09-19 it read a bare
650
+ // "(timeout)" as "the file is too large" and spent ~15 of 26
651
+ // iterations on subshell exports and agent restarts.
652
+ return {
653
+ result: `Tool "${name}" was not run: the operator did not answer the approval prompt within ${formatWindow(confirmTimeoutMs)}.` +
654
+ (confirmReason ? ` Approval was required because ${confirmReason}.` : "") +
655
+ ` Do not retry this call; ask the operator in plain words instead.`,
656
+ denied: true,
657
+ };
658
+ } else if (answer !== "yes") {
659
+ logSecurity("DENIED_USER", name, args, "user denied");
660
+ return {
661
+ result: `Tool "${name}" was denied by the operator.` +
662
+ (confirmReason ? ` Approval was required because ${confirmReason}.` : "") +
663
+ ` Do not retry this call or route around the denial; say what you needed and why.`,
664
+ denied: true,
665
+ };
666
+ }
667
+ }
668
+
669
+ // 3. Execute tool
670
+ let result = await executeTool(name, args);
671
+
672
+ // 4. Run after hooks
673
+ for (const hook of afterHooks) {
674
+ const transformed = await hook(name, args, result);
675
+ if (transformed != null) {
676
+ result = transformed;
677
+ }
678
+ }
679
+
680
+ return { result, denied: false };
681
+ }