flint-agent 1.14.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (171) hide show
  1. package/.env.example +108 -0
  2. package/CHANGELOG.md +55 -0
  3. package/FEATURES.md +298 -0
  4. package/LICENSE +21 -0
  5. package/README.md +435 -0
  6. package/bin/flint.js +47 -0
  7. package/config/classifier-prompt.md +218 -0
  8. package/config/models-curated.json +4 -0
  9. package/config/providers.json +74 -0
  10. package/package.json +92 -0
  11. package/patches/ink+6.8.0.patch +78 -0
  12. package/profiles/desktop.md +65 -0
  13. package/profiles/generic.md +20 -0
  14. package/profiles/marketer.md +20 -0
  15. package/profiles/profiles.json +34 -0
  16. package/profiles/ux-reviewer.md +25 -0
  17. package/src/agent/agent.js +1743 -0
  18. package/src/agent/auto.js +346 -0
  19. package/src/agent/backoff.js +143 -0
  20. package/src/agent/compression.js +310 -0
  21. package/src/agent/content-resolver.js +180 -0
  22. package/src/agent/flow-controller.js +309 -0
  23. package/src/agent/intent-manifest.js +231 -0
  24. package/src/agent/intent-timeout.js +46 -0
  25. package/src/agent/intent.js +633 -0
  26. package/src/agent/knowledge.js +114 -0
  27. package/src/agent/learning.js +180 -0
  28. package/src/agent/modes.js +187 -0
  29. package/src/agent/outcome-ask.js +91 -0
  30. package/src/agent/project-context.js +76 -0
  31. package/src/agent/prompt-budget.js +117 -0
  32. package/src/agent/reflection-extractor.js +140 -0
  33. package/src/agent/steering.js +86 -0
  34. package/src/agent/supervisor.js +430 -0
  35. package/src/agent/swap.js +443 -0
  36. package/src/agent/system-prompt.js +446 -0
  37. package/src/agent/time-stamp.js +48 -0
  38. package/src/agent/tool-guard.js +201 -0
  39. package/src/agent/toolcall-text.js +162 -0
  40. package/src/agent/usage.js +297 -0
  41. package/src/agent/vision.js +94 -0
  42. package/src/agent/watchdog.js +139 -0
  43. package/src/agent/workspace-changes.js +177 -0
  44. package/src/api/address.js +14 -0
  45. package/src/api/client.js +280 -0
  46. package/src/api/server.js +535 -0
  47. package/src/api/stream-pipe.js +113 -0
  48. package/src/app-state.js +39 -0
  49. package/src/bootstrap.js +501 -0
  50. package/src/bus/drain-loop.js +497 -0
  51. package/src/bus/index.js +270 -0
  52. package/src/bus/plugins.js +65 -0
  53. package/src/child-idle.js +14 -0
  54. package/src/cli.js +118 -0
  55. package/src/commands/commands.js +1297 -0
  56. package/src/commands/registry.js +132 -0
  57. package/src/components/App.js +491 -0
  58. package/src/components/CarefulMenu.js +145 -0
  59. package/src/components/HistoryWriter.js +86 -0
  60. package/src/components/LineInput.js +69 -0
  61. package/src/components/LiveZone.js +294 -0
  62. package/src/components/OverlayMenu.js +179 -0
  63. package/src/components/SystemPanel.js +156 -0
  64. package/src/components/Table.js +54 -0
  65. package/src/config.js +249 -0
  66. package/src/free-models.js +230 -0
  67. package/src/index.js +1111 -0
  68. package/src/input-handler.js +13 -0
  69. package/src/input-text.js +123 -0
  70. package/src/launcher.js +129 -0
  71. package/src/logging/api-log.js +95 -0
  72. package/src/logging/chat-log-follower.js +113 -0
  73. package/src/logging/chat-log.js +15 -0
  74. package/src/logging/log-collector.js +182 -0
  75. package/src/logging/logger.js +112 -0
  76. package/src/logging/tool-log.js +20 -0
  77. package/src/mcp-client.js +314 -0
  78. package/src/memory/conversation-digest.js +113 -0
  79. package/src/memory/extract-facts.js +98 -0
  80. package/src/memory/facts.js +181 -0
  81. package/src/memory/inbox.js +63 -0
  82. package/src/memory/markdown.js +38 -0
  83. package/src/memory/patterns.js +185 -0
  84. package/src/memory/project.js +66 -0
  85. package/src/memory/reflections.js +74 -0
  86. package/src/memory/retrieval.js +84 -0
  87. package/src/memory/rules.js +105 -0
  88. package/src/memory/session-facts.js +125 -0
  89. package/src/memory/skills.js +191 -0
  90. package/src/memory/sqlite-store.js +653 -0
  91. package/src/memory/store.js +208 -0
  92. package/src/memory/tools.js +196 -0
  93. package/src/memory/user-model.js +86 -0
  94. package/src/message-handler.js +775 -0
  95. package/src/model-check.js +218 -0
  96. package/src/plugins/loader.js +120 -0
  97. package/src/plugins/manager.js +88 -0
  98. package/src/production-env.js +22 -0
  99. package/src/profiles.js +42 -0
  100. package/src/providers/adapters/anthropic.js +270 -0
  101. package/src/providers/adapters/openai.js +120 -0
  102. package/src/providers/keys-dpapi.js +41 -0
  103. package/src/providers/keys-fallback.js +31 -0
  104. package/src/providers/keys.js +132 -0
  105. package/src/providers/models.js +154 -0
  106. package/src/providers/registry.js +56 -0
  107. package/src/providers/state.js +56 -0
  108. package/src/registry.js +96 -0
  109. package/src/restart.js +29 -0
  110. package/src/sandbox/backend.js +130 -0
  111. package/src/security/api-auth.js +132 -0
  112. package/src/security/audit.js +98 -0
  113. package/src/security/child-policy.js +41 -0
  114. package/src/security/command-guard.js +173 -0
  115. package/src/security/content-fence.js +250 -0
  116. package/src/security/content-validator.js +132 -0
  117. package/src/security/index.js +143 -0
  118. package/src/security/network-guard.js +126 -0
  119. package/src/security/pairing.js +180 -0
  120. package/src/security/path-guard.js +140 -0
  121. package/src/security/persona-guard.js +67 -0
  122. package/src/security/policies.js +452 -0
  123. package/src/security/safety-constants.js +34 -0
  124. package/src/security/watchdog.js +107 -0
  125. package/src/sessions.js +130 -0
  126. package/src/spend.js +97 -0
  127. package/src/startup-watchdog.js +59 -0
  128. package/src/stdio/args.js +71 -0
  129. package/src/stdio/guard.js +59 -0
  130. package/src/stdio/protocol.js +167 -0
  131. package/src/stdio/run.js +106 -0
  132. package/src/stdio/session.js +180 -0
  133. package/src/store/agent-slice.js +306 -0
  134. package/src/store/dataset-slice.js +73 -0
  135. package/src/store/index.js +22 -0
  136. package/src/store/process-slice.js +135 -0
  137. package/src/store/session-slice.js +191 -0
  138. package/src/store/ui-slice.js +119 -0
  139. package/src/tasks/db.js +184 -0
  140. package/src/tasks/queries.js +589 -0
  141. package/src/tools/agent-tools.js +473 -0
  142. package/src/tools/checkpoint.js +152 -0
  143. package/src/tools/command-approvals.js +180 -0
  144. package/src/tools/dataset.js +50 -0
  145. package/src/tools/filesystem.js +682 -0
  146. package/src/tools/inbox-tools.js +48 -0
  147. package/src/tools/mesh.js +135 -0
  148. package/src/tools/own-env.js +136 -0
  149. package/src/tools/permissions.js +681 -0
  150. package/src/tools/plugin-tools.js +123 -0
  151. package/src/tools/process-tools.js +595 -0
  152. package/src/tools/registry.js +307 -0
  153. package/src/tools/swap-tools.js +72 -0
  154. package/src/tools/system.js +662 -0
  155. package/src/tools/tasks.js +532 -0
  156. package/src/tools/tool-search.js +171 -0
  157. package/src/ui/header.js +140 -0
  158. package/src/ui/input-cursor.js +23 -0
  159. package/src/ui/last-line.js +25 -0
  160. package/src/ui/line-edit.js +135 -0
  161. package/src/ui/output.js +399 -0
  162. package/src/ui/paste-tokens.js +131 -0
  163. package/src/ui/prompt-attention.js +134 -0
  164. package/src/ui/render-options.js +13 -0
  165. package/src/ui/replay.js +94 -0
  166. package/src/ui/splash.js +49 -0
  167. package/src/ui/status-level.js +36 -0
  168. package/src/ui/tool-ledger.js +203 -0
  169. package/src/ui/window-title.js +150 -0
  170. package/src/update.js +205 -0
  171. package/system.md +63 -0
@@ -0,0 +1,114 @@
1
+ /**
2
+ * Knowledge Store — backend-agnostic facts & patterns
3
+ *
4
+ * Stores learned facts and patterns. Retrieves by keyword match.
5
+ * Backend adapters: text file (default), mesh, agent-memory.
6
+ *
7
+ * Used by:
8
+ * - Layer 3 (Execution): retrieve before planning
9
+ * - Layer 4 (Verification Gate): store after successful task
10
+ */
11
+
12
+ import { readFileSync, writeFileSync, existsSync, mkdirSync } from "node:fs";
13
+ import { join } from "node:path";
14
+ import { createLogger } from "../logging/logger.js";
15
+ import { config } from "../config.js";
16
+
17
+ const log = createLogger("knowledge");
18
+
19
+ // --- Storage path ---
20
+ const KNOWLEDGE_DIR = join(config.sessionsDir || process.cwd(), "..", "knowledge");
21
+ const FACTS_FILE = join(KNOWLEDGE_DIR, "facts.jsonl");
22
+
23
+ // --- In-memory cache ---
24
+ let _facts = [];
25
+ let _loaded = false;
26
+
27
+ /**
28
+ * Load facts from disk (lazy, once).
29
+ */
30
+ function _ensureLoaded() {
31
+ if (_loaded) return;
32
+ _loaded = true;
33
+ try {
34
+ if (existsSync(FACTS_FILE)) {
35
+ const lines = readFileSync(FACTS_FILE, "utf-8").split("\n").filter(Boolean);
36
+ _facts = lines.map(l => { try { return JSON.parse(l); } catch { return null; } }).filter(Boolean);
37
+ log.info("loaded", { count: _facts.length });
38
+ }
39
+ } catch (e) {
40
+ log.warn("load-failed", { error: e.message });
41
+ }
42
+ }
43
+
44
+ /**
45
+ * Store a fact or pattern.
46
+ * @param {object} entry - { type: "fact"|"pattern", text, context, confidence, triggers }
47
+ */
48
+ export function store(entry) {
49
+ _ensureLoaded();
50
+ entry.timestamp = new Date().toISOString();
51
+ entry.confidence = entry.confidence || 0.5;
52
+ _facts.push(entry);
53
+
54
+ try {
55
+ mkdirSync(KNOWLEDGE_DIR, { recursive: true });
56
+ writeFileSync(FACTS_FILE, _facts.map(f => JSON.stringify(f)).join("\n") + "\n");
57
+ log.info("stored", { type: entry.type, text: (entry.text || "").slice(0, 60) });
58
+ } catch (e) {
59
+ log.warn("store-failed", { error: e.message });
60
+ }
61
+ }
62
+
63
+ /**
64
+ * Retrieve relevant facts/patterns for a task.
65
+ * @param {string} query - task description or keywords
66
+ * @param {number} limit - max results
67
+ * @returns {Array} matching entries, sorted by relevance
68
+ */
69
+ export function retrieve(query, limit = 3) {
70
+ _ensureLoaded();
71
+ if (!_facts.length || !query) return [];
72
+
73
+ const words = query.toLowerCase().split(/\s+/).filter(w => w.length > 2);
74
+ const scored = _facts.map(f => {
75
+ const text = `${f.text || ""} ${(f.triggers || []).join(" ")} ${f.context || ""}`.toLowerCase();
76
+ const matches = words.filter(w => text.includes(w)).length;
77
+ return { ...f, score: matches * (f.confidence || 0.5) };
78
+ }).filter(f => f.score > 0);
79
+
80
+ scored.sort((a, b) => b.score - a.score);
81
+ return scored.slice(0, limit);
82
+ }
83
+
84
+ /**
85
+ * Update confidence of a fact (called when pattern reused successfully).
86
+ * @param {number} index - fact index
87
+ * @param {number} delta - confidence change (+0.1 on success, -0.1 on failure)
88
+ */
89
+ export function updateConfidence(index, delta) {
90
+ _ensureLoaded();
91
+ if (index >= 0 && index < _facts.length) {
92
+ _facts[index].confidence = Math.max(0, Math.min(1, (_facts[index].confidence || 0.5) + delta));
93
+ try {
94
+ writeFileSync(FACTS_FILE, _facts.map(f => JSON.stringify(f)).join("\n") + "\n");
95
+ } catch {}
96
+ }
97
+ }
98
+
99
+ /**
100
+ * Get all facts (for debugging/display).
101
+ */
102
+ export function getAll() {
103
+ _ensureLoaded();
104
+ return [..._facts];
105
+ }
106
+
107
+ /**
108
+ * Format retrieved knowledge for injection into prompt.
109
+ */
110
+ export function formatForPrompt(entries) {
111
+ if (!entries.length) return null;
112
+ const lines = entries.map(e => `- [${e.type}] ${e.text}`);
113
+ return `[KNOWLEDGE — learned from previous tasks]\n${lines.join("\n")}`;
114
+ }
@@ -0,0 +1,180 @@
1
+ /**
2
+ * Learning System — extract facts & patterns from completed tasks
3
+ *
4
+ * After successful task completion:
5
+ * 1. Analyze tool call sequence
6
+ * 2. Extract reusable patterns (e.g. "search → read → save → open")
7
+ * 3. Extract facts (e.g. "desktop_shell works for remote file creation")
8
+ * 4. Store via knowledge.js
9
+ *
10
+ * No LLM calls — pure code extraction from action history.
11
+ */
12
+
13
+ import { store as storeKnowledge, retrieve } from "./knowledge.js";
14
+ import { createLogger } from "../logging/logger.js";
15
+
16
+ const log = createLogger("learning");
17
+
18
+ /**
19
+ * Extract learnings from a completed task.
20
+ * @param {object} opts
21
+ * @param {Array} opts.messages - conversation messages from agent loop
22
+ * @param {string} opts.task - original task description
23
+ * @param {object} opts.plan - completed plan (goal + tasks)
24
+ * @param {string} opts.outcome - "success" | "partial" | "failed"
25
+ */
26
+ export function extractLearnings({ messages, task, plan, outcome }) {
27
+ if (outcome === "failed") return; // don't learn from failures (yet)
28
+
29
+ const toolCalls = _extractToolSequence(messages);
30
+ if (toolCalls.length < 2) return; // nothing to learn from 1 tool call
31
+
32
+ // Check if we already know this pattern
33
+ const existing = retrieve(task, 1);
34
+ if (existing.length > 0 && existing[0].score > 2) {
35
+ log.info("pattern-already-known", { task: task.slice(0, 40) });
36
+ return;
37
+ }
38
+
39
+ // Extract pattern: ordered sequence of unique tool names
40
+ const pattern = _buildPattern(toolCalls, task);
41
+ if (pattern) {
42
+ storeKnowledge(pattern);
43
+ log.info("pattern-stored", { type: pattern.type, text: pattern.text.slice(0, 60) });
44
+ }
45
+
46
+ // Extract facts: tools that worked in specific contexts
47
+ const facts = _buildFacts(toolCalls, task);
48
+ for (const fact of facts) {
49
+ storeKnowledge(fact);
50
+ log.info("fact-stored", { text: fact.text.slice(0, 60) });
51
+ }
52
+ }
53
+
54
+ /**
55
+ * Extract tool call sequence from messages.
56
+ */
57
+ function _extractToolSequence(messages) {
58
+ const calls = [];
59
+ for (const m of messages) {
60
+ if (m.role === "assistant" && m.tool_calls) {
61
+ for (const tc of m.tool_calls) {
62
+ const name = tc.function?.name || "unknown";
63
+ let args = {};
64
+ try { args = JSON.parse(tc.function?.arguments || "{}"); } catch {}
65
+ calls.push({ name, args });
66
+ }
67
+ }
68
+ }
69
+ return calls;
70
+ }
71
+
72
+ /**
73
+ * Build a reusable pattern from tool sequence.
74
+ */
75
+ function _buildPattern(toolCalls, task) {
76
+ // Deduplicate consecutive same-tool calls
77
+ const steps = [];
78
+ let prev = null;
79
+ for (const tc of toolCalls) {
80
+ const shortName = _shortToolName(tc.name);
81
+ if (shortName !== prev) {
82
+ steps.push(shortName);
83
+ prev = shortName;
84
+ }
85
+ }
86
+
87
+ if (steps.length < 2 || steps.length > 15) return null;
88
+
89
+ // Extract task keywords for triggers
90
+ const triggers = task.toLowerCase()
91
+ .split(/\s+/)
92
+ .filter(w => w.length > 3)
93
+ .filter(w => !["this", "that", "then", "with", "from", "using", "into", "open", "the"].includes(w))
94
+ .slice(0, 5);
95
+
96
+ return {
97
+ type: "pattern",
98
+ text: `For "${_summarizeTask(task)}": ${steps.join(" → ")}`,
99
+ context: task.slice(0, 100),
100
+ triggers,
101
+ confidence: 0.6,
102
+ steps,
103
+ };
104
+ }
105
+
106
+ /**
107
+ * Build facts from tool usage in context.
108
+ */
109
+ function _buildFacts(toolCalls, task) {
110
+ const facts = [];
111
+ const seen = new Set();
112
+
113
+ for (const tc of toolCalls) {
114
+ const shortName = _shortToolName(tc.name);
115
+
116
+ // Fact: MCP tool used for file operation
117
+ if (tc.name.includes("mcp_") && tc.args?.command && tc.args.command.includes("echo")) {
118
+ const key = "remote-echo";
119
+ if (!seen.has(key)) {
120
+ seen.add(key);
121
+ facts.push({
122
+ type: "fact",
123
+ text: `Use ${shortName} (remote shell) for file creation on remote systems, not local write_file`,
124
+ context: "remote file operations",
125
+ triggers: ["file", "save", "create", "remote", "desktop"],
126
+ confidence: 0.7,
127
+ });
128
+ }
129
+ }
130
+
131
+ // Fact: successful tool for reading web content
132
+ if (tc.name.includes("chrome") && tc.args?.action === "page_read") {
133
+ const key = "chrome-read";
134
+ if (!seen.has(key)) {
135
+ seen.add(key);
136
+ facts.push({
137
+ type: "fact",
138
+ text: `Use chrome page_read for extracting web page content (faster than screenshot+OCR)`,
139
+ context: "web content extraction",
140
+ triggers: ["read", "page", "content", "extract", "web"],
141
+ confidence: 0.8,
142
+ });
143
+ }
144
+ }
145
+ }
146
+
147
+ return facts;
148
+ }
149
+
150
+ /**
151
+ * Shorten tool name: mcp_screenbox_desktop_shell → desktop_shell
152
+ */
153
+ function _shortToolName(name) {
154
+ if (name.startsWith("mcp_")) {
155
+ const parts = name.split("_");
156
+ return parts.slice(2).join("_") || name;
157
+ }
158
+ return name;
159
+ }
160
+
161
+ /**
162
+ * Summarize task to ~30 chars.
163
+ */
164
+ function _summarizeTask(task) {
165
+ if (task.length <= 30) return task;
166
+ return task.slice(0, 27) + "...";
167
+ }
168
+
169
+ /**
170
+ * True if a tool result looks like a genuine failure. Matches structural
171
+ * failure markers (non-zero exit, "Error:" prefix, permission/path errors)
172
+ * -- deliberately NOT the bare word "error", which appears in legitimate
173
+ * content like a command string or a README about error handling.
174
+ *
175
+ * Shared with agent.js's failure-recovery gate so both use one definition.
176
+ */
177
+ export function isFailureResult(content) {
178
+ if (typeof content !== "string") return false;
179
+ return /error\s*\(exit|\berror:|\berror executing\b|\(exit\s+[1-9]|exit\s+code\s+[1-9]|exit\s+status\s+[1-9]|\bpermission denied\b|\bno such file\b|\bcommand not found\b|\bfailed to\b|\beconnrefused\b|\beconnreset\b|\benoent\b|\beacces\b|traceback \(most recent/i.test(content);
180
+ }
@@ -0,0 +1,187 @@
1
+ // Mode registry — maps intent classes to behavioral modes.
2
+ //
3
+ // Each mode defines:
4
+ // - promptAddition: behavioral rules injected into system prompt
5
+ // - maxIterations: iteration budget override
6
+ // - model: optional model override (e.g. use Sonnet for project, Flash for chat)
7
+ //
8
+ // Model override: set via AGENT_MODEL_<MODE> env var or mode.model field.
9
+ // If not set, uses session default (config.model).
10
+ //
11
+ // 22 intent classes → 12 modes (many-to-one mapping).
12
+
13
+ export const MODES = {
14
+ conversation: {
15
+ name: "conversation",
16
+ intents: ["chat", "knowledge_qa", "reasoning"],
17
+ model: null, // cheapest model OK — override via AGENT_MODEL_CONVERSATION
18
+ promptAddition: `[MODE: conversation]
19
+ - Prefer responding in text without tools. Use tools only if the question requires checking live data.
20
+ - Keep responses concise. No preamble. No trailing offers.
21
+ - Match the user's language and register.`,
22
+ maxIterations: 5,
23
+ },
24
+
25
+ quick_action: {
26
+ name: "quick_action",
27
+ intents: ["shell_command", "file_read"],
28
+ promptAddition: `[MODE: quick_action]
29
+ - Execute in 1-3 tool calls. No planning phase needed.
30
+ - Show the result immediately. Minimal explanation.
31
+ - If the result is an error — report it clearly, suggest a fix.`,
32
+ maxIterations: 5,
33
+ },
34
+
35
+ planning: {
36
+ name: "planning",
37
+ intents: ["task_create", "task_update", "task_view"],
38
+ promptAddition: `[MODE: planning]
39
+ - Create or manage tasks/plans. Use task tools (create_plan, update_task, list_tasks).
40
+ - When user says "plan X" — produce a text plan (bullet list). Do NOT auto-execute it.
41
+ - Only create a formal plan (via create_plan tool) if user explicitly confirms.`,
42
+ maxIterations: 5,
43
+ },
44
+
45
+ files: {
46
+ name: "files",
47
+ intents: ["file_write", "file_edit", "file_manage"],
48
+ promptAddition: `[MODE: files]
49
+ - File operations: create, edit, move, copy, delete, search.
50
+ - Always verify the file exists before editing. Show relevant content after changes.
51
+ - For edits: read first, modify, write back. Show the diff or changed section.
52
+ - Respect scope: only touch files the user mentioned.`,
53
+ maxIterations: 8,
54
+ },
55
+
56
+ project: {
57
+ name: "project",
58
+ intents: ["complex_multi"],
59
+ promptAddition: `[MODE: project]
60
+ - Multi-step feature work, refactoring, or bug fixing.
61
+ - Plan before executing: list what you'll change and why.
62
+ - After changes: run tests if available. Report results.
63
+ - Commit-ready: changes should be minimal and focused.`,
64
+ maxIterations: 30,
65
+ },
66
+
67
+ autonomous: {
68
+ name: "autonomous",
69
+ intents: [],
70
+ promptAddition: `[MODE: autonomous]
71
+ - Long self-directed work. You have extended budget.
72
+ - Write intermediate progress reports every 5-10 iterations.
73
+ - Self-verify: when you think you're done, check once more.
74
+ - If stuck after 3 attempts on same approach — try a different strategy.`,
75
+ maxIterations: 50,
76
+ },
77
+
78
+ research: {
79
+ name: "research",
80
+ intents: ["web_search", "web_fetch"],
81
+ promptAddition: `[MODE: research]
82
+ - Gather information from web and local sources.
83
+ - Structure findings: summary first, then details.
84
+ - Cite sources when using web results.
85
+ - If multiple sources disagree — note the discrepancy.`,
86
+ maxIterations: 10,
87
+ },
88
+
89
+ computer_use: {
90
+ name: "computer_use",
91
+ intents: ["desktop"],
92
+ promptAddition: `[MODE: computer_use]
93
+ - Desktop GUI interaction via screenshot/click/type tools.
94
+ - MANDATORY workflow: screenshot → look (OCR) → click. Never click blind.
95
+ - Prefer keyboard shortcuts over mouse when possible.
96
+ - If Chrome is open: use Chrome semantics (page_read, page_map) before OCR.`,
97
+ maxIterations: 20,
98
+ },
99
+
100
+ delegation: {
101
+ name: "delegation",
102
+ intents: ["agent_spawn"],
103
+ promptAddition: `[MODE: delegation]
104
+ - Spawn child agents for parallel work.
105
+ - Give each agent a clear, specific task description.
106
+ - Wait for results. Aggregate and summarize for the user.
107
+ - Max 3 concurrent child agents unless user specifies more.`,
108
+ maxIterations: 10,
109
+ },
110
+
111
+ secretary: {
112
+ name: "secretary",
113
+ intents: [],
114
+ promptAddition: `[MODE: secretary]
115
+ - Communication tasks: email, messages, scheduling.
116
+ - Use formal tone for external communications.
117
+ - Always show draft before sending. Wait for user approval.
118
+ - Include all relevant context (who, what, when, where).`,
119
+ maxIterations: 5,
120
+ },
121
+
122
+ monitoring: {
123
+ name: "monitoring",
124
+ intents: [],
125
+ promptAddition: `[MODE: monitoring]
126
+ - Periodic checks and alerts.
127
+ - Run check → compare with previous state → report changes.
128
+ - Only alert on meaningful changes, not noise.
129
+ - Keep logs of what was checked and when.`,
130
+ maxIterations: 5,
131
+ },
132
+
133
+ creative: {
134
+ name: "creative",
135
+ intents: ["creative_text"],
136
+ promptAddition: `[MODE: creative]
137
+ - Content generation: writing, editing, formatting.
138
+ - Longer output is acceptable. Use markdown formatting.
139
+ - If user asks to save result to file — use write_file.`,
140
+ maxIterations: 5,
141
+ },
142
+ };
143
+
144
+ // Build reverse map: intent → mode
145
+ const _intentToMode = new Map();
146
+ for (const [modeName, mode] of Object.entries(MODES)) {
147
+ for (const intent of mode.intents) {
148
+ _intentToMode.set(intent, modeName);
149
+ }
150
+ }
151
+
152
+ /**
153
+ * Get mode config for a given intent class.
154
+ * Falls back to "project" for unknown intents (safest default).
155
+ * Resolves model: env AGENT_MODEL_<MODE> > mode.model > null (use session default).
156
+ * @param {string} intentClass — from intent classifier
157
+ * @returns {{ name, promptAddition, maxIterations, model }}
158
+ */
159
+ export function getModeForIntent(intentClass) {
160
+ const modeName = _intentToMode.get(intentClass) || "project";
161
+ const mode = MODES[modeName];
162
+ // Resolve model: env var takes priority
163
+ const envKey = `AGENT_MODEL_${modeName.toUpperCase()}`;
164
+ const resolvedModel = process.env[envKey] || mode.model || null;
165
+ return { ...mode, model: resolvedModel };
166
+ }
167
+
168
+ /**
169
+ * Get mode by name directly.
170
+ * @param {string} modeName
171
+ * @returns {object|null}
172
+ */
173
+ export function getMode(modeName) {
174
+ return MODES[modeName] || null;
175
+ }
176
+
177
+ /**
178
+ * List all available modes (for /help or tool listing).
179
+ * @returns {Array<{name, description}>}
180
+ */
181
+ export function listModes() {
182
+ return Object.values(MODES).map(m => ({
183
+ name: m.name,
184
+ intents: m.intents,
185
+ maxIterations: m.maxIterations,
186
+ }));
187
+ }
@@ -0,0 +1,91 @@
1
+ // Which of the three things happened on a turn that changed nothing.
2
+ //
3
+ // A turn that changed nothing must tell the operator WHICH: the agent did not find what to
4
+ // change, or found it and could not apply the change, or there was nothing to
5
+ // change. Only the model that did the turn can say, so it is asked.
6
+ //
7
+ // It used to be asked inside the conversation: a [OUTCOME] system message, then
8
+ // `continue`, and the loop took the next reply as the turn's answer. The
9
+ // question says "in one sentence", the model obeys, and the one sentence
10
+ // REPLACED the answer. Measured on ten readiness probes, 2026-09-21: eight
11
+ // turns of ten came back to the operator as a single sentence about change,
12
+ // their actual answer lost inside the loop. cut-0003 asked "go to the server",
13
+ // ran a command, and the whole answer was "The request was not asking for a
14
+ // change."
15
+ //
16
+ // So the question is not a turn of the conversation. It is one small call of
17
+ // its own, like the classifier, and its answer is a line of explanation, never
18
+ // the answer itself. That also makes it cheaper: the in-conversation ask
19
+ // re-sent the whole context and the tool schemas, around 9k prompt tokens on
20
+ // the measured run; this sends the evidence and nothing else.
21
+
22
+ import { config } from "../config.js";
23
+ import { chatCompletion } from "../api/client.js";
24
+ import { createLogger } from "../logging/logger.js";
25
+
26
+ const log = createLogger("outcome-ask");
27
+
28
+ const SYSTEM = `You are reporting on one turn of an agent that finished without changing any file on disk.
29
+
30
+ Say, in ONE short sentence addressed to the operator, which of these is true:
31
+ - you did not find what to change;
32
+ - you found what to change but did not apply the change;
33
+ - there was nothing to change;
34
+ - the request was not asking for a change.
35
+
36
+ Reply with that sentence and nothing else. No preamble, no JSON, no quotes.`;
37
+
38
+ /**
39
+ * Ask why the turn changed nothing.
40
+ *
41
+ * @param {object} turn
42
+ * @param {string} turn.request what the user asked for
43
+ * @param {string} turn.answer what the agent replied
44
+ * @param {string[]} turn.tools names of the tools the turn called, in order
45
+ * @param {AbortSignal} [turn.signal]
46
+ * @returns {Promise<string|null>} one sentence, or null when the call did not happen
47
+ */
48
+ export async function askOutcome({ request, answer, tools = [], signal }) {
49
+ if (!config.apiKey) return null;
50
+
51
+ const evidence = [
52
+ `REQUEST: ${String(request || "").slice(0, 1500)}`,
53
+ `YOUR ANSWER: ${String(answer || "").slice(0, 1500)}`,
54
+ `TOOLS YOU CALLED: ${tools.length ? tools.join(", ") : "(none)"}`,
55
+ "NOTHING ON DISK CHANGED.",
56
+ ].join("\n\n");
57
+
58
+ try {
59
+ const { message } = await chatCompletion(
60
+ [
61
+ { role: "system", content: SYSTEM },
62
+ { role: "user", content: evidence },
63
+ ],
64
+ [],
65
+ null,
66
+ {
67
+ source: "outcome",
68
+ model: config.model,
69
+ maxTokens: 80,
70
+ temperature: 0,
71
+ stream: false,
72
+ signal,
73
+ // A ceiling of its own, always: the turn's signal cancels the ask when
74
+ // the user aborts, but it never expires on its own, and a hung reason
75
+ // must not hold back the answer that is already written.
76
+ timeoutMs: 15000,
77
+ },
78
+ );
79
+ const text = (message?.content || "").trim();
80
+ if (!text) return null;
81
+ // One sentence is what was asked for; a model that writes three gets cut,
82
+ // because this line sits next to the answer and must not compete with it.
83
+ return text.split("\n")[0].slice(0, 300);
84
+ } catch (err) {
85
+ // The reason is best effort. The FACT that nothing changed is not: the
86
+ // caller states that either way. Losing the
87
+ // reason must never cost the operator the fact.
88
+ log.warn("outcome-ask failed", { error: err.message });
89
+ return null;
90
+ }
91
+ }
@@ -0,0 +1,76 @@
1
+ // Project context (FLINT.md) — the agent's own notebook for the project
2
+ // it's currently working in. Walks up from CWD looking for FLINT.md,
3
+ // scans for prompt injection, returns content. Cached per-CWD for the
4
+ // life of the process (invalidated by file mtime change).
5
+ //
6
+ // Injected at the top of EVERY LLM prompt — classifier, agent loop,
7
+ // reflection, etc. — so the agent's own project rules are visible
8
+ // before any other context.
9
+
10
+ import { existsSync, readFileSync, statSync } from "node:fs";
11
+ import { join } from "node:path";
12
+ import { config } from "../config.js";
13
+ import { detectInjection } from "../security/content-fence.js";
14
+
15
+ const MAX_CHARS = config.maxContextChars;
16
+ const _cache = new Map(); // path → { content, mtimeMs }
17
+
18
+ function findFile(startDir) {
19
+ let dir = startDir;
20
+ for (let i = 0; i < 10; i++) {
21
+ const p = join(dir, "FLINT.md");
22
+ if (existsSync(p)) return p;
23
+ const parent = join(dir, "..");
24
+ if (parent === dir) break;
25
+ dir = parent;
26
+ }
27
+ return null;
28
+ }
29
+
30
+ export function loadFlintMd(cwd = process.cwd()) {
31
+ const p = findFile(cwd);
32
+ if (!p) return null;
33
+
34
+ let mtimeMs;
35
+ try {
36
+ mtimeMs = statSync(p).mtimeMs;
37
+ } catch {
38
+ return null;
39
+ }
40
+
41
+ const cached = _cache.get(p);
42
+ if (cached && cached.mtimeMs === mtimeMs) return cached.content;
43
+
44
+ let content;
45
+ try {
46
+ content = readFileSync(p, "utf-8");
47
+ } catch {
48
+ return null;
49
+ }
50
+ if (content.length > MAX_CHARS) {
51
+ content = content.slice(0, MAX_CHARS) + "\n... (truncated)";
52
+ }
53
+
54
+ // Injection scan — FLINT.md is user- or agent-written, but we still
55
+ // scan because the agent itself could be fooled into writing malicious
56
+ // instructions there via a crafted user message in a prior turn.
57
+ const scan = detectInjection(content);
58
+ if (scan.detected && scan.score >= 3) {
59
+ const blocked = `[BLOCKED: FLINT.md at ${p} failed injection scan (score ${scan.score})]`;
60
+ _cache.set(p, { content: blocked, mtimeMs });
61
+ return blocked;
62
+ }
63
+
64
+ _cache.set(p, { content, mtimeMs });
65
+ return content;
66
+ }
67
+
68
+ export function formatFlintMdBlock(cwd = process.cwd()) {
69
+ const content = loadFlintMd(cwd);
70
+ if (!content) return "";
71
+ return [
72
+ "<trusted-context name=\"project-rules\" source=\"FLINT.md\">",
73
+ content,
74
+ "</trusted-context>",
75
+ ].join("\n");
76
+ }