flint-agent 1.14.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.example +108 -0
- package/CHANGELOG.md +55 -0
- package/FEATURES.md +298 -0
- package/LICENSE +21 -0
- package/README.md +435 -0
- package/bin/flint.js +47 -0
- package/config/classifier-prompt.md +218 -0
- package/config/models-curated.json +4 -0
- package/config/providers.json +74 -0
- package/package.json +92 -0
- package/patches/ink+6.8.0.patch +78 -0
- package/profiles/desktop.md +65 -0
- package/profiles/generic.md +20 -0
- package/profiles/marketer.md +20 -0
- package/profiles/profiles.json +34 -0
- package/profiles/ux-reviewer.md +25 -0
- package/src/agent/agent.js +1743 -0
- package/src/agent/auto.js +346 -0
- package/src/agent/backoff.js +143 -0
- package/src/agent/compression.js +310 -0
- package/src/agent/content-resolver.js +180 -0
- package/src/agent/flow-controller.js +309 -0
- package/src/agent/intent-manifest.js +231 -0
- package/src/agent/intent-timeout.js +46 -0
- package/src/agent/intent.js +633 -0
- package/src/agent/knowledge.js +114 -0
- package/src/agent/learning.js +180 -0
- package/src/agent/modes.js +187 -0
- package/src/agent/outcome-ask.js +91 -0
- package/src/agent/project-context.js +76 -0
- package/src/agent/prompt-budget.js +117 -0
- package/src/agent/reflection-extractor.js +140 -0
- package/src/agent/steering.js +86 -0
- package/src/agent/supervisor.js +430 -0
- package/src/agent/swap.js +443 -0
- package/src/agent/system-prompt.js +446 -0
- package/src/agent/time-stamp.js +48 -0
- package/src/agent/tool-guard.js +201 -0
- package/src/agent/toolcall-text.js +162 -0
- package/src/agent/usage.js +297 -0
- package/src/agent/vision.js +94 -0
- package/src/agent/watchdog.js +139 -0
- package/src/agent/workspace-changes.js +177 -0
- package/src/api/address.js +14 -0
- package/src/api/client.js +280 -0
- package/src/api/server.js +535 -0
- package/src/api/stream-pipe.js +113 -0
- package/src/app-state.js +39 -0
- package/src/bootstrap.js +501 -0
- package/src/bus/drain-loop.js +497 -0
- package/src/bus/index.js +270 -0
- package/src/bus/plugins.js +65 -0
- package/src/child-idle.js +14 -0
- package/src/cli.js +118 -0
- package/src/commands/commands.js +1297 -0
- package/src/commands/registry.js +132 -0
- package/src/components/App.js +491 -0
- package/src/components/CarefulMenu.js +145 -0
- package/src/components/HistoryWriter.js +86 -0
- package/src/components/LineInput.js +69 -0
- package/src/components/LiveZone.js +294 -0
- package/src/components/OverlayMenu.js +179 -0
- package/src/components/SystemPanel.js +156 -0
- package/src/components/Table.js +54 -0
- package/src/config.js +249 -0
- package/src/free-models.js +230 -0
- package/src/index.js +1111 -0
- package/src/input-handler.js +13 -0
- package/src/input-text.js +123 -0
- package/src/launcher.js +129 -0
- package/src/logging/api-log.js +95 -0
- package/src/logging/chat-log-follower.js +113 -0
- package/src/logging/chat-log.js +15 -0
- package/src/logging/log-collector.js +182 -0
- package/src/logging/logger.js +112 -0
- package/src/logging/tool-log.js +20 -0
- package/src/mcp-client.js +314 -0
- package/src/memory/conversation-digest.js +113 -0
- package/src/memory/extract-facts.js +98 -0
- package/src/memory/facts.js +181 -0
- package/src/memory/inbox.js +63 -0
- package/src/memory/markdown.js +38 -0
- package/src/memory/patterns.js +185 -0
- package/src/memory/project.js +66 -0
- package/src/memory/reflections.js +74 -0
- package/src/memory/retrieval.js +84 -0
- package/src/memory/rules.js +105 -0
- package/src/memory/session-facts.js +125 -0
- package/src/memory/skills.js +191 -0
- package/src/memory/sqlite-store.js +653 -0
- package/src/memory/store.js +208 -0
- package/src/memory/tools.js +196 -0
- package/src/memory/user-model.js +86 -0
- package/src/message-handler.js +775 -0
- package/src/model-check.js +218 -0
- package/src/plugins/loader.js +120 -0
- package/src/plugins/manager.js +88 -0
- package/src/production-env.js +22 -0
- package/src/profiles.js +42 -0
- package/src/providers/adapters/anthropic.js +270 -0
- package/src/providers/adapters/openai.js +120 -0
- package/src/providers/keys-dpapi.js +41 -0
- package/src/providers/keys-fallback.js +31 -0
- package/src/providers/keys.js +132 -0
- package/src/providers/models.js +154 -0
- package/src/providers/registry.js +56 -0
- package/src/providers/state.js +56 -0
- package/src/registry.js +96 -0
- package/src/restart.js +29 -0
- package/src/sandbox/backend.js +130 -0
- package/src/security/api-auth.js +132 -0
- package/src/security/audit.js +98 -0
- package/src/security/child-policy.js +41 -0
- package/src/security/command-guard.js +173 -0
- package/src/security/content-fence.js +250 -0
- package/src/security/content-validator.js +132 -0
- package/src/security/index.js +143 -0
- package/src/security/network-guard.js +126 -0
- package/src/security/pairing.js +180 -0
- package/src/security/path-guard.js +140 -0
- package/src/security/persona-guard.js +67 -0
- package/src/security/policies.js +452 -0
- package/src/security/safety-constants.js +34 -0
- package/src/security/watchdog.js +107 -0
- package/src/sessions.js +130 -0
- package/src/spend.js +97 -0
- package/src/startup-watchdog.js +59 -0
- package/src/stdio/args.js +71 -0
- package/src/stdio/guard.js +59 -0
- package/src/stdio/protocol.js +167 -0
- package/src/stdio/run.js +106 -0
- package/src/stdio/session.js +180 -0
- package/src/store/agent-slice.js +306 -0
- package/src/store/dataset-slice.js +73 -0
- package/src/store/index.js +22 -0
- package/src/store/process-slice.js +135 -0
- package/src/store/session-slice.js +191 -0
- package/src/store/ui-slice.js +119 -0
- package/src/tasks/db.js +184 -0
- package/src/tasks/queries.js +589 -0
- package/src/tools/agent-tools.js +473 -0
- package/src/tools/checkpoint.js +152 -0
- package/src/tools/command-approvals.js +180 -0
- package/src/tools/dataset.js +50 -0
- package/src/tools/filesystem.js +682 -0
- package/src/tools/inbox-tools.js +48 -0
- package/src/tools/mesh.js +135 -0
- package/src/tools/own-env.js +136 -0
- package/src/tools/permissions.js +681 -0
- package/src/tools/plugin-tools.js +123 -0
- package/src/tools/process-tools.js +595 -0
- package/src/tools/registry.js +307 -0
- package/src/tools/swap-tools.js +72 -0
- package/src/tools/system.js +662 -0
- package/src/tools/tasks.js +532 -0
- package/src/tools/tool-search.js +171 -0
- package/src/ui/header.js +140 -0
- package/src/ui/input-cursor.js +23 -0
- package/src/ui/last-line.js +25 -0
- package/src/ui/line-edit.js +135 -0
- package/src/ui/output.js +399 -0
- package/src/ui/paste-tokens.js +131 -0
- package/src/ui/prompt-attention.js +134 -0
- package/src/ui/render-options.js +13 -0
- package/src/ui/replay.js +94 -0
- package/src/ui/splash.js +49 -0
- package/src/ui/status-level.js +36 -0
- package/src/ui/tool-ledger.js +203 -0
- package/src/ui/window-title.js +150 -0
- package/src/update.js +205 -0
- package/system.md +63 -0
|
@@ -0,0 +1,114 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Knowledge Store — backend-agnostic facts & patterns
|
|
3
|
+
*
|
|
4
|
+
* Stores learned facts and patterns. Retrieves by keyword match.
|
|
5
|
+
* Backend adapters: text file (default), mesh, agent-memory.
|
|
6
|
+
*
|
|
7
|
+
* Used by:
|
|
8
|
+
* - Layer 3 (Execution): retrieve before planning
|
|
9
|
+
* - Layer 4 (Verification Gate): store after successful task
|
|
10
|
+
*/
|
|
11
|
+
|
|
12
|
+
import { readFileSync, writeFileSync, existsSync, mkdirSync } from "node:fs";
|
|
13
|
+
import { join } from "node:path";
|
|
14
|
+
import { createLogger } from "../logging/logger.js";
|
|
15
|
+
import { config } from "../config.js";
|
|
16
|
+
|
|
17
|
+
const log = createLogger("knowledge");
|
|
18
|
+
|
|
19
|
+
// --- Storage path ---
|
|
20
|
+
const KNOWLEDGE_DIR = join(config.sessionsDir || process.cwd(), "..", "knowledge");
|
|
21
|
+
const FACTS_FILE = join(KNOWLEDGE_DIR, "facts.jsonl");
|
|
22
|
+
|
|
23
|
+
// --- In-memory cache ---
|
|
24
|
+
let _facts = [];
|
|
25
|
+
let _loaded = false;
|
|
26
|
+
|
|
27
|
+
/**
|
|
28
|
+
* Load facts from disk (lazy, once).
|
|
29
|
+
*/
|
|
30
|
+
function _ensureLoaded() {
|
|
31
|
+
if (_loaded) return;
|
|
32
|
+
_loaded = true;
|
|
33
|
+
try {
|
|
34
|
+
if (existsSync(FACTS_FILE)) {
|
|
35
|
+
const lines = readFileSync(FACTS_FILE, "utf-8").split("\n").filter(Boolean);
|
|
36
|
+
_facts = lines.map(l => { try { return JSON.parse(l); } catch { return null; } }).filter(Boolean);
|
|
37
|
+
log.info("loaded", { count: _facts.length });
|
|
38
|
+
}
|
|
39
|
+
} catch (e) {
|
|
40
|
+
log.warn("load-failed", { error: e.message });
|
|
41
|
+
}
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
/**
|
|
45
|
+
* Store a fact or pattern.
|
|
46
|
+
* @param {object} entry - { type: "fact"|"pattern", text, context, confidence, triggers }
|
|
47
|
+
*/
|
|
48
|
+
export function store(entry) {
|
|
49
|
+
_ensureLoaded();
|
|
50
|
+
entry.timestamp = new Date().toISOString();
|
|
51
|
+
entry.confidence = entry.confidence || 0.5;
|
|
52
|
+
_facts.push(entry);
|
|
53
|
+
|
|
54
|
+
try {
|
|
55
|
+
mkdirSync(KNOWLEDGE_DIR, { recursive: true });
|
|
56
|
+
writeFileSync(FACTS_FILE, _facts.map(f => JSON.stringify(f)).join("\n") + "\n");
|
|
57
|
+
log.info("stored", { type: entry.type, text: (entry.text || "").slice(0, 60) });
|
|
58
|
+
} catch (e) {
|
|
59
|
+
log.warn("store-failed", { error: e.message });
|
|
60
|
+
}
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
/**
|
|
64
|
+
* Retrieve relevant facts/patterns for a task.
|
|
65
|
+
* @param {string} query - task description or keywords
|
|
66
|
+
* @param {number} limit - max results
|
|
67
|
+
* @returns {Array} matching entries, sorted by relevance
|
|
68
|
+
*/
|
|
69
|
+
export function retrieve(query, limit = 3) {
|
|
70
|
+
_ensureLoaded();
|
|
71
|
+
if (!_facts.length || !query) return [];
|
|
72
|
+
|
|
73
|
+
const words = query.toLowerCase().split(/\s+/).filter(w => w.length > 2);
|
|
74
|
+
const scored = _facts.map(f => {
|
|
75
|
+
const text = `${f.text || ""} ${(f.triggers || []).join(" ")} ${f.context || ""}`.toLowerCase();
|
|
76
|
+
const matches = words.filter(w => text.includes(w)).length;
|
|
77
|
+
return { ...f, score: matches * (f.confidence || 0.5) };
|
|
78
|
+
}).filter(f => f.score > 0);
|
|
79
|
+
|
|
80
|
+
scored.sort((a, b) => b.score - a.score);
|
|
81
|
+
return scored.slice(0, limit);
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
/**
|
|
85
|
+
* Update confidence of a fact (called when pattern reused successfully).
|
|
86
|
+
* @param {number} index - fact index
|
|
87
|
+
* @param {number} delta - confidence change (+0.1 on success, -0.1 on failure)
|
|
88
|
+
*/
|
|
89
|
+
export function updateConfidence(index, delta) {
|
|
90
|
+
_ensureLoaded();
|
|
91
|
+
if (index >= 0 && index < _facts.length) {
|
|
92
|
+
_facts[index].confidence = Math.max(0, Math.min(1, (_facts[index].confidence || 0.5) + delta));
|
|
93
|
+
try {
|
|
94
|
+
writeFileSync(FACTS_FILE, _facts.map(f => JSON.stringify(f)).join("\n") + "\n");
|
|
95
|
+
} catch {}
|
|
96
|
+
}
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
/**
|
|
100
|
+
* Get all facts (for debugging/display).
|
|
101
|
+
*/
|
|
102
|
+
export function getAll() {
|
|
103
|
+
_ensureLoaded();
|
|
104
|
+
return [..._facts];
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
/**
|
|
108
|
+
* Format retrieved knowledge for injection into prompt.
|
|
109
|
+
*/
|
|
110
|
+
export function formatForPrompt(entries) {
|
|
111
|
+
if (!entries.length) return null;
|
|
112
|
+
const lines = entries.map(e => `- [${e.type}] ${e.text}`);
|
|
113
|
+
return `[KNOWLEDGE — learned from previous tasks]\n${lines.join("\n")}`;
|
|
114
|
+
}
|
|
@@ -0,0 +1,180 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Learning System — extract facts & patterns from completed tasks
|
|
3
|
+
*
|
|
4
|
+
* After successful task completion:
|
|
5
|
+
* 1. Analyze tool call sequence
|
|
6
|
+
* 2. Extract reusable patterns (e.g. "search → read → save → open")
|
|
7
|
+
* 3. Extract facts (e.g. "desktop_shell works for remote file creation")
|
|
8
|
+
* 4. Store via knowledge.js
|
|
9
|
+
*
|
|
10
|
+
* No LLM calls — pure code extraction from action history.
|
|
11
|
+
*/
|
|
12
|
+
|
|
13
|
+
import { store as storeKnowledge, retrieve } from "./knowledge.js";
|
|
14
|
+
import { createLogger } from "../logging/logger.js";
|
|
15
|
+
|
|
16
|
+
const log = createLogger("learning");
|
|
17
|
+
|
|
18
|
+
/**
|
|
19
|
+
* Extract learnings from a completed task.
|
|
20
|
+
* @param {object} opts
|
|
21
|
+
* @param {Array} opts.messages - conversation messages from agent loop
|
|
22
|
+
* @param {string} opts.task - original task description
|
|
23
|
+
* @param {object} opts.plan - completed plan (goal + tasks)
|
|
24
|
+
* @param {string} opts.outcome - "success" | "partial" | "failed"
|
|
25
|
+
*/
|
|
26
|
+
export function extractLearnings({ messages, task, plan, outcome }) {
|
|
27
|
+
if (outcome === "failed") return; // don't learn from failures (yet)
|
|
28
|
+
|
|
29
|
+
const toolCalls = _extractToolSequence(messages);
|
|
30
|
+
if (toolCalls.length < 2) return; // nothing to learn from 1 tool call
|
|
31
|
+
|
|
32
|
+
// Check if we already know this pattern
|
|
33
|
+
const existing = retrieve(task, 1);
|
|
34
|
+
if (existing.length > 0 && existing[0].score > 2) {
|
|
35
|
+
log.info("pattern-already-known", { task: task.slice(0, 40) });
|
|
36
|
+
return;
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
// Extract pattern: ordered sequence of unique tool names
|
|
40
|
+
const pattern = _buildPattern(toolCalls, task);
|
|
41
|
+
if (pattern) {
|
|
42
|
+
storeKnowledge(pattern);
|
|
43
|
+
log.info("pattern-stored", { type: pattern.type, text: pattern.text.slice(0, 60) });
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
// Extract facts: tools that worked in specific contexts
|
|
47
|
+
const facts = _buildFacts(toolCalls, task);
|
|
48
|
+
for (const fact of facts) {
|
|
49
|
+
storeKnowledge(fact);
|
|
50
|
+
log.info("fact-stored", { text: fact.text.slice(0, 60) });
|
|
51
|
+
}
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
/**
|
|
55
|
+
* Extract tool call sequence from messages.
|
|
56
|
+
*/
|
|
57
|
+
function _extractToolSequence(messages) {
|
|
58
|
+
const calls = [];
|
|
59
|
+
for (const m of messages) {
|
|
60
|
+
if (m.role === "assistant" && m.tool_calls) {
|
|
61
|
+
for (const tc of m.tool_calls) {
|
|
62
|
+
const name = tc.function?.name || "unknown";
|
|
63
|
+
let args = {};
|
|
64
|
+
try { args = JSON.parse(tc.function?.arguments || "{}"); } catch {}
|
|
65
|
+
calls.push({ name, args });
|
|
66
|
+
}
|
|
67
|
+
}
|
|
68
|
+
}
|
|
69
|
+
return calls;
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
/**
|
|
73
|
+
* Build a reusable pattern from tool sequence.
|
|
74
|
+
*/
|
|
75
|
+
function _buildPattern(toolCalls, task) {
|
|
76
|
+
// Deduplicate consecutive same-tool calls
|
|
77
|
+
const steps = [];
|
|
78
|
+
let prev = null;
|
|
79
|
+
for (const tc of toolCalls) {
|
|
80
|
+
const shortName = _shortToolName(tc.name);
|
|
81
|
+
if (shortName !== prev) {
|
|
82
|
+
steps.push(shortName);
|
|
83
|
+
prev = shortName;
|
|
84
|
+
}
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
if (steps.length < 2 || steps.length > 15) return null;
|
|
88
|
+
|
|
89
|
+
// Extract task keywords for triggers
|
|
90
|
+
const triggers = task.toLowerCase()
|
|
91
|
+
.split(/\s+/)
|
|
92
|
+
.filter(w => w.length > 3)
|
|
93
|
+
.filter(w => !["this", "that", "then", "with", "from", "using", "into", "open", "the"].includes(w))
|
|
94
|
+
.slice(0, 5);
|
|
95
|
+
|
|
96
|
+
return {
|
|
97
|
+
type: "pattern",
|
|
98
|
+
text: `For "${_summarizeTask(task)}": ${steps.join(" → ")}`,
|
|
99
|
+
context: task.slice(0, 100),
|
|
100
|
+
triggers,
|
|
101
|
+
confidence: 0.6,
|
|
102
|
+
steps,
|
|
103
|
+
};
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
/**
|
|
107
|
+
* Build facts from tool usage in context.
|
|
108
|
+
*/
|
|
109
|
+
function _buildFacts(toolCalls, task) {
|
|
110
|
+
const facts = [];
|
|
111
|
+
const seen = new Set();
|
|
112
|
+
|
|
113
|
+
for (const tc of toolCalls) {
|
|
114
|
+
const shortName = _shortToolName(tc.name);
|
|
115
|
+
|
|
116
|
+
// Fact: MCP tool used for file operation
|
|
117
|
+
if (tc.name.includes("mcp_") && tc.args?.command && tc.args.command.includes("echo")) {
|
|
118
|
+
const key = "remote-echo";
|
|
119
|
+
if (!seen.has(key)) {
|
|
120
|
+
seen.add(key);
|
|
121
|
+
facts.push({
|
|
122
|
+
type: "fact",
|
|
123
|
+
text: `Use ${shortName} (remote shell) for file creation on remote systems, not local write_file`,
|
|
124
|
+
context: "remote file operations",
|
|
125
|
+
triggers: ["file", "save", "create", "remote", "desktop"],
|
|
126
|
+
confidence: 0.7,
|
|
127
|
+
});
|
|
128
|
+
}
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
// Fact: successful tool for reading web content
|
|
132
|
+
if (tc.name.includes("chrome") && tc.args?.action === "page_read") {
|
|
133
|
+
const key = "chrome-read";
|
|
134
|
+
if (!seen.has(key)) {
|
|
135
|
+
seen.add(key);
|
|
136
|
+
facts.push({
|
|
137
|
+
type: "fact",
|
|
138
|
+
text: `Use chrome page_read for extracting web page content (faster than screenshot+OCR)`,
|
|
139
|
+
context: "web content extraction",
|
|
140
|
+
triggers: ["read", "page", "content", "extract", "web"],
|
|
141
|
+
confidence: 0.8,
|
|
142
|
+
});
|
|
143
|
+
}
|
|
144
|
+
}
|
|
145
|
+
}
|
|
146
|
+
|
|
147
|
+
return facts;
|
|
148
|
+
}
|
|
149
|
+
|
|
150
|
+
/**
|
|
151
|
+
* Shorten tool name: mcp_screenbox_desktop_shell → desktop_shell
|
|
152
|
+
*/
|
|
153
|
+
function _shortToolName(name) {
|
|
154
|
+
if (name.startsWith("mcp_")) {
|
|
155
|
+
const parts = name.split("_");
|
|
156
|
+
return parts.slice(2).join("_") || name;
|
|
157
|
+
}
|
|
158
|
+
return name;
|
|
159
|
+
}
|
|
160
|
+
|
|
161
|
+
/**
|
|
162
|
+
* Summarize task to ~30 chars.
|
|
163
|
+
*/
|
|
164
|
+
function _summarizeTask(task) {
|
|
165
|
+
if (task.length <= 30) return task;
|
|
166
|
+
return task.slice(0, 27) + "...";
|
|
167
|
+
}
|
|
168
|
+
|
|
169
|
+
/**
|
|
170
|
+
* True if a tool result looks like a genuine failure. Matches structural
|
|
171
|
+
* failure markers (non-zero exit, "Error:" prefix, permission/path errors)
|
|
172
|
+
* -- deliberately NOT the bare word "error", which appears in legitimate
|
|
173
|
+
* content like a command string or a README about error handling.
|
|
174
|
+
*
|
|
175
|
+
* Shared with agent.js's failure-recovery gate so both use one definition.
|
|
176
|
+
*/
|
|
177
|
+
export function isFailureResult(content) {
|
|
178
|
+
if (typeof content !== "string") return false;
|
|
179
|
+
return /error\s*\(exit|\berror:|\berror executing\b|\(exit\s+[1-9]|exit\s+code\s+[1-9]|exit\s+status\s+[1-9]|\bpermission denied\b|\bno such file\b|\bcommand not found\b|\bfailed to\b|\beconnrefused\b|\beconnreset\b|\benoent\b|\beacces\b|traceback \(most recent/i.test(content);
|
|
180
|
+
}
|
|
@@ -0,0 +1,187 @@
|
|
|
1
|
+
// Mode registry — maps intent classes to behavioral modes.
|
|
2
|
+
//
|
|
3
|
+
// Each mode defines:
|
|
4
|
+
// - promptAddition: behavioral rules injected into system prompt
|
|
5
|
+
// - maxIterations: iteration budget override
|
|
6
|
+
// - model: optional model override (e.g. use Sonnet for project, Flash for chat)
|
|
7
|
+
//
|
|
8
|
+
// Model override: set via AGENT_MODEL_<MODE> env var or mode.model field.
|
|
9
|
+
// If not set, uses session default (config.model).
|
|
10
|
+
//
|
|
11
|
+
// 22 intent classes → 12 modes (many-to-one mapping).
|
|
12
|
+
|
|
13
|
+
export const MODES = {
|
|
14
|
+
conversation: {
|
|
15
|
+
name: "conversation",
|
|
16
|
+
intents: ["chat", "knowledge_qa", "reasoning"],
|
|
17
|
+
model: null, // cheapest model OK — override via AGENT_MODEL_CONVERSATION
|
|
18
|
+
promptAddition: `[MODE: conversation]
|
|
19
|
+
- Prefer responding in text without tools. Use tools only if the question requires checking live data.
|
|
20
|
+
- Keep responses concise. No preamble. No trailing offers.
|
|
21
|
+
- Match the user's language and register.`,
|
|
22
|
+
maxIterations: 5,
|
|
23
|
+
},
|
|
24
|
+
|
|
25
|
+
quick_action: {
|
|
26
|
+
name: "quick_action",
|
|
27
|
+
intents: ["shell_command", "file_read"],
|
|
28
|
+
promptAddition: `[MODE: quick_action]
|
|
29
|
+
- Execute in 1-3 tool calls. No planning phase needed.
|
|
30
|
+
- Show the result immediately. Minimal explanation.
|
|
31
|
+
- If the result is an error — report it clearly, suggest a fix.`,
|
|
32
|
+
maxIterations: 5,
|
|
33
|
+
},
|
|
34
|
+
|
|
35
|
+
planning: {
|
|
36
|
+
name: "planning",
|
|
37
|
+
intents: ["task_create", "task_update", "task_view"],
|
|
38
|
+
promptAddition: `[MODE: planning]
|
|
39
|
+
- Create or manage tasks/plans. Use task tools (create_plan, update_task, list_tasks).
|
|
40
|
+
- When user says "plan X" — produce a text plan (bullet list). Do NOT auto-execute it.
|
|
41
|
+
- Only create a formal plan (via create_plan tool) if user explicitly confirms.`,
|
|
42
|
+
maxIterations: 5,
|
|
43
|
+
},
|
|
44
|
+
|
|
45
|
+
files: {
|
|
46
|
+
name: "files",
|
|
47
|
+
intents: ["file_write", "file_edit", "file_manage"],
|
|
48
|
+
promptAddition: `[MODE: files]
|
|
49
|
+
- File operations: create, edit, move, copy, delete, search.
|
|
50
|
+
- Always verify the file exists before editing. Show relevant content after changes.
|
|
51
|
+
- For edits: read first, modify, write back. Show the diff or changed section.
|
|
52
|
+
- Respect scope: only touch files the user mentioned.`,
|
|
53
|
+
maxIterations: 8,
|
|
54
|
+
},
|
|
55
|
+
|
|
56
|
+
project: {
|
|
57
|
+
name: "project",
|
|
58
|
+
intents: ["complex_multi"],
|
|
59
|
+
promptAddition: `[MODE: project]
|
|
60
|
+
- Multi-step feature work, refactoring, or bug fixing.
|
|
61
|
+
- Plan before executing: list what you'll change and why.
|
|
62
|
+
- After changes: run tests if available. Report results.
|
|
63
|
+
- Commit-ready: changes should be minimal and focused.`,
|
|
64
|
+
maxIterations: 30,
|
|
65
|
+
},
|
|
66
|
+
|
|
67
|
+
autonomous: {
|
|
68
|
+
name: "autonomous",
|
|
69
|
+
intents: [],
|
|
70
|
+
promptAddition: `[MODE: autonomous]
|
|
71
|
+
- Long self-directed work. You have extended budget.
|
|
72
|
+
- Write intermediate progress reports every 5-10 iterations.
|
|
73
|
+
- Self-verify: when you think you're done, check once more.
|
|
74
|
+
- If stuck after 3 attempts on same approach — try a different strategy.`,
|
|
75
|
+
maxIterations: 50,
|
|
76
|
+
},
|
|
77
|
+
|
|
78
|
+
research: {
|
|
79
|
+
name: "research",
|
|
80
|
+
intents: ["web_search", "web_fetch"],
|
|
81
|
+
promptAddition: `[MODE: research]
|
|
82
|
+
- Gather information from web and local sources.
|
|
83
|
+
- Structure findings: summary first, then details.
|
|
84
|
+
- Cite sources when using web results.
|
|
85
|
+
- If multiple sources disagree — note the discrepancy.`,
|
|
86
|
+
maxIterations: 10,
|
|
87
|
+
},
|
|
88
|
+
|
|
89
|
+
computer_use: {
|
|
90
|
+
name: "computer_use",
|
|
91
|
+
intents: ["desktop"],
|
|
92
|
+
promptAddition: `[MODE: computer_use]
|
|
93
|
+
- Desktop GUI interaction via screenshot/click/type tools.
|
|
94
|
+
- MANDATORY workflow: screenshot → look (OCR) → click. Never click blind.
|
|
95
|
+
- Prefer keyboard shortcuts over mouse when possible.
|
|
96
|
+
- If Chrome is open: use Chrome semantics (page_read, page_map) before OCR.`,
|
|
97
|
+
maxIterations: 20,
|
|
98
|
+
},
|
|
99
|
+
|
|
100
|
+
delegation: {
|
|
101
|
+
name: "delegation",
|
|
102
|
+
intents: ["agent_spawn"],
|
|
103
|
+
promptAddition: `[MODE: delegation]
|
|
104
|
+
- Spawn child agents for parallel work.
|
|
105
|
+
- Give each agent a clear, specific task description.
|
|
106
|
+
- Wait for results. Aggregate and summarize for the user.
|
|
107
|
+
- Max 3 concurrent child agents unless user specifies more.`,
|
|
108
|
+
maxIterations: 10,
|
|
109
|
+
},
|
|
110
|
+
|
|
111
|
+
secretary: {
|
|
112
|
+
name: "secretary",
|
|
113
|
+
intents: [],
|
|
114
|
+
promptAddition: `[MODE: secretary]
|
|
115
|
+
- Communication tasks: email, messages, scheduling.
|
|
116
|
+
- Use formal tone for external communications.
|
|
117
|
+
- Always show draft before sending. Wait for user approval.
|
|
118
|
+
- Include all relevant context (who, what, when, where).`,
|
|
119
|
+
maxIterations: 5,
|
|
120
|
+
},
|
|
121
|
+
|
|
122
|
+
monitoring: {
|
|
123
|
+
name: "monitoring",
|
|
124
|
+
intents: [],
|
|
125
|
+
promptAddition: `[MODE: monitoring]
|
|
126
|
+
- Periodic checks and alerts.
|
|
127
|
+
- Run check → compare with previous state → report changes.
|
|
128
|
+
- Only alert on meaningful changes, not noise.
|
|
129
|
+
- Keep logs of what was checked and when.`,
|
|
130
|
+
maxIterations: 5,
|
|
131
|
+
},
|
|
132
|
+
|
|
133
|
+
creative: {
|
|
134
|
+
name: "creative",
|
|
135
|
+
intents: ["creative_text"],
|
|
136
|
+
promptAddition: `[MODE: creative]
|
|
137
|
+
- Content generation: writing, editing, formatting.
|
|
138
|
+
- Longer output is acceptable. Use markdown formatting.
|
|
139
|
+
- If user asks to save result to file — use write_file.`,
|
|
140
|
+
maxIterations: 5,
|
|
141
|
+
},
|
|
142
|
+
};
|
|
143
|
+
|
|
144
|
+
// Build reverse map: intent → mode
|
|
145
|
+
const _intentToMode = new Map();
|
|
146
|
+
for (const [modeName, mode] of Object.entries(MODES)) {
|
|
147
|
+
for (const intent of mode.intents) {
|
|
148
|
+
_intentToMode.set(intent, modeName);
|
|
149
|
+
}
|
|
150
|
+
}
|
|
151
|
+
|
|
152
|
+
/**
|
|
153
|
+
* Get mode config for a given intent class.
|
|
154
|
+
* Falls back to "project" for unknown intents (safest default).
|
|
155
|
+
* Resolves model: env AGENT_MODEL_<MODE> > mode.model > null (use session default).
|
|
156
|
+
* @param {string} intentClass — from intent classifier
|
|
157
|
+
* @returns {{ name, promptAddition, maxIterations, model }}
|
|
158
|
+
*/
|
|
159
|
+
export function getModeForIntent(intentClass) {
|
|
160
|
+
const modeName = _intentToMode.get(intentClass) || "project";
|
|
161
|
+
const mode = MODES[modeName];
|
|
162
|
+
// Resolve model: env var takes priority
|
|
163
|
+
const envKey = `AGENT_MODEL_${modeName.toUpperCase()}`;
|
|
164
|
+
const resolvedModel = process.env[envKey] || mode.model || null;
|
|
165
|
+
return { ...mode, model: resolvedModel };
|
|
166
|
+
}
|
|
167
|
+
|
|
168
|
+
/**
|
|
169
|
+
* Get mode by name directly.
|
|
170
|
+
* @param {string} modeName
|
|
171
|
+
* @returns {object|null}
|
|
172
|
+
*/
|
|
173
|
+
export function getMode(modeName) {
|
|
174
|
+
return MODES[modeName] || null;
|
|
175
|
+
}
|
|
176
|
+
|
|
177
|
+
/**
|
|
178
|
+
* List all available modes (for /help or tool listing).
|
|
179
|
+
* @returns {Array<{name, description}>}
|
|
180
|
+
*/
|
|
181
|
+
export function listModes() {
|
|
182
|
+
return Object.values(MODES).map(m => ({
|
|
183
|
+
name: m.name,
|
|
184
|
+
intents: m.intents,
|
|
185
|
+
maxIterations: m.maxIterations,
|
|
186
|
+
}));
|
|
187
|
+
}
|
|
@@ -0,0 +1,91 @@
|
|
|
1
|
+
// Which of the three things happened on a turn that changed nothing.
|
|
2
|
+
//
|
|
3
|
+
// A turn that changed nothing must tell the operator WHICH: the agent did not find what to
|
|
4
|
+
// change, or found it and could not apply the change, or there was nothing to
|
|
5
|
+
// change. Only the model that did the turn can say, so it is asked.
|
|
6
|
+
//
|
|
7
|
+
// It used to be asked inside the conversation: a [OUTCOME] system message, then
|
|
8
|
+
// `continue`, and the loop took the next reply as the turn's answer. The
|
|
9
|
+
// question says "in one sentence", the model obeys, and the one sentence
|
|
10
|
+
// REPLACED the answer. Measured on ten readiness probes, 2026-09-21: eight
|
|
11
|
+
// turns of ten came back to the operator as a single sentence about change,
|
|
12
|
+
// their actual answer lost inside the loop. cut-0003 asked "go to the server",
|
|
13
|
+
// ran a command, and the whole answer was "The request was not asking for a
|
|
14
|
+
// change."
|
|
15
|
+
//
|
|
16
|
+
// So the question is not a turn of the conversation. It is one small call of
|
|
17
|
+
// its own, like the classifier, and its answer is a line of explanation, never
|
|
18
|
+
// the answer itself. That also makes it cheaper: the in-conversation ask
|
|
19
|
+
// re-sent the whole context and the tool schemas, around 9k prompt tokens on
|
|
20
|
+
// the measured run; this sends the evidence and nothing else.
|
|
21
|
+
|
|
22
|
+
import { config } from "../config.js";
|
|
23
|
+
import { chatCompletion } from "../api/client.js";
|
|
24
|
+
import { createLogger } from "../logging/logger.js";
|
|
25
|
+
|
|
26
|
+
const log = createLogger("outcome-ask");
|
|
27
|
+
|
|
28
|
+
const SYSTEM = `You are reporting on one turn of an agent that finished without changing any file on disk.
|
|
29
|
+
|
|
30
|
+
Say, in ONE short sentence addressed to the operator, which of these is true:
|
|
31
|
+
- you did not find what to change;
|
|
32
|
+
- you found what to change but did not apply the change;
|
|
33
|
+
- there was nothing to change;
|
|
34
|
+
- the request was not asking for a change.
|
|
35
|
+
|
|
36
|
+
Reply with that sentence and nothing else. No preamble, no JSON, no quotes.`;
|
|
37
|
+
|
|
38
|
+
/**
|
|
39
|
+
* Ask why the turn changed nothing.
|
|
40
|
+
*
|
|
41
|
+
* @param {object} turn
|
|
42
|
+
* @param {string} turn.request what the user asked for
|
|
43
|
+
* @param {string} turn.answer what the agent replied
|
|
44
|
+
* @param {string[]} turn.tools names of the tools the turn called, in order
|
|
45
|
+
* @param {AbortSignal} [turn.signal]
|
|
46
|
+
* @returns {Promise<string|null>} one sentence, or null when the call did not happen
|
|
47
|
+
*/
|
|
48
|
+
export async function askOutcome({ request, answer, tools = [], signal }) {
|
|
49
|
+
if (!config.apiKey) return null;
|
|
50
|
+
|
|
51
|
+
const evidence = [
|
|
52
|
+
`REQUEST: ${String(request || "").slice(0, 1500)}`,
|
|
53
|
+
`YOUR ANSWER: ${String(answer || "").slice(0, 1500)}`,
|
|
54
|
+
`TOOLS YOU CALLED: ${tools.length ? tools.join(", ") : "(none)"}`,
|
|
55
|
+
"NOTHING ON DISK CHANGED.",
|
|
56
|
+
].join("\n\n");
|
|
57
|
+
|
|
58
|
+
try {
|
|
59
|
+
const { message } = await chatCompletion(
|
|
60
|
+
[
|
|
61
|
+
{ role: "system", content: SYSTEM },
|
|
62
|
+
{ role: "user", content: evidence },
|
|
63
|
+
],
|
|
64
|
+
[],
|
|
65
|
+
null,
|
|
66
|
+
{
|
|
67
|
+
source: "outcome",
|
|
68
|
+
model: config.model,
|
|
69
|
+
maxTokens: 80,
|
|
70
|
+
temperature: 0,
|
|
71
|
+
stream: false,
|
|
72
|
+
signal,
|
|
73
|
+
// A ceiling of its own, always: the turn's signal cancels the ask when
|
|
74
|
+
// the user aborts, but it never expires on its own, and a hung reason
|
|
75
|
+
// must not hold back the answer that is already written.
|
|
76
|
+
timeoutMs: 15000,
|
|
77
|
+
},
|
|
78
|
+
);
|
|
79
|
+
const text = (message?.content || "").trim();
|
|
80
|
+
if (!text) return null;
|
|
81
|
+
// One sentence is what was asked for; a model that writes three gets cut,
|
|
82
|
+
// because this line sits next to the answer and must not compete with it.
|
|
83
|
+
return text.split("\n")[0].slice(0, 300);
|
|
84
|
+
} catch (err) {
|
|
85
|
+
// The reason is best effort. The FACT that nothing changed is not: the
|
|
86
|
+
// caller states that either way. Losing the
|
|
87
|
+
// reason must never cost the operator the fact.
|
|
88
|
+
log.warn("outcome-ask failed", { error: err.message });
|
|
89
|
+
return null;
|
|
90
|
+
}
|
|
91
|
+
}
|
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
// Project context (FLINT.md) — the agent's own notebook for the project
|
|
2
|
+
// it's currently working in. Walks up from CWD looking for FLINT.md,
|
|
3
|
+
// scans for prompt injection, returns content. Cached per-CWD for the
|
|
4
|
+
// life of the process (invalidated by file mtime change).
|
|
5
|
+
//
|
|
6
|
+
// Injected at the top of EVERY LLM prompt — classifier, agent loop,
|
|
7
|
+
// reflection, etc. — so the agent's own project rules are visible
|
|
8
|
+
// before any other context.
|
|
9
|
+
|
|
10
|
+
import { existsSync, readFileSync, statSync } from "node:fs";
|
|
11
|
+
import { join } from "node:path";
|
|
12
|
+
import { config } from "../config.js";
|
|
13
|
+
import { detectInjection } from "../security/content-fence.js";
|
|
14
|
+
|
|
15
|
+
const MAX_CHARS = config.maxContextChars;
|
|
16
|
+
const _cache = new Map(); // path → { content, mtimeMs }
|
|
17
|
+
|
|
18
|
+
function findFile(startDir) {
|
|
19
|
+
let dir = startDir;
|
|
20
|
+
for (let i = 0; i < 10; i++) {
|
|
21
|
+
const p = join(dir, "FLINT.md");
|
|
22
|
+
if (existsSync(p)) return p;
|
|
23
|
+
const parent = join(dir, "..");
|
|
24
|
+
if (parent === dir) break;
|
|
25
|
+
dir = parent;
|
|
26
|
+
}
|
|
27
|
+
return null;
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
export function loadFlintMd(cwd = process.cwd()) {
|
|
31
|
+
const p = findFile(cwd);
|
|
32
|
+
if (!p) return null;
|
|
33
|
+
|
|
34
|
+
let mtimeMs;
|
|
35
|
+
try {
|
|
36
|
+
mtimeMs = statSync(p).mtimeMs;
|
|
37
|
+
} catch {
|
|
38
|
+
return null;
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
const cached = _cache.get(p);
|
|
42
|
+
if (cached && cached.mtimeMs === mtimeMs) return cached.content;
|
|
43
|
+
|
|
44
|
+
let content;
|
|
45
|
+
try {
|
|
46
|
+
content = readFileSync(p, "utf-8");
|
|
47
|
+
} catch {
|
|
48
|
+
return null;
|
|
49
|
+
}
|
|
50
|
+
if (content.length > MAX_CHARS) {
|
|
51
|
+
content = content.slice(0, MAX_CHARS) + "\n... (truncated)";
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
// Injection scan — FLINT.md is user- or agent-written, but we still
|
|
55
|
+
// scan because the agent itself could be fooled into writing malicious
|
|
56
|
+
// instructions there via a crafted user message in a prior turn.
|
|
57
|
+
const scan = detectInjection(content);
|
|
58
|
+
if (scan.detected && scan.score >= 3) {
|
|
59
|
+
const blocked = `[BLOCKED: FLINT.md at ${p} failed injection scan (score ${scan.score})]`;
|
|
60
|
+
_cache.set(p, { content: blocked, mtimeMs });
|
|
61
|
+
return blocked;
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
_cache.set(p, { content, mtimeMs });
|
|
65
|
+
return content;
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
export function formatFlintMdBlock(cwd = process.cwd()) {
|
|
69
|
+
const content = loadFlintMd(cwd);
|
|
70
|
+
if (!content) return "";
|
|
71
|
+
return [
|
|
72
|
+
"<trusted-context name=\"project-rules\" source=\"FLINT.md\">",
|
|
73
|
+
content,
|
|
74
|
+
"</trusted-context>",
|
|
75
|
+
].join("\n");
|
|
76
|
+
}
|