flint-agent 1.14.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (171) hide show
  1. package/.env.example +108 -0
  2. package/CHANGELOG.md +55 -0
  3. package/FEATURES.md +298 -0
  4. package/LICENSE +21 -0
  5. package/README.md +435 -0
  6. package/bin/flint.js +47 -0
  7. package/config/classifier-prompt.md +218 -0
  8. package/config/models-curated.json +4 -0
  9. package/config/providers.json +74 -0
  10. package/package.json +92 -0
  11. package/patches/ink+6.8.0.patch +78 -0
  12. package/profiles/desktop.md +65 -0
  13. package/profiles/generic.md +20 -0
  14. package/profiles/marketer.md +20 -0
  15. package/profiles/profiles.json +34 -0
  16. package/profiles/ux-reviewer.md +25 -0
  17. package/src/agent/agent.js +1743 -0
  18. package/src/agent/auto.js +346 -0
  19. package/src/agent/backoff.js +143 -0
  20. package/src/agent/compression.js +310 -0
  21. package/src/agent/content-resolver.js +180 -0
  22. package/src/agent/flow-controller.js +309 -0
  23. package/src/agent/intent-manifest.js +231 -0
  24. package/src/agent/intent-timeout.js +46 -0
  25. package/src/agent/intent.js +633 -0
  26. package/src/agent/knowledge.js +114 -0
  27. package/src/agent/learning.js +180 -0
  28. package/src/agent/modes.js +187 -0
  29. package/src/agent/outcome-ask.js +91 -0
  30. package/src/agent/project-context.js +76 -0
  31. package/src/agent/prompt-budget.js +117 -0
  32. package/src/agent/reflection-extractor.js +140 -0
  33. package/src/agent/steering.js +86 -0
  34. package/src/agent/supervisor.js +430 -0
  35. package/src/agent/swap.js +443 -0
  36. package/src/agent/system-prompt.js +446 -0
  37. package/src/agent/time-stamp.js +48 -0
  38. package/src/agent/tool-guard.js +201 -0
  39. package/src/agent/toolcall-text.js +162 -0
  40. package/src/agent/usage.js +297 -0
  41. package/src/agent/vision.js +94 -0
  42. package/src/agent/watchdog.js +139 -0
  43. package/src/agent/workspace-changes.js +177 -0
  44. package/src/api/address.js +14 -0
  45. package/src/api/client.js +280 -0
  46. package/src/api/server.js +535 -0
  47. package/src/api/stream-pipe.js +113 -0
  48. package/src/app-state.js +39 -0
  49. package/src/bootstrap.js +501 -0
  50. package/src/bus/drain-loop.js +497 -0
  51. package/src/bus/index.js +270 -0
  52. package/src/bus/plugins.js +65 -0
  53. package/src/child-idle.js +14 -0
  54. package/src/cli.js +118 -0
  55. package/src/commands/commands.js +1297 -0
  56. package/src/commands/registry.js +132 -0
  57. package/src/components/App.js +491 -0
  58. package/src/components/CarefulMenu.js +145 -0
  59. package/src/components/HistoryWriter.js +86 -0
  60. package/src/components/LineInput.js +69 -0
  61. package/src/components/LiveZone.js +294 -0
  62. package/src/components/OverlayMenu.js +179 -0
  63. package/src/components/SystemPanel.js +156 -0
  64. package/src/components/Table.js +54 -0
  65. package/src/config.js +249 -0
  66. package/src/free-models.js +230 -0
  67. package/src/index.js +1111 -0
  68. package/src/input-handler.js +13 -0
  69. package/src/input-text.js +123 -0
  70. package/src/launcher.js +129 -0
  71. package/src/logging/api-log.js +95 -0
  72. package/src/logging/chat-log-follower.js +113 -0
  73. package/src/logging/chat-log.js +15 -0
  74. package/src/logging/log-collector.js +182 -0
  75. package/src/logging/logger.js +112 -0
  76. package/src/logging/tool-log.js +20 -0
  77. package/src/mcp-client.js +314 -0
  78. package/src/memory/conversation-digest.js +113 -0
  79. package/src/memory/extract-facts.js +98 -0
  80. package/src/memory/facts.js +181 -0
  81. package/src/memory/inbox.js +63 -0
  82. package/src/memory/markdown.js +38 -0
  83. package/src/memory/patterns.js +185 -0
  84. package/src/memory/project.js +66 -0
  85. package/src/memory/reflections.js +74 -0
  86. package/src/memory/retrieval.js +84 -0
  87. package/src/memory/rules.js +105 -0
  88. package/src/memory/session-facts.js +125 -0
  89. package/src/memory/skills.js +191 -0
  90. package/src/memory/sqlite-store.js +653 -0
  91. package/src/memory/store.js +208 -0
  92. package/src/memory/tools.js +196 -0
  93. package/src/memory/user-model.js +86 -0
  94. package/src/message-handler.js +775 -0
  95. package/src/model-check.js +218 -0
  96. package/src/plugins/loader.js +120 -0
  97. package/src/plugins/manager.js +88 -0
  98. package/src/production-env.js +22 -0
  99. package/src/profiles.js +42 -0
  100. package/src/providers/adapters/anthropic.js +270 -0
  101. package/src/providers/adapters/openai.js +120 -0
  102. package/src/providers/keys-dpapi.js +41 -0
  103. package/src/providers/keys-fallback.js +31 -0
  104. package/src/providers/keys.js +132 -0
  105. package/src/providers/models.js +154 -0
  106. package/src/providers/registry.js +56 -0
  107. package/src/providers/state.js +56 -0
  108. package/src/registry.js +96 -0
  109. package/src/restart.js +29 -0
  110. package/src/sandbox/backend.js +130 -0
  111. package/src/security/api-auth.js +132 -0
  112. package/src/security/audit.js +98 -0
  113. package/src/security/child-policy.js +41 -0
  114. package/src/security/command-guard.js +173 -0
  115. package/src/security/content-fence.js +250 -0
  116. package/src/security/content-validator.js +132 -0
  117. package/src/security/index.js +143 -0
  118. package/src/security/network-guard.js +126 -0
  119. package/src/security/pairing.js +180 -0
  120. package/src/security/path-guard.js +140 -0
  121. package/src/security/persona-guard.js +67 -0
  122. package/src/security/policies.js +452 -0
  123. package/src/security/safety-constants.js +34 -0
  124. package/src/security/watchdog.js +107 -0
  125. package/src/sessions.js +130 -0
  126. package/src/spend.js +97 -0
  127. package/src/startup-watchdog.js +59 -0
  128. package/src/stdio/args.js +71 -0
  129. package/src/stdio/guard.js +59 -0
  130. package/src/stdio/protocol.js +167 -0
  131. package/src/stdio/run.js +106 -0
  132. package/src/stdio/session.js +180 -0
  133. package/src/store/agent-slice.js +306 -0
  134. package/src/store/dataset-slice.js +73 -0
  135. package/src/store/index.js +22 -0
  136. package/src/store/process-slice.js +135 -0
  137. package/src/store/session-slice.js +191 -0
  138. package/src/store/ui-slice.js +119 -0
  139. package/src/tasks/db.js +184 -0
  140. package/src/tasks/queries.js +589 -0
  141. package/src/tools/agent-tools.js +473 -0
  142. package/src/tools/checkpoint.js +152 -0
  143. package/src/tools/command-approvals.js +180 -0
  144. package/src/tools/dataset.js +50 -0
  145. package/src/tools/filesystem.js +682 -0
  146. package/src/tools/inbox-tools.js +48 -0
  147. package/src/tools/mesh.js +135 -0
  148. package/src/tools/own-env.js +136 -0
  149. package/src/tools/permissions.js +681 -0
  150. package/src/tools/plugin-tools.js +123 -0
  151. package/src/tools/process-tools.js +595 -0
  152. package/src/tools/registry.js +307 -0
  153. package/src/tools/swap-tools.js +72 -0
  154. package/src/tools/system.js +662 -0
  155. package/src/tools/tasks.js +532 -0
  156. package/src/tools/tool-search.js +171 -0
  157. package/src/ui/header.js +140 -0
  158. package/src/ui/input-cursor.js +23 -0
  159. package/src/ui/last-line.js +25 -0
  160. package/src/ui/line-edit.js +135 -0
  161. package/src/ui/output.js +399 -0
  162. package/src/ui/paste-tokens.js +131 -0
  163. package/src/ui/prompt-attention.js +134 -0
  164. package/src/ui/render-options.js +13 -0
  165. package/src/ui/replay.js +94 -0
  166. package/src/ui/splash.js +49 -0
  167. package/src/ui/status-level.js +36 -0
  168. package/src/ui/tool-ledger.js +203 -0
  169. package/src/ui/window-title.js +150 -0
  170. package/src/update.js +205 -0
  171. package/system.md +63 -0
@@ -0,0 +1,117 @@
1
+ // Prompt Budget Allocator — waterfall allocation with min/max per section.
2
+ //
3
+ // Each section declares priority, min tokens, max tokens.
4
+ // Allocator gives min to all sections first, then distributes remaining
5
+ // budget up to max in priority order. Conversation history gets the remainder.
6
+ //
7
+ // Token estimation: chars / 4 (rough but fast, no tiktoken dependency).
8
+
9
+ import { config } from "../config.js";
10
+
11
+ const DEFAULT_MAX_PROMPT_TOKENS = 100_000;
12
+ const CHARS_PER_TOKEN = 4; // rough approximation
13
+
14
+ /**
15
+ * Estimate token count from string content.
16
+ */
17
+ export function estimateTokens(text) {
18
+ if (!text) return 0;
19
+ return Math.ceil(text.length / CHARS_PER_TOKEN);
20
+ }
21
+
22
+ /**
23
+ * Create a budget allocator instance.
24
+ * @param {number} [maxTokens] - Total prompt budget in tokens
25
+ */
26
+ export function createBudgetAllocator(maxTokens) {
27
+ const budget = maxTokens || config.maxPromptTokens || DEFAULT_MAX_PROMPT_TOKENS;
28
+ const sections = [];
29
+
30
+ return {
31
+ /**
32
+ * Add a named section with content and budget constraints.
33
+ * @param {string} name - Section identifier
34
+ * @param {string} content - Section text content
35
+ * @param {object} opts
36
+ * @param {number} opts.priority - Lower = higher priority (1 = must-have, 10 = nice-to-have)
37
+ * @param {number} [opts.min] - Minimum tokens (guaranteed if content available)
38
+ * @param {number} [opts.max] - Maximum tokens (cap even if budget allows more)
39
+ * @param {boolean} [opts.fixed] - If true, include as-is without truncation
40
+ */
41
+ addSection(name, content, { priority, min = 0, max = Infinity, fixed = false } = {}) {
42
+ if (!content) return;
43
+ const tokens = estimateTokens(content);
44
+ sections.push({ name, content, tokens, priority, min, max, fixed });
45
+ },
46
+
47
+ /**
48
+ * Build the final prompt string respecting the budget.
49
+ * Returns { prompt, stats } where stats shows per-section token usage.
50
+ */
51
+ build() {
52
+ if (!sections.length) return { prompt: "", stats: {} };
53
+
54
+ // Sort by priority (lower = higher priority)
55
+ const sorted = [...sections].sort((a, b) => a.priority - b.priority);
56
+
57
+ // Phase 1: give minimum to all sections
58
+ let used = 0;
59
+ const allocations = new Map();
60
+
61
+ for (const s of sorted) {
62
+ if (s.fixed) {
63
+ allocations.set(s.name, s.tokens);
64
+ used += s.tokens;
65
+ } else {
66
+ const minAlloc = Math.min(s.min, s.tokens); // don't allocate more than content
67
+ allocations.set(s.name, minAlloc);
68
+ used += minAlloc;
69
+ }
70
+ }
71
+
72
+ // Phase 2: distribute remaining budget up to max, in priority order
73
+ let remaining = budget - used;
74
+ for (const s of sorted) {
75
+ if (s.fixed || remaining <= 0) continue;
76
+ const current = allocations.get(s.name);
77
+ const want = Math.min(s.tokens, s.max) - current;
78
+ if (want <= 0) continue;
79
+ const give = Math.min(want, remaining);
80
+ allocations.set(s.name, current + give);
81
+ remaining -= give;
82
+ }
83
+
84
+ // Phase 3: build output, truncating sections to their allocation
85
+ const parts = [];
86
+ const stats = {};
87
+
88
+ // Maintain original insertion order
89
+ for (const s of sections) {
90
+ const allocated = allocations.get(s.name) || 0;
91
+ stats[s.name] = { tokens: s.tokens, allocated, truncated: false };
92
+
93
+ if (s.fixed || s.tokens <= allocated) {
94
+ parts.push(s.content);
95
+ } else {
96
+ // Truncate: keep head (70%) + tail (20%) + marker
97
+ const maxChars = allocated * CHARS_PER_TOKEN;
98
+ if (maxChars <= 0) continue;
99
+ const headChars = Math.floor(maxChars * 0.7);
100
+ const tailChars = Math.floor(maxChars * 0.2);
101
+ const head = s.content.slice(0, headChars);
102
+ const tail = s.content.slice(-tailChars);
103
+ parts.push(`${head}\n\n[...truncated ${s.name}: ${s.tokens} tokens -> ${allocated} tokens...]\n\n${tail}`);
104
+ stats[s.name].truncated = true;
105
+ }
106
+ }
107
+
108
+ return {
109
+ prompt: parts.join("\n\n"),
110
+ stats,
111
+ budget,
112
+ used: [...allocations.values()].reduce((a, b) => a + b, 0),
113
+ remaining: budget - [...allocations.values()].reduce((a, b) => a + b, 0),
114
+ };
115
+ },
116
+ };
117
+ }
@@ -0,0 +1,140 @@
1
+ // Extract a reflection from a session transcript using an LLM call.
2
+ //
3
+ // Prompts the model with 3 questions — what did I do, what did I do wrong,
4
+ // how can I do better — and parses the structured response.
5
+ //
6
+ // Part of Layer 1 memory architecture.
7
+
8
+ import { chatCompletion } from "../api/client.js";
9
+ import { appendReflection } from "../memory/reflections.js";
10
+ import { createLogger } from "../logging/logger.js";
11
+
12
+ const log = createLogger("reflection");
13
+
14
+ const EXTRACTION_SYSTEM = `You are a self-reflection analyzer for an AI agent.
15
+
16
+ Given a session transcript, answer three questions about the AGENT'S behavior (not the user's):
17
+
18
+ 1. DID — What did the agent do this session? One concise sentence.
19
+ 2. WRONG — What did the agent do wrong or suboptimally? Up to 5 bullet points. Each must be CONCRETE and ACTIONABLE (e.g. "used python3 on Windows where it's unavailable" not "had trouble with tools"). If nothing went wrong, use empty list.
20
+ 3. BETTER — How should the agent behave next time to avoid those mistakes? Up to 5 bullet points. Each must be a specific actionable rule the agent can follow next time.
21
+
22
+ Output ONLY valid JSON matching this schema, no other text:
23
+ {"did": "one sentence", "wrong": ["point1", "point2"], "better": ["rule1", "rule2"], "tags": ["tag1", "tag2"]}
24
+
25
+ Tags are short keywords (e.g. "verbosity", "tool-choice", "file-ops", "consistency", "windows-paths").
26
+
27
+ If the session was trivial or empty (no real agent action), output:
28
+ {"did": "", "wrong": [], "better": [], "tags": []}
29
+ — this will be discarded.`;
30
+
31
+ /**
32
+ * Extract a reflection from a session transcript.
33
+ * @param {Array} messages — array of {role, content} message objects
34
+ * @param {string} sessionId — session id to tag the reflection with
35
+ * @param {object} opts — { signal?, maxTranscriptChars? }
36
+ * @returns {Promise<string|null>} id of appended reflection, or null if discarded/failed
37
+ */
38
+ export async function extractAndStoreReflection(messages, sessionId, opts = {}) {
39
+ const maxChars = opts.maxTranscriptChars || 12000;
40
+ const transcript = buildTranscript(messages, maxChars);
41
+
42
+ // Skip if session too short to be meaningful
43
+ if (transcript.length < 100) {
44
+ log.debug("reflection-skipped", { reason: "transcript too short", sessionId, chars: transcript.length });
45
+ return null;
46
+ }
47
+
48
+ try {
49
+ const extractionMessages = [
50
+ { role: "system", content: EXTRACTION_SYSTEM },
51
+ { role: "user", content: `Session transcript:\n\n${transcript}\n\nExtract the reflection. JSON only, no preamble.` },
52
+ ];
53
+
54
+ const result = await chatCompletion(extractionMessages, [], null, {
55
+ signal: opts.signal,
56
+ source: "extractor",
57
+ });
58
+ const text = result?.text || result?.message?.content || "";
59
+
60
+ const parsed = parseReflectionJson(text);
61
+ if (!parsed) {
62
+ log.warn("reflection-parse-failed", { sessionId, rawLen: text.length });
63
+ return null;
64
+ }
65
+
66
+ // Discard if empty (model decided nothing to reflect on)
67
+ if (!parsed.did && parsed.wrong.length === 0 && parsed.better.length === 0) {
68
+ log.info("reflection-empty", { sessionId });
69
+ return null;
70
+ }
71
+
72
+ const id = appendReflection({
73
+ session_id: sessionId,
74
+ did: parsed.did,
75
+ wrong: parsed.wrong,
76
+ better: parsed.better,
77
+ tags: parsed.tags,
78
+ });
79
+ log.info("reflection-stored", { sessionId, id, wrong: parsed.wrong.length, better: parsed.better.length });
80
+ return id;
81
+ } catch (e) {
82
+ log.warn("reflection-failed", { sessionId, error: e.message });
83
+ return null;
84
+ }
85
+ }
86
+
87
+ /**
88
+ * Build a compact transcript from the message array. Strips system messages,
89
+ * trims tool outputs, caps total length.
90
+ */
91
+ function buildTranscript(messages, maxChars) {
92
+ const out = [];
93
+ for (const m of messages || []) {
94
+ if (m.role === "system") continue;
95
+ const content = typeof m.content === "string" ? m.content : JSON.stringify(m.content || "");
96
+ if (!content.trim()) continue;
97
+ // Trim long tool results
98
+ const trimmed = content.length > 500 ? content.slice(0, 400) + "... [truncated]" : content;
99
+ const tool_calls = Array.isArray(m.tool_calls) ? m.tool_calls.map(t => t.function?.name || t.name).filter(Boolean) : [];
100
+ const prefix = tool_calls.length ? `[${m.role}, used: ${tool_calls.join(",")}]` : `[${m.role}]`;
101
+ out.push(`${prefix} ${trimmed}`);
102
+ }
103
+ let transcript = out.join("\n");
104
+ if (transcript.length > maxChars) {
105
+ // Keep head and tail — middle of long sessions is least interesting
106
+ const head = transcript.slice(0, Math.floor(maxChars * 0.4));
107
+ const tail = transcript.slice(-Math.floor(maxChars * 0.5));
108
+ transcript = head + "\n\n... [middle truncated] ...\n\n" + tail;
109
+ }
110
+ return transcript;
111
+ }
112
+
113
+ /**
114
+ * Extract the JSON object from a model response that may have extra text around it.
115
+ */
116
+ function parseReflectionJson(text) {
117
+ if (!text) return null;
118
+ // Try direct parse
119
+ try {
120
+ return normalize(JSON.parse(text.trim()));
121
+ } catch {}
122
+ // Find first { ... } block
123
+ const m = text.match(/\{[\s\S]*\}/);
124
+ if (!m) return null;
125
+ try {
126
+ return normalize(JSON.parse(m[0]));
127
+ } catch {
128
+ return null;
129
+ }
130
+ }
131
+
132
+ function normalize(obj) {
133
+ if (!obj || typeof obj !== "object") return null;
134
+ return {
135
+ did: typeof obj.did === "string" ? obj.did : "",
136
+ wrong: Array.isArray(obj.wrong) ? obj.wrong.map(String).filter(s => s.trim()) : [],
137
+ better: Array.isArray(obj.better) ? obj.better.map(String).filter(s => s.trim()) : [],
138
+ tags: Array.isArray(obj.tags) ? obj.tags.map(String).filter(s => s.trim()) : [],
139
+ };
140
+ }
@@ -0,0 +1,86 @@
1
+ // Steering that applies to the NEXT completion only.
2
+ //
3
+ // Before this, twelve places in agent.js pushed a system message straight into
4
+ // the conversation array, and nothing ever took one out. The array is also the
5
+ // session and the source of the rolling context window, so every nudge became a
6
+ // standing instruction: it survived the turn that raised it, crossed into the
7
+ // next turn, and stacked with copies of itself.
8
+ //
9
+ // Measured on ten readiness probes, 2026-09-21, from the payloads in
10
+ // sessions/*.messages.jsonl:
11
+ //
12
+ // [OUTCOME] raised 8 times sent to the model 202 times
13
+ // [SAFETY] sent to the model 192 times
14
+ // most system messages in one call 20
15
+ // most copies of one nudge in one call 3
16
+ // leftover nudges present at the start of a turn 6
17
+ //
18
+ // What that did to answers: three standing orders to answer "in one
19
+ // sentence" and the answer collapses to one sentence.
20
+ //
21
+ // A nudge now lives in this slot instead. It is assembled into ONE system
22
+ // message appended to the payload of the next completion, and dropped. It never
23
+ // touches the conversation, so it cannot be saved, cannot be re-sent and cannot
24
+ // pile up.
25
+
26
+ /**
27
+ * Order the nudges are read in, most urgent first. Not a filter: everything
28
+ * raised in one iteration is sent in one message, because dropping a nudge
29
+ * silently is how a loop-breaker goes missing. Order is what resolves a
30
+ * disagreement between two of them.
31
+ */
32
+ export const STEER_KINDS = {
33
+ security: 100,
34
+ loop: 90,
35
+ // A call of the model's own that was malformed and not run: it has to hear
36
+ // that before anything about budget or style, or it waits for a result.
37
+ "tool-call": 85,
38
+ budget: 80,
39
+ verify: 70,
40
+ outcome: 65,
41
+ supervisor: 60,
42
+ safety: 50,
43
+ reflection: 40,
44
+ };
45
+
46
+ /**
47
+ * One slot per agent run.
48
+ *
49
+ * @returns {{add: (kind: string, text: string) => void, take: () => object|null, size: () => number}}
50
+ */
51
+ export function createSteering() {
52
+ /** @type {Array<{rank: number, seq: number, text: string}>} */
53
+ let pending = [];
54
+ let seq = 0;
55
+
56
+ return {
57
+ /**
58
+ * Raise a nudge for the next completion.
59
+ *
60
+ * An unknown kind throws rather than defaulting: a nudge with no rank would
61
+ * be ordered by accident, and silently taking the lowest rank is exactly the
62
+ * kind of fallback that hides a typo until someone reads a payload.
63
+ */
64
+ add(kind, text) {
65
+ const rank = STEER_KINDS[kind];
66
+ if (rank === undefined) throw new Error(`unknown steering kind: ${kind}`);
67
+ if (typeof text !== "string" || !text.trim()) return;
68
+ // The same line twice in one payload teaches the model nothing and was
69
+ // half of what the measurement found.
70
+ if (pending.some((p) => p.text === text)) return;
71
+ pending.push({ rank, seq: seq++, text });
72
+ },
73
+
74
+ /** Take everything raised since the last completion, as one message. Clears. */
75
+ take() {
76
+ if (!pending.length) return null;
77
+ const ordered = [...pending].sort((a, b) => b.rank - a.rank || a.seq - b.seq);
78
+ pending = [];
79
+ return { role: "system", content: ordered.map((p) => p.text).join("\n\n") };
80
+ },
81
+
82
+ size() {
83
+ return pending.length;
84
+ },
85
+ };
86
+ }