@miphamai/cli 0.14.0 → 0.15.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@miphamai/cli",
3
- "version": "0.14.0",
3
+ "version": "0.15.0",
4
4
  "description": "Mipham Code — Multi-model open-core intelligent coding terminal by MiphamAI",
5
5
  "keywords": [
6
6
  "ai",
@@ -4,6 +4,7 @@ import type { SubAgentType, SubAgentOptions, AgentDefinition } from './types'
4
4
  import { createAgentContext } from './agent-context'
5
5
  import { getBackgroundAgentRegistry } from './background-registry'
6
6
  import type { HookEngine } from '../core/hooks'
7
+ import type { PermissionSystem } from '../core/permission'
7
8
 
8
9
  const TYPE_SYSTEM_PROMPTS: Record<SubAgentType, string> = {
9
10
  general: 'You are a focused sub-agent. Complete the assigned task thoroughly and return results.',
@@ -27,6 +28,7 @@ export class SubAgent {
27
28
  constructor(
28
29
  private registry: ProviderRegistry,
29
30
  private toolRegistry: Map<string, ToolDefinition>,
31
+ private permission?: PermissionSystem,
30
32
  private hookEngine?: HookEngine,
31
33
  ) {}
32
34
 
@@ -113,6 +115,9 @@ export class SubAgent {
113
115
  const agentType = options.type || 'general'
114
116
  const agentDef = options.agentDef
115
117
 
118
+ // Resolve execution directory: worktree isolation or process cwd
119
+ const execCwd = options.worktreePath || process.cwd()
120
+
116
121
  // Resolve system prompt: agentDef > options.systemPrompt > builtin type
117
122
  const systemPrompt =
118
123
  agentDef?.systemPrompt || options.systemPrompt || TYPE_SYSTEM_PROMPTS[agentType]
@@ -229,9 +234,21 @@ export class SubAgent {
229
234
  continue
230
235
  }
231
236
 
237
+ // Security: check permission before executing
238
+ // Sub-agents run without user interaction — tools requiring approval are rejected
239
+ if (this.permission?.needsApproval(tool, tu.input)) {
240
+ currentMessages.push({
241
+ role: 'user' as const,
242
+ content:
243
+ `Tool "${tu.name}" requires user approval (permission: ask). ` +
244
+ `Cannot execute in non-interactive sub-agent context.`,
245
+ })
246
+ continue
247
+ }
248
+
232
249
  try {
233
250
  const result = await tool.execute(tu.input, {
234
- cwd: process.cwd(),
251
+ cwd: execCwd,
235
252
  sessionId: 'sub-agent',
236
253
  provider: '',
237
254
  model: resolvedModel,
@@ -44,4 +44,6 @@ export interface SubAgentOptions {
44
44
  runInBackground?: boolean
45
45
  /** Optional callback for streaming progress chunks during background execution. */
46
46
  onProgress?: (chunk: string) => void
47
+ /** When set, tool executions use this path as cwd (git worktree isolation). */
48
+ worktreePath?: string
47
49
  }
@@ -16,6 +16,7 @@ import type { AgentViewManager } from '../agent-view/agent-view-manager'
16
16
  import type { SkillsLoader } from '../skills/loader'
17
17
  import { getBackgroundAgentRegistry } from '../agent/background-registry'
18
18
  import { RulesLoader } from './rules-loader'
19
+ import { UsageTracker } from './usage-tracker'
19
20
  import { buildRequest, sendInferenceCheck, isInferenceHookEnabled } from './inference-hook'
20
21
 
21
22
  export class QueryEngine {
@@ -82,6 +83,7 @@ export class QueryEngine {
82
83
  private rulesLoader?: RulesLoader
83
84
  /** Files touched in the current turn (for rules matching). */
84
85
  private touchedFiles: Set<string> = new Set()
86
+ private usageTracker = new UsageTracker()
85
87
  /** Inference hook (DLP) configuration. */
86
88
  private inferenceHookConfig?: InferenceHookConfig
87
89
 
@@ -238,6 +240,10 @@ export class QueryEngine {
238
240
  return this.permission
239
241
  }
240
242
 
243
+ getUsageTracker(): UsageTracker {
244
+ return this.usageTracker
245
+ }
246
+
241
247
  async *process(userInput: string, signal?: AbortSignal): AsyncGenerator<StreamChunk> {
242
248
  // Fire UserPromptSubmit hooks before processing
243
249
  if (this.hookEngine) {
@@ -293,6 +299,8 @@ export class QueryEngine {
293
299
  let assistantContent = ''
294
300
  let reasoningContent = ''
295
301
  let thinkingContent = ''
302
+ let turnApiInputTokens = 0
303
+ let turnApiOutputTokens = 0
296
304
  const toolUses: Array<{ id: string; name: string; input: Record<string, unknown> }> = []
297
305
 
298
306
  // Stream model response
@@ -331,6 +339,12 @@ export class QueryEngine {
331
339
  })
332
340
  }
333
341
 
342
+ if (chunk.type === 'usage' && chunk.inputTokens !== undefined) {
343
+ // Accumulate API-reported token counts for this turn
344
+ turnApiInputTokens += chunk.inputTokens
345
+ turnApiOutputTokens += chunk.outputTokens || 0
346
+ }
347
+
334
348
  if (chunk.type === 'stop') {
335
349
  // Add assistant response to context
336
350
  if (assistantContent || reasoningContent || thinkingContent) {
@@ -417,6 +431,31 @@ export class QueryEngine {
417
431
  })
418
432
  }
419
433
 
434
+ // Record API token usage for this turn, attributed to executed tools
435
+ if (turnApiInputTokens > 0 || turnApiOutputTokens > 0) {
436
+ if (toolUses.length > 0) {
437
+ // Attribute tokens equally across all tools invoked this turn
438
+ const perTool = toolUses.length
439
+ for (const tu of toolUses) {
440
+ this.usageTracker.recordApiUsage(
441
+ Math.round(turnApiInputTokens / perTool),
442
+ Math.round(turnApiOutputTokens / perTool),
443
+ tu.name,
444
+ )
445
+ }
446
+ } else {
447
+ this.usageTracker.recordApiUsage(turnApiInputTokens, turnApiOutputTokens, 'chat')
448
+ }
449
+ } else {
450
+ // Fallback: API doesn't report usage — use char-based estimate
451
+ const estimated = Math.round((assistantContent.length + userInput.length) / 4)
452
+ if (toolUses.length > 0) {
453
+ for (const tu of toolUses) {
454
+ this.usageTracker.recordEstimatedUsage(Math.round(estimated / toolUses.length), tu.name)
455
+ }
456
+ }
457
+ }
458
+
420
459
  // Inject path-scoped rules for touched files
421
460
  this.injectRules()
422
461
 
@@ -721,6 +760,7 @@ export class QueryEngine {
721
760
  artifactServer: this.artifactServer,
722
761
  agentRegistry: this.agentRegistry,
723
762
  backgroundAgentRegistry: getBackgroundAgentRegistry(),
763
+ permissionSystem: this.permission,
724
764
  })
725
765
 
726
766
  // Track touched files for rules matching
@@ -223,6 +223,11 @@ export class PermissionSystem {
223
223
  case 'auto':
224
224
  // Safety checks handled by hook layer (PreToolUse hooks).
225
225
  // Bypass the static permission system so hooks are the sole gate.
226
+ // Exception: SendMessage always goes through the permission classifier
227
+ // so deny/allow rules are honored for cross-session messages.
228
+ if (tool.name === 'SendMessage') {
229
+ return 'mode-baseline'
230
+ }
226
231
  return 'bypass'
227
232
 
228
233
  case 'dontAsk':
@@ -0,0 +1,103 @@
1
+ /**
2
+ * UsageTracker — per-session token usage tracking with per-tool attribution.
3
+ *
4
+ * Tracks actual API-reported token counts (when available) and falls back
5
+ * to character-estimated counts. Token costs are attributed to the tool
6
+ * invoked in each turn — MCP tools (prefixed `mcp__`) get correct
7
+ * per-invocation attribution, not inflated cumulative costs.
8
+ */
9
+
10
+ export interface ToolUsage {
11
+ /** Total input tokens attributed to this tool. */
12
+ inputTokens: number
13
+ /** Total output tokens attributed to this tool. */
14
+ outputTokens: number
15
+ /** Number of times this tool was invoked. */
16
+ calls: number
17
+ }
18
+
19
+ export interface UsageSummary {
20
+ /** Total input tokens from API (0 if API data unavailable). */
21
+ apiInputTokens: number
22
+ /** Total output tokens from API (0 if API data unavailable). */
23
+ apiOutputTokens: number
24
+ /** Estimated total tokens (chars/4 heuristic, always available). */
25
+ estimatedTokens: number
26
+ /** Per-tool breakdown, keyed by tool name. "chat" = text-only turns. */
27
+ tools: Record<string, ToolUsage>
28
+ }
29
+
30
+ export class UsageTracker {
31
+ private _apiInputTokens = 0
32
+ private _apiOutputTokens = 0
33
+ private _estimatedTokens = 0
34
+ private _toolUsage = new Map<string, ToolUsage>()
35
+
36
+ /** Record API-reported token usage for a turn, attributed to the given tool. */
37
+ recordApiUsage(inputTokens: number, outputTokens: number, toolName?: string): void {
38
+ this._apiInputTokens += inputTokens
39
+ this._apiOutputTokens += outputTokens
40
+ this._recordTool(toolName || 'chat', inputTokens, outputTokens)
41
+ }
42
+
43
+ /** Record character-estimated token usage for a turn. */
44
+ recordEstimatedUsage(tokenCount: number, toolName?: string): void {
45
+ this._estimatedTokens += tokenCount
46
+ this._recordToolEstimated(toolName || 'chat', tokenCount)
47
+ }
48
+
49
+ /** Get a summary of all usage for the current session. */
50
+ getSummary(): UsageSummary {
51
+ const tools: Record<string, ToolUsage> = {}
52
+ for (const [name, usage] of this._toolUsage) {
53
+ tools[name] = { ...usage }
54
+ }
55
+ return {
56
+ apiInputTokens: this._apiInputTokens,
57
+ apiOutputTokens: this._apiOutputTokens,
58
+ estimatedTokens: this._estimatedTokens,
59
+ tools,
60
+ }
61
+ }
62
+
63
+ /** Reset all counters for a new session. */
64
+ reset(): void {
65
+ this._apiInputTokens = 0
66
+ this._apiOutputTokens = 0
67
+ this._estimatedTokens = 0
68
+ this._toolUsage.clear()
69
+ }
70
+
71
+ /** Total API tokens (input + output). */
72
+ get totalApiTokens(): number {
73
+ return this._apiInputTokens + this._apiOutputTokens
74
+ }
75
+
76
+ /** Whether any real API usage data has been recorded. */
77
+ get hasApiData(): boolean {
78
+ return this._apiInputTokens > 0 || this._apiOutputTokens > 0
79
+ }
80
+
81
+ // ── Private helpers ──
82
+
83
+ private _recordTool(name: string, inputTokens: number, outputTokens: number): void {
84
+ const existing = this._toolUsage.get(name)
85
+ if (existing) {
86
+ existing.inputTokens += inputTokens
87
+ existing.outputTokens += outputTokens
88
+ existing.calls++
89
+ } else {
90
+ this._toolUsage.set(name, { inputTokens, outputTokens, calls: 1 })
91
+ }
92
+ }
93
+
94
+ private _recordToolEstimated(name: string, tokenCount: number): void {
95
+ const existing = this._toolUsage.get(name)
96
+ if (existing) {
97
+ existing.inputTokens += tokenCount
98
+ existing.calls++
99
+ } else {
100
+ this._toolUsage.set(name, { inputTokens: tokenCount, outputTokens: 0, calls: 1 })
101
+ }
102
+ }
103
+ }
@@ -194,7 +194,15 @@ export class AnthropicProvider implements ProviderInstance {
194
194
  }
195
195
 
196
196
  case 'message_delta': {
197
- // Contains stop_reason and usage info
197
+ // Capture token usage for accurate cost tracking
198
+ if (event.usage) {
199
+ yield {
200
+ type: 'usage',
201
+ inputTokens: event.usage.input_tokens,
202
+ outputTokens: event.usage.output_tokens,
203
+ }
204
+ }
205
+ // Contains stop_reason; also handles late input_json_delta
198
206
  if (event.delta?.type === 'input_json_delta' && event.delta.partial_json) {
199
207
  accumulatedToolInput += event.delta.partial_json
200
208
  }
@@ -99,6 +99,16 @@ export class OpenAICompatProvider implements ProviderInstance {
99
99
  try {
100
100
  const parsed = JSON.parse(data)
101
101
  const choice = parsed.choices?.[0]
102
+
103
+ // Capture token usage when available (final chunk with stream_options.include_usage)
104
+ if (parsed.usage) {
105
+ yield {
106
+ type: 'usage',
107
+ inputTokens: parsed.usage.prompt_tokens,
108
+ outputTokens: parsed.usage.completion_tokens,
109
+ }
110
+ }
111
+
102
112
  if (!choice) continue
103
113
 
104
114
  const delta = choice.delta
@@ -81,6 +81,7 @@ export interface ToolContext {
81
81
  artifactServer?: import('../artifacts/server').ArtifactServer
82
82
  agentRegistry?: import('../agent/agent-registry').AgentRegistry
83
83
  backgroundAgentRegistry?: import('../agent/background-registry').BackgroundAgentRegistry
84
+ permissionSystem?: import('../core/permission').PermissionSystem
84
85
  }
85
86
 
86
87
  // ── Artifact Types ──
@@ -128,7 +129,15 @@ export interface TaskNotification {
128
129
 
129
130
  // ── Stream Types ──
130
131
  export interface StreamChunk {
131
- type: 'text' | 'tool_use' | 'tool_result' | 'thinking' | 'stop' | 'error' | 'task_notification'
132
+ type:
133
+ | 'text'
134
+ | 'tool_use'
135
+ | 'tool_result'
136
+ | 'thinking'
137
+ | 'stop'
138
+ | 'error'
139
+ | 'task_notification'
140
+ | 'usage'
132
141
  content?: string
133
142
  toolUse?: ToolUseContent
134
143
  tool_use_id?: string
@@ -139,6 +148,10 @@ export interface StreamChunk {
139
148
  thinking?: string
140
149
  /** Background task notification payload (type: 'task_notification'). */
141
150
  taskNotification?: TaskNotification
151
+ /** API-reported input token count (type: 'usage'). */
152
+ inputTokens?: number
153
+ /** API-reported output token count (type: 'usage'). */
154
+ outputTokens?: number
142
155
  }
143
156
 
144
157
  // ── Config Types ──
@@ -2,6 +2,7 @@ import { SubAgent } from '../agent/sub-agent'
2
2
  import type { ProviderRegistry } from '../providers/registry'
3
3
  import type { ToolDefinition, SkillDefinition } from '../shared/index.ts'
4
4
  import type { AgentDefinition } from '../agent/types'
5
+ import type { PermissionSystem } from '../core/permission'
5
6
 
6
7
  /**
7
8
  * Execute a skill in an isolated subagent context (context: fork).
@@ -15,6 +16,7 @@ export async function executeForkedSkill(
15
16
  args: string,
16
17
  registry: ProviderRegistry,
17
18
  toolRegistry: Map<string, ToolDefinition>,
19
+ permissionSystem?: PermissionSystem,
18
20
  ): Promise<string> {
19
21
  const agentDef: AgentDefinition = {
20
22
  name: `skill:${skill.name}`,
@@ -27,7 +29,7 @@ export async function executeForkedSkill(
27
29
  source: 'builtin',
28
30
  }
29
31
 
30
- const sub = new SubAgent(registry, toolRegistry)
32
+ const sub = new SubAgent(registry, toolRegistry, permissionSystem)
31
33
  const prompt = args
32
34
  ? `Execute the "${skill.name}" skill with arguments: ${args}`
33
35
  : `Execute the "${skill.name}" skill.`
@@ -60,7 +60,7 @@ export const agentTool: ToolDefinition = {
60
60
  const agentDef = ctx.agentRegistry?.resolve(agentType)
61
61
 
62
62
  try {
63
- const sub = new SubAgent(registry, toolRegistry)
63
+ const sub = new SubAgent(registry, toolRegistry, ctx.permissionSystem)
64
64
  const result = await sub.execute(prompt, description, {
65
65
  type: agentType,
66
66
  agentDef,
@@ -52,6 +52,7 @@ export const skillTool: ToolDefinition = {
52
52
  args,
53
53
  registry,
54
54
  ctx.toolRegistry || new Map(),
55
+ ctx.permissionSystem,
55
56
  )
56
57
  // Return to AI as internal context
57
58
  return { success: true, content: `[Forked skill "${skillName}" result]:\n${result}` }
@@ -689,22 +689,47 @@ const recapCmd: CommandHandler = (ctx) => {
689
689
 
690
690
  const usageCmd: CommandHandler = (ctx) => {
691
691
  const c = ctx.engine.getContext()
692
- const tokens = c.getEstimatedTokens()
692
+ const estTokens = c.getEstimatedTokens()
693
693
  const msgs = c.getMessages()
694
694
  const maxTokens = 200_000
695
- const pct = ((tokens / maxTokens) * 100).toFixed(1)
695
+ const pct = ((estTokens / maxTokens) * 100).toFixed(1)
696
+
697
+ const tracker = ctx.engine.getUsageTracker()
698
+ const summary = tracker.getSummary()
699
+
700
+ // Build per-tool breakdown
701
+ const toolLines: string[] = []
702
+ const sortedTools = Object.entries(summary.tools).sort(
703
+ (a, b) => b[1].inputTokens + b[1].outputTokens - (a[1].inputTokens + a[1].outputTokens),
704
+ )
705
+ for (const [name, usage] of sortedTools) {
706
+ const total = usage.inputTokens + usage.outputTokens
707
+ const prefix = name.startsWith('mcp__') ? '🔌 ' : ' '
708
+ toolLines.push(
709
+ `${prefix}${name.padEnd(18)} ${total.toLocaleString().padStart(8)} tokens (${usage.calls} call${usage.calls !== 1 ? 's' : ''})`,
710
+ )
711
+ }
712
+
713
+ const apiTotal = summary.apiInputTokens + summary.apiOutputTokens
714
+ const apiLine =
715
+ summary.apiInputTokens > 0 || summary.apiOutputTokens > 0
716
+ ? `API tokens: ${summary.apiInputTokens.toLocaleString().padStart(8)} in / ${summary.apiOutputTokens.toLocaleString().padStart(6)} out (${apiTotal.toLocaleString()} total)`
717
+ : 'API tokens: (no API usage data yet)'
718
+
719
+ const toolSection = toolLines.length > 0 ? `\n── Per-Tool ──\n${toolLines.join('\n')}` : ''
696
720
 
697
721
  return {
698
722
  content: stripIndent`
699
723
  ── Usage Dashboard ──
700
- Context tokens: ~${tokens.toLocaleString()} / ${maxTokens.toLocaleString()} (${pct}%)
724
+ ${apiLine}
725
+ Context tokens: ~${estTokens.toLocaleString()} / ${maxTokens.toLocaleString()} (${pct}%)
701
726
  Messages: ${msgs.length}
702
727
  Provider: ${ctx.providerId}
703
728
  Model: ${ctx.modelId}
704
729
 
705
730
  ${'█'.repeat(Math.ceil(Number(pct) / 5))}${'░'.repeat(20 - Math.ceil(Number(pct) / 5))} ${pct}%
731
+ ${toolSection}
706
732
 
707
- Note: Token counting is approximate (chars/4).
708
733
  Use /context for detailed stats, /compact to free space.
709
734
  `,
710
735
  }
@@ -4093,8 +4118,14 @@ const forkCmd: CommandHandler = async (ctx, args) => {
4093
4118
  'general',
4094
4119
  async (_signal) => {
4095
4120
  const { SubAgent } = await import('../agent/sub-agent')
4096
- const sa = new SubAgent(ctx.engine.getRegistry(), ctx.engine.getTools())
4097
- const result = await sa.execute(prompt, 'fork: ' + prompt.slice(0, 60), {})
4121
+ const sa = new SubAgent(
4122
+ ctx.engine.getRegistry(),
4123
+ ctx.engine.getTools(),
4124
+ ctx.engine.getPermission(),
4125
+ )
4126
+ const result = await sa.execute(prompt, 'fork: ' + prompt.slice(0, 60), {
4127
+ worktreePath: wtPath,
4128
+ })
4098
4129
  try {
4099
4130
  const { execSync: ex } = await import('node:child_process')
4100
4131
  ex(`git -C ${wtPath} add -A`, { stdio: 'ignore', timeout: 10_000 })
@@ -1,6 +1,7 @@
1
1
  import { SubAgent } from '../../agent/sub-agent'
2
2
  import type { ProviderRegistry } from '../../providers/registry'
3
3
  import type { ToolDefinition } from '../../shared/index.ts'
4
+ import type { PermissionSystem } from '../../core/permission'
4
5
  import { validateJSONSchema, formatValidationErrors } from '../schema-validator'
5
6
 
6
7
  export interface WorkflowAgentOpts {
@@ -12,17 +13,28 @@ export interface WorkflowAgentOpts {
12
13
  effort?: 'low' | 'medium' | 'high' | 'max'
13
14
  /** Maximum retries on schema validation failure (default: 2). */
14
15
  maxRetries?: number
16
+ /** Permission system for sub-agent tool execution (optional). */
17
+ permissionSystem?: PermissionSystem
18
+ /** Run the agent in an isolated git worktree (optional). */
19
+ isolation?: 'worktree'
15
20
  }
16
21
 
17
22
  /**
18
23
  * Workflow agent() primitive — creates a SubAgent with optional
19
- * provider/model override and structured output schema validation.
24
+ * provider/model override, structured output schema validation,
25
+ * and git worktree isolation.
20
26
  *
21
27
  * When `schema` is provided:
22
28
  * 1. The sub-agent is prompted to return valid JSON matching the schema.
23
29
  * 2. The result is JSON.parsed and validated against the schema.
24
30
  * 3. On validation failure, the sub-agent is retried with error feedback.
25
31
  * 4. Returns the validated object, or { raw, validationErrors } on final failure.
32
+ *
33
+ * When `isolation: 'worktree'` is set:
34
+ * 1. A git worktree is created at .claude/worktrees/wf-<slug>
35
+ * 2. The sub-agent runs with its cwd set to the worktree path
36
+ * 3. Changes are auto-committed (best-effort)
37
+ * 4. The worktree is cleaned up after execution
26
38
  */
27
39
  export async function workflowAgent(
28
40
  prompt: string,
@@ -39,6 +51,26 @@ export async function workflowAgent(
39
51
 
40
52
  const maxRetries = opts.maxRetries ?? 2
41
53
 
54
+ // ── Worktree isolation setup ──
55
+ let worktreePath: string | undefined
56
+ let worktreeBranch: string | undefined
57
+
58
+ if (opts.isolation === 'worktree') {
59
+ const slug = `wf-${Date.now().toString(36)}-${Math.random().toString(36).slice(2, 6)}`
60
+ worktreeBranch = `worktree/${slug}`
61
+ worktreePath = `.claude/worktrees/${slug}`
62
+
63
+ const proc = Bun.spawn(['git', 'worktree', 'add', '-b', worktreeBranch, worktreePath, 'HEAD'], {
64
+ stdout: 'pipe',
65
+ stderr: 'pipe',
66
+ })
67
+ const exitCode = await proc.exited
68
+ if (exitCode !== 0) {
69
+ const stderr = await new Response(proc.stderr).text()
70
+ throw new Error(`Worktree creation failed: ${stderr.slice(0, 500)}`)
71
+ }
72
+ }
73
+
42
74
  // Build the prompt — if schema is requested, instruct the model to return JSON
43
75
  let effectivePrompt = prompt
44
76
  if (opts.schema) {
@@ -50,64 +82,118 @@ export async function workflowAgent(
50
82
  `Return ONLY the JSON object, no other text. Do not wrap in markdown code fences.`
51
83
  }
52
84
 
53
- const sub = new SubAgent(registry, toolRegistry)
85
+ const sub = new SubAgent(registry, toolRegistry, opts.permissionSystem)
54
86
 
55
- // ── Attempt execution with optional schema validation + retry ──
87
+ // ── Execute with optional schema validation + retry ──
88
+ let result: unknown = ''
56
89
  let lastResult = ''
57
90
  let lastErrors: string[] = []
58
91
 
59
- for (let attempt = 0; attempt <= maxRetries; attempt++) {
60
- const retryPrompt =
61
- attempt === 0
62
- ? effectivePrompt
63
- : `${effectivePrompt}\n\n` +
64
- `[RETRY #${attempt}] Your previous response did NOT match the required schema.\n` +
65
- `Validation errors:\n${lastErrors.map((e) => ` • ${e}`).join('\n')}\n\n` +
66
- `Please fix the errors and return a valid JSON object matching the schema exactly.`
67
-
68
- const result = await sub.execute(retryPrompt, opts.label || 'workflow-agent', {
69
- type: 'general',
70
- modelOverride: opts.model,
71
- allowedTools: undefined, // use all tools by default
72
- })
92
+ try {
93
+ for (let attempt = 0; attempt <= maxRetries; attempt++) {
94
+ const retryPrompt =
95
+ attempt === 0
96
+ ? effectivePrompt
97
+ : `${effectivePrompt}\n\n` +
98
+ `[RETRY #${attempt}] Your previous response did NOT match the required schema.\n` +
99
+ `Validation errors:\n${lastErrors.map((e) => ` • ${e}`).join('\n')}\n\n` +
100
+ `Please fix the errors and return a valid JSON object matching the schema exactly.`
73
101
 
74
- lastResult = result
102
+ const textResult = await sub.execute(retryPrompt, opts.label || 'workflow-agent', {
103
+ type: 'general',
104
+ modelOverride: opts.model,
105
+ allowedTools: undefined, // use all tools by default
106
+ worktreePath,
107
+ })
75
108
 
76
- // No schema — return raw result
77
- if (!opts.schema) {
78
- return result
79
- }
109
+ lastResult = textResult
80
110
 
81
- // Try to parse and validate
82
- try {
83
- // Extract JSON from potential markdown fences
84
- let jsonStr = result.trim()
85
- const fenceMatch = jsonStr.match(/```(?:json)?\s*\n?([\s\S]*?)\n?```/)
86
- if (fenceMatch) {
87
- jsonStr = fenceMatch[1]!.trim()
111
+ // No schema — return raw result
112
+ if (!opts.schema) {
113
+ result = textResult
114
+ return result
88
115
  }
89
116
 
90
- const parsed = JSON.parse(jsonStr)
91
- const errors = validateJSONSchema(parsed, opts.schema as Record<string, unknown>)
117
+ // Try to parse and validate
118
+ try {
119
+ // Extract JSON from potential markdown fences
120
+ let jsonStr = textResult.trim()
121
+ const fenceMatch = jsonStr.match(/```(?:json)?\s*\n?([\s\S]*?)\n?```/)
122
+ if (fenceMatch) {
123
+ jsonStr = fenceMatch[1]!.trim()
124
+ }
92
125
 
93
- if (errors.length === 0) {
94
- // Valid! Return the parsed object
95
- return parsed
126
+ const parsed = JSON.parse(jsonStr)
127
+ const errors = validateJSONSchema(parsed, opts.schema as Record<string, unknown>)
128
+
129
+ if (errors.length === 0) {
130
+ // Valid! Return the parsed object
131
+ result = parsed
132
+ return result
133
+ }
134
+
135
+ // Not valid — collect errors for retry
136
+ lastErrors = [formatValidationErrors(errors)]
137
+ } catch (err) {
138
+ // JSON parse error
139
+ lastErrors = [`JSON parse error: ${String(err)}. Ensure your response is valid JSON only.`]
96
140
  }
141
+ }
97
142
 
98
- // Not valid — collect errors for retry
99
- lastErrors = [formatValidationErrors(errors)]
100
- } catch (err) {
101
- // JSON parse error
102
- lastErrors = [`JSON parse error: ${String(err)}. Ensure your response is valid JSON only.`]
143
+ // All retries exhausted — return raw result with validation errors
144
+ try {
145
+ const parsed = JSON.parse(lastResult)
146
+ result = { raw: lastResult, parsed, validationErrors: lastErrors }
147
+ } catch {
148
+ result = { raw: lastResult, validationErrors: lastErrors }
103
149
  }
104
- }
105
150
 
106
- // All retries exhausted — return raw result with validation errors
107
- try {
108
- const parsed = JSON.parse(lastResult)
109
- return { raw: lastResult, parsed, validationErrors: lastErrors }
110
- } catch {
111
- return { raw: lastResult, validationErrors: lastErrors }
151
+ return result
152
+ } finally {
153
+ // ── Cleanup worktree ──
154
+ if (worktreePath) {
155
+ // Best-effort: auto-commit any changes
156
+ try {
157
+ const addProc = Bun.spawn(['git', '-C', worktreePath, 'add', '-A'], {
158
+ stdout: 'pipe',
159
+ stderr: 'pipe',
160
+ })
161
+ await addProc.exited
162
+
163
+ const statusProc = Bun.spawn(['git', '-C', worktreePath, 'status', '--porcelain'], {
164
+ stdout: 'pipe',
165
+ stderr: 'pipe',
166
+ })
167
+ const status = await new Response(statusProc.stdout).text()
168
+ if (status.trim()) {
169
+ const commitMsg = opts.label ? `workflow: ${opts.label}` : 'workflow: agent changes'
170
+ const commitProc = Bun.spawn(['git', '-C', worktreePath, 'commit', '-m', commitMsg], {
171
+ stdout: 'pipe',
172
+ stderr: 'pipe',
173
+ })
174
+ await commitProc.exited
175
+ }
176
+ } catch {
177
+ /* best-effort */
178
+ }
179
+
180
+ // Remove worktree and branch
181
+ try {
182
+ const rmProc = Bun.spawn(['git', 'worktree', 'remove', '--force', worktreePath], {
183
+ stdout: 'pipe',
184
+ stderr: 'pipe',
185
+ })
186
+ await rmProc.exited
187
+ if (worktreeBranch) {
188
+ const brProc = Bun.spawn(['git', 'branch', '-D', worktreeBranch], {
189
+ stdout: 'pipe',
190
+ stderr: 'pipe',
191
+ })
192
+ await brProc.exited
193
+ }
194
+ } catch {
195
+ /* best-effort */
196
+ }
197
+ }
112
198
  }
113
199
  }
@@ -96,6 +96,7 @@ export async function runWorkflow(
96
96
 
97
97
  const registry: ProviderRegistry = engine.getRegistry()
98
98
  const toolRegistry = engine.getTools()
99
+ const permission = engine.getPermission()
99
100
 
100
101
  const budget = createBudget(budgetTotal)
101
102
 
@@ -111,12 +112,10 @@ export async function runWorkflow(
111
112
  }
112
113
  cacheMisses++
113
114
 
114
- const result = await workflowAgent(
115
- prompt,
116
- registry,
117
- toolRegistry,
118
- opts as Record<string, unknown>,
119
- )
115
+ const result = await workflowAgent(prompt, registry, toolRegistry, {
116
+ ...(opts || {}),
117
+ permissionSystem: permission,
118
+ } as Record<string, unknown>)
120
119
  appendJournal(runId, {
121
120
  type: 'agent',
122
121
  prompt,