@miphamai/cli 0.14.0 → 0.16.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -10,6 +10,7 @@ import type { SkillsLoader } from '../skills/loader'
10
10
  import type { PluginManager } from '../plugin/plugin-manager'
11
11
  import { McpClient } from '../mcp/client'
12
12
  import { NPM_INSTALL_COMMAND, NPM_UPDATE_COMMAND, PACKAGE_VERSION } from '../shared/index.ts'
13
+ import { getPreference } from '../config/preferences'
13
14
 
14
15
  export interface CommandContext {
15
16
  engine: QueryEngine
@@ -17,6 +18,7 @@ export interface CommandContext {
17
18
  providerId: string
18
19
  modelId: string
19
20
  version: string
21
+ sessionId: string
20
22
  // Callbacks for commands that mutate App state
21
23
  setSessionTitle: (title: string) => void
22
24
  setFastMode: (on: boolean) => void
@@ -689,22 +691,47 @@ const recapCmd: CommandHandler = (ctx) => {
689
691
 
690
692
  const usageCmd: CommandHandler = (ctx) => {
691
693
  const c = ctx.engine.getContext()
692
- const tokens = c.getEstimatedTokens()
694
+ const estTokens = c.getEstimatedTokens()
693
695
  const msgs = c.getMessages()
694
696
  const maxTokens = 200_000
695
- const pct = ((tokens / maxTokens) * 100).toFixed(1)
697
+ const pct = ((estTokens / maxTokens) * 100).toFixed(1)
698
+
699
+ const tracker = ctx.engine.getUsageTracker()
700
+ const summary = tracker.getSummary()
701
+
702
+ // Build per-tool breakdown
703
+ const toolLines: string[] = []
704
+ const sortedTools = Object.entries(summary.tools).sort(
705
+ (a, b) => b[1].inputTokens + b[1].outputTokens - (a[1].inputTokens + a[1].outputTokens),
706
+ )
707
+ for (const [name, usage] of sortedTools) {
708
+ const total = usage.inputTokens + usage.outputTokens
709
+ const prefix = name.startsWith('mcp__') ? 'šŸ”Œ ' : ' '
710
+ toolLines.push(
711
+ `${prefix}${name.padEnd(18)} ${total.toLocaleString().padStart(8)} tokens (${usage.calls} call${usage.calls !== 1 ? 's' : ''})`,
712
+ )
713
+ }
714
+
715
+ const apiTotal = summary.apiInputTokens + summary.apiOutputTokens
716
+ const apiLine =
717
+ summary.apiInputTokens > 0 || summary.apiOutputTokens > 0
718
+ ? `API tokens: ${summary.apiInputTokens.toLocaleString().padStart(8)} in / ${summary.apiOutputTokens.toLocaleString().padStart(6)} out (${apiTotal.toLocaleString()} total)`
719
+ : 'API tokens: (no API usage data yet)'
720
+
721
+ const toolSection = toolLines.length > 0 ? `\n── Per-Tool ──\n${toolLines.join('\n')}` : ''
696
722
 
697
723
  return {
698
724
  content: stripIndent`
699
725
  ── Usage Dashboard ──
700
- Context tokens: ~${tokens.toLocaleString()} / ${maxTokens.toLocaleString()} (${pct}%)
726
+ ${apiLine}
727
+ Context tokens: ~${estTokens.toLocaleString()} / ${maxTokens.toLocaleString()} (${pct}%)
701
728
  Messages: ${msgs.length}
702
729
  Provider: ${ctx.providerId}
703
730
  Model: ${ctx.modelId}
704
731
 
705
732
  ${'ā–ˆ'.repeat(Math.ceil(Number(pct) / 5))}${'ā–‘'.repeat(20 - Math.ceil(Number(pct) / 5))} ${pct}%
733
+ ${toolSection}
706
734
 
707
- Note: Token counting is approximate (chars/4).
708
735
  Use /context for detailed stats, /compact to free space.
709
736
  `,
710
737
  }
@@ -774,7 +801,7 @@ const browseSkillsCmd: CommandHandler = async () => {
774
801
  return { content: lines.join('\n') }
775
802
  }
776
803
 
777
- const installSkillCmd: CommandHandler = async (_ctx, args) => {
804
+ const installSkillCmd: CommandHandler = async (ctx, args) => {
778
805
  const { installSkill, installSkillFromUrl } = await import('../skills/registry')
779
806
 
780
807
  const target = args[0]
@@ -782,12 +809,13 @@ const installSkillCmd: CommandHandler = async (_ctx, args) => {
782
809
  return { content: 'Usage: /install-skill <skill-name> or /install-skill <url>' }
783
810
  }
784
811
 
812
+ const marketplaceConfig = ctx.config.marketplace
785
813
  let result: { success: boolean; name: string; message: string }
786
814
 
787
815
  if (target.startsWith('http://') || target.startsWith('https://')) {
788
- result = installSkillFromUrl(target)
816
+ result = installSkillFromUrl(target, marketplaceConfig)
789
817
  } else {
790
- result = installSkill(target)
818
+ result = installSkill(target, marketplaceConfig)
791
819
  }
792
820
 
793
821
  return {
@@ -1656,7 +1684,7 @@ function gitDiffBridgeCmd(opts: {
1656
1684
  label: string
1657
1685
  noChangesHint: string
1658
1686
  runningMsg: string
1659
- forwardToAI: string
1687
+ forwardToAI: string | (() => string)
1660
1688
  }): CommandHandler {
1661
1689
  return async () => {
1662
1690
  try {
@@ -1667,7 +1695,7 @@ function gitDiffBridgeCmd(opts: {
1667
1695
  }
1668
1696
  return {
1669
1697
  content: `─ ${opts.label} ─\n\n${opts.runningMsg}\n\nChanged files:\n${diff}`,
1670
- forwardToAI: opts.forwardToAI,
1698
+ forwardToAI: typeof opts.forwardToAI === 'function' ? opts.forwardToAI() : opts.forwardToAI,
1671
1699
  }
1672
1700
  } catch {
1673
1701
  return {
@@ -1683,8 +1711,8 @@ const codeReviewCmd = gitDiffBridgeCmd({
1683
1711
  'No uncommitted changes to review.\n\nTo review a specific file: /code-review path/to/file.ts',
1684
1712
  runningMsg:
1685
1713
  'Reviewing uncommitted changes with the code-review skill (7 dimensions: correctness, security, performance, code quality, architecture, testing, language-specific)...',
1686
- forwardToAI:
1687
- 'use the code-review skill to review all uncommitted changes. Check all 7 dimensions: correctness, security, performance, code quality, architecture & design, testing, and language-specific issues.',
1714
+ forwardToAI: () =>
1715
+ `use the code-review skill to review all uncommitted changes. Check all 7 dimensions: correctness, security, performance, code quality, architecture & design, testing, and language-specific issues. Use effort level: ${getPreference('lastCodeReviewEffort', 'high')}.`,
1688
1716
  })
1689
1717
 
1690
1718
  const simplifyCmd = gitDiffBridgeCmd({
@@ -1862,7 +1890,7 @@ const summaryCmd: CommandHandler = (ctx) => {
1862
1890
  }
1863
1891
  }
1864
1892
 
1865
- const cdCmd: CommandHandler = async (_ctx, args) => {
1893
+ const cdCmd: CommandHandler = async (ctx, args) => {
1866
1894
  const target = args[0]
1867
1895
  if (!target) {
1868
1896
  return {
@@ -1887,6 +1915,22 @@ const cdCmd: CommandHandler = async (_ctx, args) => {
1887
1915
 
1888
1916
  try {
1889
1917
  process.chdir(resolved)
1918
+
1919
+ // Persist cwd to active session (best-effort)
1920
+ try {
1921
+ const { SessionStore } = await import('../core/session-store')
1922
+ const saved = SessionStore.load(ctx.sessionId)
1923
+ if (saved) {
1924
+ SessionStore.save(ctx.sessionId, saved.messages, {
1925
+ provider: saved.metadata.provider,
1926
+ model: saved.metadata.model,
1927
+ cwd: resolved,
1928
+ })
1929
+ }
1930
+ } catch {
1931
+ /* session persistence is best-effort */
1932
+ }
1933
+
1890
1934
  return {
1891
1935
  content: [
1892
1936
  '── Directory Changed ──',
@@ -2014,48 +2058,10 @@ const exportCmd: CommandHandler = async (ctx) => {
2014
2058
  }
2015
2059
 
2016
2060
  // ═══════════════════════════════════════════════════════════════
2017
- // Review — code review workflow
2061
+ // Review — alias for /code-review (P1: 2026-08-06 polish)
2018
2062
  // ═══════════════════════════════════════════════════════════════
2019
2063
 
2020
- const reviewCmd: CommandHandler = async () => {
2021
- try {
2022
- const { execSync } = await import('node:child_process')
2023
- const diff = execSync('git diff --stat', { encoding: 'utf-8', timeout: 5000 }).trim()
2024
- const unstaged = execSync('git diff --name-only', { encoding: 'utf-8', timeout: 3000 }).trim()
2025
- const staged = execSync('git diff --cached --name-only', {
2026
- encoding: 'utf-8',
2027
- timeout: 3000,
2028
- }).trim()
2029
-
2030
- if (!diff) {
2031
- return {
2032
- content:
2033
- '─ Code Review ─\n\nNo uncommitted changes detected.\n\nUse /pr-comments for PR-level review, or make changes first.',
2034
- }
2035
- }
2036
-
2037
- const lines: string[] = ['─ Code Review ─', '', 'Uncommitted changes:', '', diff]
2038
-
2039
- if (staged) {
2040
- lines.push('')
2041
- lines.push('Staged files (ready for commit):')
2042
- for (const f of staged.split('\n')) lines.push(` āœ“ ${f}`)
2043
- }
2044
- if (unstaged) {
2045
- lines.push('')
2046
- lines.push('Unstaged files (working directory):')
2047
- for (const f of unstaged.split('\n')) lines.push(` • ${f}`)
2048
- }
2049
-
2050
- lines.push('')
2051
- lines.push('To review with AI: type "review these changes" in chat.')
2052
- lines.push('To commit: git add -A && git commit -m "..."')
2053
-
2054
- return { content: lines.join('\n') }
2055
- } catch {
2056
- return { content: '─ Code Review ─\n\nCould not run git diff. Are you in a git repository?' }
2057
- }
2058
- }
2064
+ const reviewCmd = codeReviewCmd
2059
2065
 
2060
2066
  // ═══════════════════════════════════════════════════════════════
2061
2067
  // PR Comments
@@ -4093,8 +4099,14 @@ const forkCmd: CommandHandler = async (ctx, args) => {
4093
4099
  'general',
4094
4100
  async (_signal) => {
4095
4101
  const { SubAgent } = await import('../agent/sub-agent')
4096
- const sa = new SubAgent(ctx.engine.getRegistry(), ctx.engine.getTools())
4097
- const result = await sa.execute(prompt, 'fork: ' + prompt.slice(0, 60), {})
4102
+ const sa = new SubAgent(
4103
+ ctx.engine.getRegistry(),
4104
+ ctx.engine.getTools(),
4105
+ ctx.engine.getPermission(),
4106
+ )
4107
+ const result = await sa.execute(prompt, 'fork: ' + prompt.slice(0, 60), {
4108
+ worktreePath: wtPath,
4109
+ })
4098
4110
  try {
4099
4111
  const { execSync: ex } = await import('node:child_process')
4100
4112
  ex(`git -C ${wtPath} add -A`, { stdio: 'ignore', timeout: 10_000 })
@@ -4289,7 +4301,7 @@ const commandsListCmd: CommandHandler = () => {
4289
4301
  '/tdd': 'Workflow',
4290
4302
  '/todos': 'Workflow',
4291
4303
  '/tasks': 'Workflow',
4292
- '/review': 'Workflow',
4304
+ '/review': 'Code Quality',
4293
4305
  '/pr-comments': 'Workflow',
4294
4306
  '/diff': 'Workflow',
4295
4307
  '/workflows': 'Workflow',
@@ -4556,7 +4568,7 @@ const COMMAND_DESCRIPTIONS: Record<string, string> = {
4556
4568
  '/tdd': 'Test-Driven Development workflow (RED → GREEN → REFACTOR)',
4557
4569
  '/todos': 'Task management (list/create tasks)',
4558
4570
  '/tasks': 'Background tasks',
4559
- '/review': 'Code review workflow',
4571
+ '/review': 'Review code for bugs, security, performance (alias for /code-review)',
4560
4572
  '/pr-comments': 'PR review summary',
4561
4573
  '/diff': 'Show git diff',
4562
4574
  '/workflows': 'List workflow scripts',
@@ -1,6 +1,7 @@
1
1
  import { SubAgent } from '../../agent/sub-agent'
2
2
  import type { ProviderRegistry } from '../../providers/registry'
3
3
  import type { ToolDefinition } from '../../shared/index.ts'
4
+ import type { PermissionSystem } from '../../core/permission'
4
5
  import { validateJSONSchema, formatValidationErrors } from '../schema-validator'
5
6
 
6
7
  export interface WorkflowAgentOpts {
@@ -12,17 +13,28 @@ export interface WorkflowAgentOpts {
12
13
  effort?: 'low' | 'medium' | 'high' | 'max'
13
14
  /** Maximum retries on schema validation failure (default: 2). */
14
15
  maxRetries?: number
16
+ /** Permission system for sub-agent tool execution (optional). */
17
+ permissionSystem?: PermissionSystem
18
+ /** Run the agent in an isolated git worktree (optional). */
19
+ isolation?: 'worktree'
15
20
  }
16
21
 
17
22
  /**
18
23
  * Workflow agent() primitive — creates a SubAgent with optional
19
- * provider/model override and structured output schema validation.
24
+ * provider/model override, structured output schema validation,
25
+ * and git worktree isolation.
20
26
  *
21
27
  * When `schema` is provided:
22
28
  * 1. The sub-agent is prompted to return valid JSON matching the schema.
23
29
  * 2. The result is JSON.parsed and validated against the schema.
24
30
  * 3. On validation failure, the sub-agent is retried with error feedback.
25
31
  * 4. Returns the validated object, or { raw, validationErrors } on final failure.
32
+ *
33
+ * When `isolation: 'worktree'` is set:
34
+ * 1. A git worktree is created at .claude/worktrees/wf-<slug>
35
+ * 2. The sub-agent runs with its cwd set to the worktree path
36
+ * 3. Changes are auto-committed (best-effort)
37
+ * 4. The worktree is cleaned up after execution
26
38
  */
27
39
  export async function workflowAgent(
28
40
  prompt: string,
@@ -39,6 +51,26 @@ export async function workflowAgent(
39
51
 
40
52
  const maxRetries = opts.maxRetries ?? 2
41
53
 
54
+ // ── Worktree isolation setup ──
55
+ let worktreePath: string | undefined
56
+ let worktreeBranch: string | undefined
57
+
58
+ if (opts.isolation === 'worktree') {
59
+ const slug = `wf-${Date.now().toString(36)}-${Math.random().toString(36).slice(2, 6)}`
60
+ worktreeBranch = `worktree/${slug}`
61
+ worktreePath = `.claude/worktrees/${slug}`
62
+
63
+ const proc = Bun.spawn(['git', 'worktree', 'add', '-b', worktreeBranch, worktreePath, 'HEAD'], {
64
+ stdout: 'pipe',
65
+ stderr: 'pipe',
66
+ })
67
+ const exitCode = await proc.exited
68
+ if (exitCode !== 0) {
69
+ const stderr = await new Response(proc.stderr).text()
70
+ throw new Error(`Worktree creation failed: ${stderr.slice(0, 500)}`)
71
+ }
72
+ }
73
+
42
74
  // Build the prompt — if schema is requested, instruct the model to return JSON
43
75
  let effectivePrompt = prompt
44
76
  if (opts.schema) {
@@ -50,64 +82,118 @@ export async function workflowAgent(
50
82
  `Return ONLY the JSON object, no other text. Do not wrap in markdown code fences.`
51
83
  }
52
84
 
53
- const sub = new SubAgent(registry, toolRegistry)
85
+ const sub = new SubAgent(registry, toolRegistry, opts.permissionSystem)
54
86
 
55
- // ── Attempt execution with optional schema validation + retry ──
87
+ // ── Execute with optional schema validation + retry ──
88
+ let result: unknown = ''
56
89
  let lastResult = ''
57
90
  let lastErrors: string[] = []
58
91
 
59
- for (let attempt = 0; attempt <= maxRetries; attempt++) {
60
- const retryPrompt =
61
- attempt === 0
62
- ? effectivePrompt
63
- : `${effectivePrompt}\n\n` +
64
- `[RETRY #${attempt}] Your previous response did NOT match the required schema.\n` +
65
- `Validation errors:\n${lastErrors.map((e) => ` • ${e}`).join('\n')}\n\n` +
66
- `Please fix the errors and return a valid JSON object matching the schema exactly.`
67
-
68
- const result = await sub.execute(retryPrompt, opts.label || 'workflow-agent', {
69
- type: 'general',
70
- modelOverride: opts.model,
71
- allowedTools: undefined, // use all tools by default
72
- })
92
+ try {
93
+ for (let attempt = 0; attempt <= maxRetries; attempt++) {
94
+ const retryPrompt =
95
+ attempt === 0
96
+ ? effectivePrompt
97
+ : `${effectivePrompt}\n\n` +
98
+ `[RETRY #${attempt}] Your previous response did NOT match the required schema.\n` +
99
+ `Validation errors:\n${lastErrors.map((e) => ` • ${e}`).join('\n')}\n\n` +
100
+ `Please fix the errors and return a valid JSON object matching the schema exactly.`
73
101
 
74
- lastResult = result
102
+ const textResult = await sub.execute(retryPrompt, opts.label || 'workflow-agent', {
103
+ type: 'general',
104
+ modelOverride: opts.model,
105
+ allowedTools: undefined, // use all tools by default
106
+ worktreePath,
107
+ })
75
108
 
76
- // No schema — return raw result
77
- if (!opts.schema) {
78
- return result
79
- }
109
+ lastResult = textResult
80
110
 
81
- // Try to parse and validate
82
- try {
83
- // Extract JSON from potential markdown fences
84
- let jsonStr = result.trim()
85
- const fenceMatch = jsonStr.match(/```(?:json)?\s*\n?([\s\S]*?)\n?```/)
86
- if (fenceMatch) {
87
- jsonStr = fenceMatch[1]!.trim()
111
+ // No schema — return raw result
112
+ if (!opts.schema) {
113
+ result = textResult
114
+ return result
88
115
  }
89
116
 
90
- const parsed = JSON.parse(jsonStr)
91
- const errors = validateJSONSchema(parsed, opts.schema as Record<string, unknown>)
117
+ // Try to parse and validate
118
+ try {
119
+ // Extract JSON from potential markdown fences
120
+ let jsonStr = textResult.trim()
121
+ const fenceMatch = jsonStr.match(/```(?:json)?\s*\n?([\s\S]*?)\n?```/)
122
+ if (fenceMatch) {
123
+ jsonStr = fenceMatch[1]!.trim()
124
+ }
92
125
 
93
- if (errors.length === 0) {
94
- // Valid! Return the parsed object
95
- return parsed
126
+ const parsed = JSON.parse(jsonStr)
127
+ const errors = validateJSONSchema(parsed, opts.schema as Record<string, unknown>)
128
+
129
+ if (errors.length === 0) {
130
+ // Valid! Return the parsed object
131
+ result = parsed
132
+ return result
133
+ }
134
+
135
+ // Not valid — collect errors for retry
136
+ lastErrors = [formatValidationErrors(errors)]
137
+ } catch (err) {
138
+ // JSON parse error
139
+ lastErrors = [`JSON parse error: ${String(err)}. Ensure your response is valid JSON only.`]
96
140
  }
141
+ }
97
142
 
98
- // Not valid — collect errors for retry
99
- lastErrors = [formatValidationErrors(errors)]
100
- } catch (err) {
101
- // JSON parse error
102
- lastErrors = [`JSON parse error: ${String(err)}. Ensure your response is valid JSON only.`]
143
+ // All retries exhausted — return raw result with validation errors
144
+ try {
145
+ const parsed = JSON.parse(lastResult)
146
+ result = { raw: lastResult, parsed, validationErrors: lastErrors }
147
+ } catch {
148
+ result = { raw: lastResult, validationErrors: lastErrors }
103
149
  }
104
- }
105
150
 
106
- // All retries exhausted — return raw result with validation errors
107
- try {
108
- const parsed = JSON.parse(lastResult)
109
- return { raw: lastResult, parsed, validationErrors: lastErrors }
110
- } catch {
111
- return { raw: lastResult, validationErrors: lastErrors }
151
+ return result
152
+ } finally {
153
+ // ── Cleanup worktree ──
154
+ if (worktreePath) {
155
+ // Best-effort: auto-commit any changes
156
+ try {
157
+ const addProc = Bun.spawn(['git', '-C', worktreePath, 'add', '-A'], {
158
+ stdout: 'pipe',
159
+ stderr: 'pipe',
160
+ })
161
+ await addProc.exited
162
+
163
+ const statusProc = Bun.spawn(['git', '-C', worktreePath, 'status', '--porcelain'], {
164
+ stdout: 'pipe',
165
+ stderr: 'pipe',
166
+ })
167
+ const status = await new Response(statusProc.stdout).text()
168
+ if (status.trim()) {
169
+ const commitMsg = opts.label ? `workflow: ${opts.label}` : 'workflow: agent changes'
170
+ const commitProc = Bun.spawn(['git', '-C', worktreePath, 'commit', '-m', commitMsg], {
171
+ stdout: 'pipe',
172
+ stderr: 'pipe',
173
+ })
174
+ await commitProc.exited
175
+ }
176
+ } catch {
177
+ /* best-effort */
178
+ }
179
+
180
+ // Remove worktree and branch
181
+ try {
182
+ const rmProc = Bun.spawn(['git', 'worktree', 'remove', '--force', worktreePath], {
183
+ stdout: 'pipe',
184
+ stderr: 'pipe',
185
+ })
186
+ await rmProc.exited
187
+ if (worktreeBranch) {
188
+ const brProc = Bun.spawn(['git', 'branch', '-D', worktreeBranch], {
189
+ stdout: 'pipe',
190
+ stderr: 'pipe',
191
+ })
192
+ await brProc.exited
193
+ }
194
+ } catch {
195
+ /* best-effort */
196
+ }
197
+ }
112
198
  }
113
199
  }
@@ -1,3 +1,4 @@
1
+ import vm from 'node:vm'
1
2
  import { createSandbox } from './sandbox'
2
3
  import { createJournal, appendJournal, loadJournal, loadScript } from './journal'
3
4
  import { createBudget } from './budget'
@@ -47,7 +48,8 @@ function agentCacheKey(prompt: string, opts: Record<string, unknown> = {}): stri
47
48
  * - Returns cacheHits and cacheMisses counts
48
49
  *
49
50
  * The sandbox blocks non-deterministic APIs (Date.now, Math.random) to make
50
- * this deterministic replay possible.
51
+ * this deterministic replay possible. Scripts execute inside a node:vm.Context
52
+ * that blocks sandbox escape vectors (eval, Function, import, require, process, etc.).
51
53
  */
52
54
  export async function runWorkflow(
53
55
  script: string,
@@ -96,6 +98,7 @@ export async function runWorkflow(
96
98
 
97
99
  const registry: ProviderRegistry = engine.getRegistry()
98
100
  const toolRegistry = engine.getTools()
101
+ const permission = engine.getPermission()
99
102
 
100
103
  const budget = createBudget(budgetTotal)
101
104
 
@@ -111,12 +114,10 @@ export async function runWorkflow(
111
114
  }
112
115
  cacheMisses++
113
116
 
114
- const result = await workflowAgent(
115
- prompt,
116
- registry,
117
- toolRegistry,
118
- opts as Record<string, unknown>,
119
- )
117
+ const result = await workflowAgent(prompt, registry, toolRegistry, {
118
+ ...(opts || {}),
119
+ permissionSystem: permission,
120
+ } as Record<string, unknown>)
120
121
  appendJournal(runId, {
121
122
  type: 'agent',
122
123
  prompt,
@@ -135,42 +136,28 @@ export async function runWorkflow(
135
136
  appendJournal(runId, { type: 'log', message })
136
137
  }
137
138
 
138
- const sandbox = createSandbox(args, budget)
139
-
140
139
  // Build the script wrapper
141
140
  const wrappedScript = `
142
- return (async () => {
141
+ (async () => {
143
142
  ${script}
144
143
  })()
145
144
  `
146
145
 
147
- // Execute in sandboxed context
148
- const scriptFn = new Function(
149
- 'agent',
150
- 'parallel',
151
- 'pipeline',
152
- 'verify',
153
- 'judge',
154
- 'loopUntilConvergence',
155
- 'phase',
156
- 'log',
157
- 'args',
158
- 'budget',
159
- wrappedScript,
160
- )
161
-
162
- const result = await scriptFn(
163
- agent,
164
- parallel,
165
- pipeline,
166
- verify,
167
- judge,
168
- loopUntilConvergence,
169
- wrappedPhase,
170
- log,
171
- args,
172
- budget,
173
- )
146
+ // Execute in node:vm sandbox — no access to host globals
147
+ const vmScript = new vm.Script(wrappedScript, { filename: 'workflow.js' })
148
+ const sandboxCtx = createSandbox(args, budget)
149
+
150
+ // Inject primitives into the sandbox context (whitelist approach)
151
+ sandboxCtx.agent = agent
152
+ sandboxCtx.parallel = parallel
153
+ sandboxCtx.pipeline = pipeline
154
+ sandboxCtx.verify = verify
155
+ sandboxCtx.judge = judge
156
+ sandboxCtx.loopUntilConvergence = loopUntilConvergence
157
+ sandboxCtx.phase = wrappedPhase
158
+ sandboxCtx.log = log
159
+
160
+ const result = await vmScript.runInContext(sandboxCtx, { timeout: 120_000 })
174
161
 
175
162
  // Count journal entries from state
176
163
  const priorEntries = loadJournal(runId)
@@ -1,27 +1,39 @@
1
- /** APIs disabled in workflow scripts to ensure deterministic replay. */
1
+ import vm from 'node:vm'
2
+
3
+ /** APIs disabled in workflow scripts to ensure deterministic replay + sandbox escape prevention. */
2
4
  const FORBIDDEN = new Set(['Date.now', 'Math.random', 'crypto.randomUUID'])
3
5
 
4
6
  /**
5
- * Create a sandboxed global scope for workflow script execution.
6
- * Blocks Date.now(), Math.random(), argless new Date(), crypto.randomUUID().
7
+ * Create a sandboxed VM context for workflow script execution.
8
+ * Blocks Date.now(), Math.random(), argless new Date(), crypto.randomUUID(),
9
+ * plus explicit sandbox escape vectors: eval, Function constructor, import(),
10
+ * require(), process, Bun, fetch, setTimeout/setInterval, etc.
7
11
  */
8
12
  export function createSandbox(
9
13
  args: unknown,
10
14
  budget: { total: number | null; spent(): number; remaining(): number },
11
- ): Record<string, unknown> {
12
- const sandbox: Record<string, unknown> = {
15
+ ): vm.Context {
16
+ const sandboxObj: Record<string, unknown> = {
13
17
  args,
14
18
  budget,
15
19
  console: {
16
20
  log: (..._a: unknown[]) => {}, // no-op in sandbox
17
21
  error: (..._a: unknown[]) => {},
18
22
  },
19
- // Primitives are injected by the runtime, not the sandbox
23
+
24
+ // ── Explicit sandbox escape prevention ──
25
+ // V8 builtins that would otherwise be available in a vm.Context
26
+ eval: () => {
27
+ throw new Error('eval() is disabled in workflow sandbox.')
28
+ },
29
+ Function: () => {
30
+ throw new Error('new Function() is disabled in workflow sandbox.')
31
+ },
20
32
  }
21
33
 
22
34
  // Override Date to block now() and argless constructor
23
35
  const OriginalDate = Date
24
- sandbox.Date = new Proxy(OriginalDate, {
36
+ sandboxObj.Date = new Proxy(OriginalDate, {
25
37
  construct(_target, constructorArgs) {
26
38
  if (constructorArgs.length === 0) {
27
39
  throw new Error('new Date() is disabled in workflow sandbox. Pass timestamps via args.')
@@ -42,7 +54,7 @@ export function createSandbox(
42
54
  })
43
55
 
44
56
  // Override Math.random
45
- sandbox.Math = new Proxy(Math, {
57
+ sandboxObj.Math = new Proxy(Math, {
46
58
  get(_target, prop) {
47
59
  if (prop === 'random') {
48
60
  throw new Error('Math.random() is disabled in workflow sandbox. Use a seed from args.')
@@ -57,7 +69,7 @@ export function createSandbox(
57
69
  | { randomUUID?: unknown; [key: string]: unknown }
58
70
  | undefined
59
71
  if (globalCrypto) {
60
- sandbox.crypto = new Proxy(globalCrypto, {
72
+ sandboxObj.crypto = new Proxy(globalCrypto, {
61
73
  get(_target, prop) {
62
74
  if (prop === 'randomUUID') {
63
75
  throw new Error('crypto.randomUUID() is disabled in workflow sandbox.')
@@ -70,7 +82,7 @@ export function createSandbox(
70
82
  })
71
83
  }
72
84
 
73
- return sandbox
85
+ return vm.createContext(sandboxObj)
74
86
  }
75
87
 
76
88
  /** Check whether a given API identifier is in the forbidden set. */