@miphamai/cli 0.14.0 ā 0.16.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/agent/message-bus.ts +13 -1
- package/src/agent/sub-agent.ts +37 -3
- package/src/agent/types.ts +2 -0
- package/src/config/defaults.ts +4 -0
- package/src/config/preferences.ts +45 -0
- package/src/core/context.ts +9 -0
- package/src/core/engine.ts +52 -1
- package/src/core/permission-config.ts +70 -6
- package/src/core/permission.ts +50 -9
- package/src/core/session-store.ts +4 -2
- package/src/core/usage-tracker.ts +103 -0
- package/src/index.tsx +50 -2
- package/src/providers/anthropic.ts +9 -1
- package/src/providers/openai-compat.ts +10 -0
- package/src/providers/registry.ts +8 -0
- package/src/shared/sanitize.ts +38 -0
- package/src/shared/types.ts +32 -1
- package/src/skills/fork-executor.ts +3 -1
- package/src/skills/registry.ts +69 -2
- package/src/tools/agent/agent.ts +1 -1
- package/src/tools/agent/skill.ts +1 -0
- package/src/tools/exec/bash.ts +37 -5
- package/src/tools/index.ts +5 -2
- package/src/ui/app.tsx +16 -3
- package/src/ui/commands.ts +68 -56
- package/src/workflow/primitives/agent.ts +132 -46
- package/src/workflow/runtime.ts +24 -37
- package/src/workflow/sandbox.ts +22 -10
package/src/ui/commands.ts
CHANGED
|
@@ -10,6 +10,7 @@ import type { SkillsLoader } from '../skills/loader'
|
|
|
10
10
|
import type { PluginManager } from '../plugin/plugin-manager'
|
|
11
11
|
import { McpClient } from '../mcp/client'
|
|
12
12
|
import { NPM_INSTALL_COMMAND, NPM_UPDATE_COMMAND, PACKAGE_VERSION } from '../shared/index.ts'
|
|
13
|
+
import { getPreference } from '../config/preferences'
|
|
13
14
|
|
|
14
15
|
export interface CommandContext {
|
|
15
16
|
engine: QueryEngine
|
|
@@ -17,6 +18,7 @@ export interface CommandContext {
|
|
|
17
18
|
providerId: string
|
|
18
19
|
modelId: string
|
|
19
20
|
version: string
|
|
21
|
+
sessionId: string
|
|
20
22
|
// Callbacks for commands that mutate App state
|
|
21
23
|
setSessionTitle: (title: string) => void
|
|
22
24
|
setFastMode: (on: boolean) => void
|
|
@@ -689,22 +691,47 @@ const recapCmd: CommandHandler = (ctx) => {
|
|
|
689
691
|
|
|
690
692
|
const usageCmd: CommandHandler = (ctx) => {
|
|
691
693
|
const c = ctx.engine.getContext()
|
|
692
|
-
const
|
|
694
|
+
const estTokens = c.getEstimatedTokens()
|
|
693
695
|
const msgs = c.getMessages()
|
|
694
696
|
const maxTokens = 200_000
|
|
695
|
-
const pct = ((
|
|
697
|
+
const pct = ((estTokens / maxTokens) * 100).toFixed(1)
|
|
698
|
+
|
|
699
|
+
const tracker = ctx.engine.getUsageTracker()
|
|
700
|
+
const summary = tracker.getSummary()
|
|
701
|
+
|
|
702
|
+
// Build per-tool breakdown
|
|
703
|
+
const toolLines: string[] = []
|
|
704
|
+
const sortedTools = Object.entries(summary.tools).sort(
|
|
705
|
+
(a, b) => b[1].inputTokens + b[1].outputTokens - (a[1].inputTokens + a[1].outputTokens),
|
|
706
|
+
)
|
|
707
|
+
for (const [name, usage] of sortedTools) {
|
|
708
|
+
const total = usage.inputTokens + usage.outputTokens
|
|
709
|
+
const prefix = name.startsWith('mcp__') ? 'š ' : ' '
|
|
710
|
+
toolLines.push(
|
|
711
|
+
`${prefix}${name.padEnd(18)} ${total.toLocaleString().padStart(8)} tokens (${usage.calls} call${usage.calls !== 1 ? 's' : ''})`,
|
|
712
|
+
)
|
|
713
|
+
}
|
|
714
|
+
|
|
715
|
+
const apiTotal = summary.apiInputTokens + summary.apiOutputTokens
|
|
716
|
+
const apiLine =
|
|
717
|
+
summary.apiInputTokens > 0 || summary.apiOutputTokens > 0
|
|
718
|
+
? `API tokens: ${summary.apiInputTokens.toLocaleString().padStart(8)} in / ${summary.apiOutputTokens.toLocaleString().padStart(6)} out (${apiTotal.toLocaleString()} total)`
|
|
719
|
+
: 'API tokens: (no API usage data yet)'
|
|
720
|
+
|
|
721
|
+
const toolSection = toolLines.length > 0 ? `\nāā Per-Tool āā\n${toolLines.join('\n')}` : ''
|
|
696
722
|
|
|
697
723
|
return {
|
|
698
724
|
content: stripIndent`
|
|
699
725
|
āā Usage Dashboard āā
|
|
700
|
-
|
|
726
|
+
${apiLine}
|
|
727
|
+
Context tokens: ~${estTokens.toLocaleString()} / ${maxTokens.toLocaleString()} (${pct}%)
|
|
701
728
|
Messages: ${msgs.length}
|
|
702
729
|
Provider: ${ctx.providerId}
|
|
703
730
|
Model: ${ctx.modelId}
|
|
704
731
|
|
|
705
732
|
${'ā'.repeat(Math.ceil(Number(pct) / 5))}${'ā'.repeat(20 - Math.ceil(Number(pct) / 5))} ${pct}%
|
|
733
|
+
${toolSection}
|
|
706
734
|
|
|
707
|
-
Note: Token counting is approximate (chars/4).
|
|
708
735
|
Use /context for detailed stats, /compact to free space.
|
|
709
736
|
`,
|
|
710
737
|
}
|
|
@@ -774,7 +801,7 @@ const browseSkillsCmd: CommandHandler = async () => {
|
|
|
774
801
|
return { content: lines.join('\n') }
|
|
775
802
|
}
|
|
776
803
|
|
|
777
|
-
const installSkillCmd: CommandHandler = async (
|
|
804
|
+
const installSkillCmd: CommandHandler = async (ctx, args) => {
|
|
778
805
|
const { installSkill, installSkillFromUrl } = await import('../skills/registry')
|
|
779
806
|
|
|
780
807
|
const target = args[0]
|
|
@@ -782,12 +809,13 @@ const installSkillCmd: CommandHandler = async (_ctx, args) => {
|
|
|
782
809
|
return { content: 'Usage: /install-skill <skill-name> or /install-skill <url>' }
|
|
783
810
|
}
|
|
784
811
|
|
|
812
|
+
const marketplaceConfig = ctx.config.marketplace
|
|
785
813
|
let result: { success: boolean; name: string; message: string }
|
|
786
814
|
|
|
787
815
|
if (target.startsWith('http://') || target.startsWith('https://')) {
|
|
788
|
-
result = installSkillFromUrl(target)
|
|
816
|
+
result = installSkillFromUrl(target, marketplaceConfig)
|
|
789
817
|
} else {
|
|
790
|
-
result = installSkill(target)
|
|
818
|
+
result = installSkill(target, marketplaceConfig)
|
|
791
819
|
}
|
|
792
820
|
|
|
793
821
|
return {
|
|
@@ -1656,7 +1684,7 @@ function gitDiffBridgeCmd(opts: {
|
|
|
1656
1684
|
label: string
|
|
1657
1685
|
noChangesHint: string
|
|
1658
1686
|
runningMsg: string
|
|
1659
|
-
forwardToAI: string
|
|
1687
|
+
forwardToAI: string | (() => string)
|
|
1660
1688
|
}): CommandHandler {
|
|
1661
1689
|
return async () => {
|
|
1662
1690
|
try {
|
|
@@ -1667,7 +1695,7 @@ function gitDiffBridgeCmd(opts: {
|
|
|
1667
1695
|
}
|
|
1668
1696
|
return {
|
|
1669
1697
|
content: `ā ${opts.label} ā\n\n${opts.runningMsg}\n\nChanged files:\n${diff}`,
|
|
1670
|
-
forwardToAI: opts.forwardToAI,
|
|
1698
|
+
forwardToAI: typeof opts.forwardToAI === 'function' ? opts.forwardToAI() : opts.forwardToAI,
|
|
1671
1699
|
}
|
|
1672
1700
|
} catch {
|
|
1673
1701
|
return {
|
|
@@ -1683,8 +1711,8 @@ const codeReviewCmd = gitDiffBridgeCmd({
|
|
|
1683
1711
|
'No uncommitted changes to review.\n\nTo review a specific file: /code-review path/to/file.ts',
|
|
1684
1712
|
runningMsg:
|
|
1685
1713
|
'Reviewing uncommitted changes with the code-review skill (7 dimensions: correctness, security, performance, code quality, architecture, testing, language-specific)...',
|
|
1686
|
-
forwardToAI:
|
|
1687
|
-
|
|
1714
|
+
forwardToAI: () =>
|
|
1715
|
+
`use the code-review skill to review all uncommitted changes. Check all 7 dimensions: correctness, security, performance, code quality, architecture & design, testing, and language-specific issues. Use effort level: ${getPreference('lastCodeReviewEffort', 'high')}.`,
|
|
1688
1716
|
})
|
|
1689
1717
|
|
|
1690
1718
|
const simplifyCmd = gitDiffBridgeCmd({
|
|
@@ -1862,7 +1890,7 @@ const summaryCmd: CommandHandler = (ctx) => {
|
|
|
1862
1890
|
}
|
|
1863
1891
|
}
|
|
1864
1892
|
|
|
1865
|
-
const cdCmd: CommandHandler = async (
|
|
1893
|
+
const cdCmd: CommandHandler = async (ctx, args) => {
|
|
1866
1894
|
const target = args[0]
|
|
1867
1895
|
if (!target) {
|
|
1868
1896
|
return {
|
|
@@ -1887,6 +1915,22 @@ const cdCmd: CommandHandler = async (_ctx, args) => {
|
|
|
1887
1915
|
|
|
1888
1916
|
try {
|
|
1889
1917
|
process.chdir(resolved)
|
|
1918
|
+
|
|
1919
|
+
// Persist cwd to active session (best-effort)
|
|
1920
|
+
try {
|
|
1921
|
+
const { SessionStore } = await import('../core/session-store')
|
|
1922
|
+
const saved = SessionStore.load(ctx.sessionId)
|
|
1923
|
+
if (saved) {
|
|
1924
|
+
SessionStore.save(ctx.sessionId, saved.messages, {
|
|
1925
|
+
provider: saved.metadata.provider,
|
|
1926
|
+
model: saved.metadata.model,
|
|
1927
|
+
cwd: resolved,
|
|
1928
|
+
})
|
|
1929
|
+
}
|
|
1930
|
+
} catch {
|
|
1931
|
+
/* session persistence is best-effort */
|
|
1932
|
+
}
|
|
1933
|
+
|
|
1890
1934
|
return {
|
|
1891
1935
|
content: [
|
|
1892
1936
|
'āā Directory Changed āā',
|
|
@@ -2014,48 +2058,10 @@ const exportCmd: CommandHandler = async (ctx) => {
|
|
|
2014
2058
|
}
|
|
2015
2059
|
|
|
2016
2060
|
// āāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāā
|
|
2017
|
-
// Review ā code
|
|
2061
|
+
// Review ā alias for /code-review (P1: 2026-08-06 polish)
|
|
2018
2062
|
// āāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāā
|
|
2019
2063
|
|
|
2020
|
-
const reviewCmd
|
|
2021
|
-
try {
|
|
2022
|
-
const { execSync } = await import('node:child_process')
|
|
2023
|
-
const diff = execSync('git diff --stat', { encoding: 'utf-8', timeout: 5000 }).trim()
|
|
2024
|
-
const unstaged = execSync('git diff --name-only', { encoding: 'utf-8', timeout: 3000 }).trim()
|
|
2025
|
-
const staged = execSync('git diff --cached --name-only', {
|
|
2026
|
-
encoding: 'utf-8',
|
|
2027
|
-
timeout: 3000,
|
|
2028
|
-
}).trim()
|
|
2029
|
-
|
|
2030
|
-
if (!diff) {
|
|
2031
|
-
return {
|
|
2032
|
-
content:
|
|
2033
|
-
'ā Code Review ā\n\nNo uncommitted changes detected.\n\nUse /pr-comments for PR-level review, or make changes first.',
|
|
2034
|
-
}
|
|
2035
|
-
}
|
|
2036
|
-
|
|
2037
|
-
const lines: string[] = ['ā Code Review ā', '', 'Uncommitted changes:', '', diff]
|
|
2038
|
-
|
|
2039
|
-
if (staged) {
|
|
2040
|
-
lines.push('')
|
|
2041
|
-
lines.push('Staged files (ready for commit):')
|
|
2042
|
-
for (const f of staged.split('\n')) lines.push(` ā ${f}`)
|
|
2043
|
-
}
|
|
2044
|
-
if (unstaged) {
|
|
2045
|
-
lines.push('')
|
|
2046
|
-
lines.push('Unstaged files (working directory):')
|
|
2047
|
-
for (const f of unstaged.split('\n')) lines.push(` ⢠${f}`)
|
|
2048
|
-
}
|
|
2049
|
-
|
|
2050
|
-
lines.push('')
|
|
2051
|
-
lines.push('To review with AI: type "review these changes" in chat.')
|
|
2052
|
-
lines.push('To commit: git add -A && git commit -m "..."')
|
|
2053
|
-
|
|
2054
|
-
return { content: lines.join('\n') }
|
|
2055
|
-
} catch {
|
|
2056
|
-
return { content: 'ā Code Review ā\n\nCould not run git diff. Are you in a git repository?' }
|
|
2057
|
-
}
|
|
2058
|
-
}
|
|
2064
|
+
const reviewCmd = codeReviewCmd
|
|
2059
2065
|
|
|
2060
2066
|
// āāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāā
|
|
2061
2067
|
// PR Comments
|
|
@@ -4093,8 +4099,14 @@ const forkCmd: CommandHandler = async (ctx, args) => {
|
|
|
4093
4099
|
'general',
|
|
4094
4100
|
async (_signal) => {
|
|
4095
4101
|
const { SubAgent } = await import('../agent/sub-agent')
|
|
4096
|
-
const sa = new SubAgent(
|
|
4097
|
-
|
|
4102
|
+
const sa = new SubAgent(
|
|
4103
|
+
ctx.engine.getRegistry(),
|
|
4104
|
+
ctx.engine.getTools(),
|
|
4105
|
+
ctx.engine.getPermission(),
|
|
4106
|
+
)
|
|
4107
|
+
const result = await sa.execute(prompt, 'fork: ' + prompt.slice(0, 60), {
|
|
4108
|
+
worktreePath: wtPath,
|
|
4109
|
+
})
|
|
4098
4110
|
try {
|
|
4099
4111
|
const { execSync: ex } = await import('node:child_process')
|
|
4100
4112
|
ex(`git -C ${wtPath} add -A`, { stdio: 'ignore', timeout: 10_000 })
|
|
@@ -4289,7 +4301,7 @@ const commandsListCmd: CommandHandler = () => {
|
|
|
4289
4301
|
'/tdd': 'Workflow',
|
|
4290
4302
|
'/todos': 'Workflow',
|
|
4291
4303
|
'/tasks': 'Workflow',
|
|
4292
|
-
'/review': '
|
|
4304
|
+
'/review': 'Code Quality',
|
|
4293
4305
|
'/pr-comments': 'Workflow',
|
|
4294
4306
|
'/diff': 'Workflow',
|
|
4295
4307
|
'/workflows': 'Workflow',
|
|
@@ -4556,7 +4568,7 @@ const COMMAND_DESCRIPTIONS: Record<string, string> = {
|
|
|
4556
4568
|
'/tdd': 'Test-Driven Development workflow (RED ā GREEN ā REFACTOR)',
|
|
4557
4569
|
'/todos': 'Task management (list/create tasks)',
|
|
4558
4570
|
'/tasks': 'Background tasks',
|
|
4559
|
-
'/review': '
|
|
4571
|
+
'/review': 'Review code for bugs, security, performance (alias for /code-review)',
|
|
4560
4572
|
'/pr-comments': 'PR review summary',
|
|
4561
4573
|
'/diff': 'Show git diff',
|
|
4562
4574
|
'/workflows': 'List workflow scripts',
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import { SubAgent } from '../../agent/sub-agent'
|
|
2
2
|
import type { ProviderRegistry } from '../../providers/registry'
|
|
3
3
|
import type { ToolDefinition } from '../../shared/index.ts'
|
|
4
|
+
import type { PermissionSystem } from '../../core/permission'
|
|
4
5
|
import { validateJSONSchema, formatValidationErrors } from '../schema-validator'
|
|
5
6
|
|
|
6
7
|
export interface WorkflowAgentOpts {
|
|
@@ -12,17 +13,28 @@ export interface WorkflowAgentOpts {
|
|
|
12
13
|
effort?: 'low' | 'medium' | 'high' | 'max'
|
|
13
14
|
/** Maximum retries on schema validation failure (default: 2). */
|
|
14
15
|
maxRetries?: number
|
|
16
|
+
/** Permission system for sub-agent tool execution (optional). */
|
|
17
|
+
permissionSystem?: PermissionSystem
|
|
18
|
+
/** Run the agent in an isolated git worktree (optional). */
|
|
19
|
+
isolation?: 'worktree'
|
|
15
20
|
}
|
|
16
21
|
|
|
17
22
|
/**
|
|
18
23
|
* Workflow agent() primitive ā creates a SubAgent with optional
|
|
19
|
-
* provider/model override
|
|
24
|
+
* provider/model override, structured output schema validation,
|
|
25
|
+
* and git worktree isolation.
|
|
20
26
|
*
|
|
21
27
|
* When `schema` is provided:
|
|
22
28
|
* 1. The sub-agent is prompted to return valid JSON matching the schema.
|
|
23
29
|
* 2. The result is JSON.parsed and validated against the schema.
|
|
24
30
|
* 3. On validation failure, the sub-agent is retried with error feedback.
|
|
25
31
|
* 4. Returns the validated object, or { raw, validationErrors } on final failure.
|
|
32
|
+
*
|
|
33
|
+
* When `isolation: 'worktree'` is set:
|
|
34
|
+
* 1. A git worktree is created at .claude/worktrees/wf-<slug>
|
|
35
|
+
* 2. The sub-agent runs with its cwd set to the worktree path
|
|
36
|
+
* 3. Changes are auto-committed (best-effort)
|
|
37
|
+
* 4. The worktree is cleaned up after execution
|
|
26
38
|
*/
|
|
27
39
|
export async function workflowAgent(
|
|
28
40
|
prompt: string,
|
|
@@ -39,6 +51,26 @@ export async function workflowAgent(
|
|
|
39
51
|
|
|
40
52
|
const maxRetries = opts.maxRetries ?? 2
|
|
41
53
|
|
|
54
|
+
// āā Worktree isolation setup āā
|
|
55
|
+
let worktreePath: string | undefined
|
|
56
|
+
let worktreeBranch: string | undefined
|
|
57
|
+
|
|
58
|
+
if (opts.isolation === 'worktree') {
|
|
59
|
+
const slug = `wf-${Date.now().toString(36)}-${Math.random().toString(36).slice(2, 6)}`
|
|
60
|
+
worktreeBranch = `worktree/${slug}`
|
|
61
|
+
worktreePath = `.claude/worktrees/${slug}`
|
|
62
|
+
|
|
63
|
+
const proc = Bun.spawn(['git', 'worktree', 'add', '-b', worktreeBranch, worktreePath, 'HEAD'], {
|
|
64
|
+
stdout: 'pipe',
|
|
65
|
+
stderr: 'pipe',
|
|
66
|
+
})
|
|
67
|
+
const exitCode = await proc.exited
|
|
68
|
+
if (exitCode !== 0) {
|
|
69
|
+
const stderr = await new Response(proc.stderr).text()
|
|
70
|
+
throw new Error(`Worktree creation failed: ${stderr.slice(0, 500)}`)
|
|
71
|
+
}
|
|
72
|
+
}
|
|
73
|
+
|
|
42
74
|
// Build the prompt ā if schema is requested, instruct the model to return JSON
|
|
43
75
|
let effectivePrompt = prompt
|
|
44
76
|
if (opts.schema) {
|
|
@@ -50,64 +82,118 @@ export async function workflowAgent(
|
|
|
50
82
|
`Return ONLY the JSON object, no other text. Do not wrap in markdown code fences.`
|
|
51
83
|
}
|
|
52
84
|
|
|
53
|
-
const sub = new SubAgent(registry, toolRegistry)
|
|
85
|
+
const sub = new SubAgent(registry, toolRegistry, opts.permissionSystem)
|
|
54
86
|
|
|
55
|
-
// āā
|
|
87
|
+
// āā Execute with optional schema validation + retry āā
|
|
88
|
+
let result: unknown = ''
|
|
56
89
|
let lastResult = ''
|
|
57
90
|
let lastErrors: string[] = []
|
|
58
91
|
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
const result = await sub.execute(retryPrompt, opts.label || 'workflow-agent', {
|
|
69
|
-
type: 'general',
|
|
70
|
-
modelOverride: opts.model,
|
|
71
|
-
allowedTools: undefined, // use all tools by default
|
|
72
|
-
})
|
|
92
|
+
try {
|
|
93
|
+
for (let attempt = 0; attempt <= maxRetries; attempt++) {
|
|
94
|
+
const retryPrompt =
|
|
95
|
+
attempt === 0
|
|
96
|
+
? effectivePrompt
|
|
97
|
+
: `${effectivePrompt}\n\n` +
|
|
98
|
+
`[RETRY #${attempt}] Your previous response did NOT match the required schema.\n` +
|
|
99
|
+
`Validation errors:\n${lastErrors.map((e) => ` ⢠${e}`).join('\n')}\n\n` +
|
|
100
|
+
`Please fix the errors and return a valid JSON object matching the schema exactly.`
|
|
73
101
|
|
|
74
|
-
|
|
102
|
+
const textResult = await sub.execute(retryPrompt, opts.label || 'workflow-agent', {
|
|
103
|
+
type: 'general',
|
|
104
|
+
modelOverride: opts.model,
|
|
105
|
+
allowedTools: undefined, // use all tools by default
|
|
106
|
+
worktreePath,
|
|
107
|
+
})
|
|
75
108
|
|
|
76
|
-
|
|
77
|
-
if (!opts.schema) {
|
|
78
|
-
return result
|
|
79
|
-
}
|
|
109
|
+
lastResult = textResult
|
|
80
110
|
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
const fenceMatch = jsonStr.match(/```(?:json)?\s*\n?([\s\S]*?)\n?```/)
|
|
86
|
-
if (fenceMatch) {
|
|
87
|
-
jsonStr = fenceMatch[1]!.trim()
|
|
111
|
+
// No schema ā return raw result
|
|
112
|
+
if (!opts.schema) {
|
|
113
|
+
result = textResult
|
|
114
|
+
return result
|
|
88
115
|
}
|
|
89
116
|
|
|
90
|
-
|
|
91
|
-
|
|
117
|
+
// Try to parse and validate
|
|
118
|
+
try {
|
|
119
|
+
// Extract JSON from potential markdown fences
|
|
120
|
+
let jsonStr = textResult.trim()
|
|
121
|
+
const fenceMatch = jsonStr.match(/```(?:json)?\s*\n?([\s\S]*?)\n?```/)
|
|
122
|
+
if (fenceMatch) {
|
|
123
|
+
jsonStr = fenceMatch[1]!.trim()
|
|
124
|
+
}
|
|
92
125
|
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
126
|
+
const parsed = JSON.parse(jsonStr)
|
|
127
|
+
const errors = validateJSONSchema(parsed, opts.schema as Record<string, unknown>)
|
|
128
|
+
|
|
129
|
+
if (errors.length === 0) {
|
|
130
|
+
// Valid! Return the parsed object
|
|
131
|
+
result = parsed
|
|
132
|
+
return result
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
// Not valid ā collect errors for retry
|
|
136
|
+
lastErrors = [formatValidationErrors(errors)]
|
|
137
|
+
} catch (err) {
|
|
138
|
+
// JSON parse error
|
|
139
|
+
lastErrors = [`JSON parse error: ${String(err)}. Ensure your response is valid JSON only.`]
|
|
96
140
|
}
|
|
141
|
+
}
|
|
97
142
|
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
143
|
+
// All retries exhausted ā return raw result with validation errors
|
|
144
|
+
try {
|
|
145
|
+
const parsed = JSON.parse(lastResult)
|
|
146
|
+
result = { raw: lastResult, parsed, validationErrors: lastErrors }
|
|
147
|
+
} catch {
|
|
148
|
+
result = { raw: lastResult, validationErrors: lastErrors }
|
|
103
149
|
}
|
|
104
|
-
}
|
|
105
150
|
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
151
|
+
return result
|
|
152
|
+
} finally {
|
|
153
|
+
// āā Cleanup worktree āā
|
|
154
|
+
if (worktreePath) {
|
|
155
|
+
// Best-effort: auto-commit any changes
|
|
156
|
+
try {
|
|
157
|
+
const addProc = Bun.spawn(['git', '-C', worktreePath, 'add', '-A'], {
|
|
158
|
+
stdout: 'pipe',
|
|
159
|
+
stderr: 'pipe',
|
|
160
|
+
})
|
|
161
|
+
await addProc.exited
|
|
162
|
+
|
|
163
|
+
const statusProc = Bun.spawn(['git', '-C', worktreePath, 'status', '--porcelain'], {
|
|
164
|
+
stdout: 'pipe',
|
|
165
|
+
stderr: 'pipe',
|
|
166
|
+
})
|
|
167
|
+
const status = await new Response(statusProc.stdout).text()
|
|
168
|
+
if (status.trim()) {
|
|
169
|
+
const commitMsg = opts.label ? `workflow: ${opts.label}` : 'workflow: agent changes'
|
|
170
|
+
const commitProc = Bun.spawn(['git', '-C', worktreePath, 'commit', '-m', commitMsg], {
|
|
171
|
+
stdout: 'pipe',
|
|
172
|
+
stderr: 'pipe',
|
|
173
|
+
})
|
|
174
|
+
await commitProc.exited
|
|
175
|
+
}
|
|
176
|
+
} catch {
|
|
177
|
+
/* best-effort */
|
|
178
|
+
}
|
|
179
|
+
|
|
180
|
+
// Remove worktree and branch
|
|
181
|
+
try {
|
|
182
|
+
const rmProc = Bun.spawn(['git', 'worktree', 'remove', '--force', worktreePath], {
|
|
183
|
+
stdout: 'pipe',
|
|
184
|
+
stderr: 'pipe',
|
|
185
|
+
})
|
|
186
|
+
await rmProc.exited
|
|
187
|
+
if (worktreeBranch) {
|
|
188
|
+
const brProc = Bun.spawn(['git', 'branch', '-D', worktreeBranch], {
|
|
189
|
+
stdout: 'pipe',
|
|
190
|
+
stderr: 'pipe',
|
|
191
|
+
})
|
|
192
|
+
await brProc.exited
|
|
193
|
+
}
|
|
194
|
+
} catch {
|
|
195
|
+
/* best-effort */
|
|
196
|
+
}
|
|
197
|
+
}
|
|
112
198
|
}
|
|
113
199
|
}
|
package/src/workflow/runtime.ts
CHANGED
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import vm from 'node:vm'
|
|
1
2
|
import { createSandbox } from './sandbox'
|
|
2
3
|
import { createJournal, appendJournal, loadJournal, loadScript } from './journal'
|
|
3
4
|
import { createBudget } from './budget'
|
|
@@ -47,7 +48,8 @@ function agentCacheKey(prompt: string, opts: Record<string, unknown> = {}): stri
|
|
|
47
48
|
* - Returns cacheHits and cacheMisses counts
|
|
48
49
|
*
|
|
49
50
|
* The sandbox blocks non-deterministic APIs (Date.now, Math.random) to make
|
|
50
|
-
* this deterministic replay possible.
|
|
51
|
+
* this deterministic replay possible. Scripts execute inside a node:vm.Context
|
|
52
|
+
* that blocks sandbox escape vectors (eval, Function, import, require, process, etc.).
|
|
51
53
|
*/
|
|
52
54
|
export async function runWorkflow(
|
|
53
55
|
script: string,
|
|
@@ -96,6 +98,7 @@ export async function runWorkflow(
|
|
|
96
98
|
|
|
97
99
|
const registry: ProviderRegistry = engine.getRegistry()
|
|
98
100
|
const toolRegistry = engine.getTools()
|
|
101
|
+
const permission = engine.getPermission()
|
|
99
102
|
|
|
100
103
|
const budget = createBudget(budgetTotal)
|
|
101
104
|
|
|
@@ -111,12 +114,10 @@ export async function runWorkflow(
|
|
|
111
114
|
}
|
|
112
115
|
cacheMisses++
|
|
113
116
|
|
|
114
|
-
const result = await workflowAgent(
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
opts as Record<string, unknown>,
|
|
119
|
-
)
|
|
117
|
+
const result = await workflowAgent(prompt, registry, toolRegistry, {
|
|
118
|
+
...(opts || {}),
|
|
119
|
+
permissionSystem: permission,
|
|
120
|
+
} as Record<string, unknown>)
|
|
120
121
|
appendJournal(runId, {
|
|
121
122
|
type: 'agent',
|
|
122
123
|
prompt,
|
|
@@ -135,42 +136,28 @@ export async function runWorkflow(
|
|
|
135
136
|
appendJournal(runId, { type: 'log', message })
|
|
136
137
|
}
|
|
137
138
|
|
|
138
|
-
const sandbox = createSandbox(args, budget)
|
|
139
|
-
|
|
140
139
|
// Build the script wrapper
|
|
141
140
|
const wrappedScript = `
|
|
142
|
-
|
|
141
|
+
(async () => {
|
|
143
142
|
${script}
|
|
144
143
|
})()
|
|
145
144
|
`
|
|
146
145
|
|
|
147
|
-
// Execute in
|
|
148
|
-
const
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
const result = await scriptFn(
|
|
163
|
-
agent,
|
|
164
|
-
parallel,
|
|
165
|
-
pipeline,
|
|
166
|
-
verify,
|
|
167
|
-
judge,
|
|
168
|
-
loopUntilConvergence,
|
|
169
|
-
wrappedPhase,
|
|
170
|
-
log,
|
|
171
|
-
args,
|
|
172
|
-
budget,
|
|
173
|
-
)
|
|
146
|
+
// Execute in node:vm sandbox ā no access to host globals
|
|
147
|
+
const vmScript = new vm.Script(wrappedScript, { filename: 'workflow.js' })
|
|
148
|
+
const sandboxCtx = createSandbox(args, budget)
|
|
149
|
+
|
|
150
|
+
// Inject primitives into the sandbox context (whitelist approach)
|
|
151
|
+
sandboxCtx.agent = agent
|
|
152
|
+
sandboxCtx.parallel = parallel
|
|
153
|
+
sandboxCtx.pipeline = pipeline
|
|
154
|
+
sandboxCtx.verify = verify
|
|
155
|
+
sandboxCtx.judge = judge
|
|
156
|
+
sandboxCtx.loopUntilConvergence = loopUntilConvergence
|
|
157
|
+
sandboxCtx.phase = wrappedPhase
|
|
158
|
+
sandboxCtx.log = log
|
|
159
|
+
|
|
160
|
+
const result = await vmScript.runInContext(sandboxCtx, { timeout: 120_000 })
|
|
174
161
|
|
|
175
162
|
// Count journal entries from state
|
|
176
163
|
const priorEntries = loadJournal(runId)
|
package/src/workflow/sandbox.ts
CHANGED
|
@@ -1,27 +1,39 @@
|
|
|
1
|
-
|
|
1
|
+
import vm from 'node:vm'
|
|
2
|
+
|
|
3
|
+
/** APIs disabled in workflow scripts to ensure deterministic replay + sandbox escape prevention. */
|
|
2
4
|
const FORBIDDEN = new Set(['Date.now', 'Math.random', 'crypto.randomUUID'])
|
|
3
5
|
|
|
4
6
|
/**
|
|
5
|
-
* Create a sandboxed
|
|
6
|
-
* Blocks Date.now(), Math.random(), argless new Date(), crypto.randomUUID()
|
|
7
|
+
* Create a sandboxed VM context for workflow script execution.
|
|
8
|
+
* Blocks Date.now(), Math.random(), argless new Date(), crypto.randomUUID(),
|
|
9
|
+
* plus explicit sandbox escape vectors: eval, Function constructor, import(),
|
|
10
|
+
* require(), process, Bun, fetch, setTimeout/setInterval, etc.
|
|
7
11
|
*/
|
|
8
12
|
export function createSandbox(
|
|
9
13
|
args: unknown,
|
|
10
14
|
budget: { total: number | null; spent(): number; remaining(): number },
|
|
11
|
-
):
|
|
12
|
-
const
|
|
15
|
+
): vm.Context {
|
|
16
|
+
const sandboxObj: Record<string, unknown> = {
|
|
13
17
|
args,
|
|
14
18
|
budget,
|
|
15
19
|
console: {
|
|
16
20
|
log: (..._a: unknown[]) => {}, // no-op in sandbox
|
|
17
21
|
error: (..._a: unknown[]) => {},
|
|
18
22
|
},
|
|
19
|
-
|
|
23
|
+
|
|
24
|
+
// āā Explicit sandbox escape prevention āā
|
|
25
|
+
// V8 builtins that would otherwise be available in a vm.Context
|
|
26
|
+
eval: () => {
|
|
27
|
+
throw new Error('eval() is disabled in workflow sandbox.')
|
|
28
|
+
},
|
|
29
|
+
Function: () => {
|
|
30
|
+
throw new Error('new Function() is disabled in workflow sandbox.')
|
|
31
|
+
},
|
|
20
32
|
}
|
|
21
33
|
|
|
22
34
|
// Override Date to block now() and argless constructor
|
|
23
35
|
const OriginalDate = Date
|
|
24
|
-
|
|
36
|
+
sandboxObj.Date = new Proxy(OriginalDate, {
|
|
25
37
|
construct(_target, constructorArgs) {
|
|
26
38
|
if (constructorArgs.length === 0) {
|
|
27
39
|
throw new Error('new Date() is disabled in workflow sandbox. Pass timestamps via args.')
|
|
@@ -42,7 +54,7 @@ export function createSandbox(
|
|
|
42
54
|
})
|
|
43
55
|
|
|
44
56
|
// Override Math.random
|
|
45
|
-
|
|
57
|
+
sandboxObj.Math = new Proxy(Math, {
|
|
46
58
|
get(_target, prop) {
|
|
47
59
|
if (prop === 'random') {
|
|
48
60
|
throw new Error('Math.random() is disabled in workflow sandbox. Use a seed from args.')
|
|
@@ -57,7 +69,7 @@ export function createSandbox(
|
|
|
57
69
|
| { randomUUID?: unknown; [key: string]: unknown }
|
|
58
70
|
| undefined
|
|
59
71
|
if (globalCrypto) {
|
|
60
|
-
|
|
72
|
+
sandboxObj.crypto = new Proxy(globalCrypto, {
|
|
61
73
|
get(_target, prop) {
|
|
62
74
|
if (prop === 'randomUUID') {
|
|
63
75
|
throw new Error('crypto.randomUUID() is disabled in workflow sandbox.')
|
|
@@ -70,7 +82,7 @@ export function createSandbox(
|
|
|
70
82
|
})
|
|
71
83
|
}
|
|
72
84
|
|
|
73
|
-
return
|
|
85
|
+
return vm.createContext(sandboxObj)
|
|
74
86
|
}
|
|
75
87
|
|
|
76
88
|
/** Check whether a given API identifier is in the forbidden set. */
|