@tanstack/ai-sandbox 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +182 -0
- package/dist/esm/agents-file.d.ts +36 -0
- package/dist/esm/agents-file.js +44 -0
- package/dist/esm/agents-file.js.map +1 -0
- package/dist/esm/approvals.d.ts +38 -0
- package/dist/esm/approvals.js +36 -0
- package/dist/esm/approvals.js.map +1 -0
- package/dist/esm/bootstrap.d.ts +17 -0
- package/dist/esm/bootstrap.js +124 -0
- package/dist/esm/bootstrap.js.map +1 -0
- package/dist/esm/bridge-events.d.ts +21 -0
- package/dist/esm/bridge-events.js +76 -0
- package/dist/esm/bridge-events.js.map +1 -0
- package/dist/esm/capabilities.d.ts +26 -0
- package/dist/esm/capabilities.js +29 -0
- package/dist/esm/capabilities.js.map +1 -0
- package/dist/esm/contracts.d.ts +211 -0
- package/dist/esm/errors.d.ts +16 -0
- package/dist/esm/errors.js +25 -0
- package/dist/esm/errors.js.map +1 -0
- package/dist/esm/git-exec.d.ts +2 -0
- package/dist/esm/git-exec.js +68 -0
- package/dist/esm/git-exec.js.map +1 -0
- package/dist/esm/harness-cwd.d.ts +2 -0
- package/dist/esm/harness-cwd.js +24 -0
- package/dist/esm/harness-cwd.js.map +1 -0
- package/dist/esm/index.d.ts +39 -0
- package/dist/esm/index.js +103 -0
- package/dist/esm/index.js.map +1 -0
- package/dist/esm/key.d.ts +20 -0
- package/dist/esm/key.js +41 -0
- package/dist/esm/key.js.map +1 -0
- package/dist/esm/middleware.d.ts +5 -0
- package/dist/esm/middleware.js +140 -0
- package/dist/esm/middleware.js.map +1 -0
- package/dist/esm/ngrok.d.ts +16 -0
- package/dist/esm/ngrok.js +54 -0
- package/dist/esm/ngrok.js.map +1 -0
- package/dist/esm/policy.d.ts +47 -0
- package/dist/esm/policy.js +44 -0
- package/dist/esm/policy.js.map +1 -0
- package/dist/esm/projection.d.ts +31 -0
- package/dist/esm/projection.js +9 -0
- package/dist/esm/projection.js.map +1 -0
- package/dist/esm/remote-tools.d.ts +48 -0
- package/dist/esm/remote-tools.js +76 -0
- package/dist/esm/remote-tools.js.map +1 -0
- package/dist/esm/run-log.d.ts +81 -0
- package/dist/esm/run-log.js +107 -0
- package/dist/esm/run-log.js.map +1 -0
- package/dist/esm/run.d.ts +58 -0
- package/dist/esm/run.js +89 -0
- package/dist/esm/run.js.map +1 -0
- package/dist/esm/runner.d.ts +21 -0
- package/dist/esm/runner.js +54 -0
- package/dist/esm/runner.js.map +1 -0
- package/dist/esm/sandbox.d.ts +79 -0
- package/dist/esm/sandbox.js +125 -0
- package/dist/esm/sandbox.js.map +1 -0
- package/dist/esm/secrets.d.ts +37 -0
- package/dist/esm/secrets.js +59 -0
- package/dist/esm/secrets.js.map +1 -0
- package/dist/esm/setup-plan.d.ts +13 -0
- package/dist/esm/setup-plan.js +16 -0
- package/dist/esm/setup-plan.js.map +1 -0
- package/dist/esm/shell.d.ts +45 -0
- package/dist/esm/shell.js +164 -0
- package/dist/esm/shell.js.map +1 -0
- package/dist/esm/store.d.ts +53 -0
- package/dist/esm/store.js +34 -0
- package/dist/esm/store.js.map +1 -0
- package/dist/esm/tool-bridge.d.ts +130 -0
- package/dist/esm/tool-bridge.js +197 -0
- package/dist/esm/tool-bridge.js.map +1 -0
- package/dist/esm/watch.d.ts +36 -0
- package/dist/esm/watch.js +144 -0
- package/dist/esm/watch.js.map +1 -0
- package/dist/esm/workspace.d.ts +128 -0
- package/dist/esm/workspace.js +42 -0
- package/dist/esm/workspace.js.map +1 -0
- package/package.json +72 -0
- package/skills/ai-sandbox/SKILL.md +366 -0
- package/src/agents-file.ts +101 -0
- package/src/approvals.ts +96 -0
- package/src/bootstrap.ts +196 -0
- package/src/bridge-events.ts +112 -0
- package/src/capabilities.ts +47 -0
- package/src/contracts.ts +236 -0
- package/src/errors.ts +31 -0
- package/src/git-exec.ts +114 -0
- package/src/harness-cwd.ts +38 -0
- package/src/index.ts +222 -0
- package/src/key.ts +70 -0
- package/src/middleware.ts +233 -0
- package/src/ngrok.ts +85 -0
- package/src/policy.ts +111 -0
- package/src/projection.ts +46 -0
- package/src/remote-tools.ts +180 -0
- package/src/run-log.ts +224 -0
- package/src/run.ts +167 -0
- package/src/runner.ts +99 -0
- package/src/sandbox.ts +259 -0
- package/src/secrets.ts +101 -0
- package/src/setup-plan.ts +25 -0
- package/src/shell.ts +288 -0
- package/src/store.ts +83 -0
- package/src/tool-bridge.ts +399 -0
- package/src/watch.ts +256 -0
- package/src/workspace.ts +151 -0
package/src/key.ts
ADDED
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Compound sandbox identity. We never key a resumable sandbox on `threadId`
|
|
3
|
+
* alone — that would resume the WRONG environment when the provider,
|
|
4
|
+
* workspace, image, or tenant changes. The key folds all of those in, so any
|
|
5
|
+
* change busts the sandbox and forces a fresh create+bootstrap (safe default).
|
|
6
|
+
*/
|
|
7
|
+
import type { WorkspaceDefinition } from './workspace'
|
|
8
|
+
|
|
9
|
+
/** Inputs that, together, identify one resumable sandbox instance. */
|
|
10
|
+
export interface SandboxKeyInput {
|
|
11
|
+
threadId: string
|
|
12
|
+
sandboxId: string
|
|
13
|
+
providerName: string
|
|
14
|
+
workspace?: WorkspaceDefinition
|
|
15
|
+
/** Optional tenant scoping pulled from runtimeContext. */
|
|
16
|
+
tenant?: { userId?: string; orgId?: string }
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
/** Deterministic, dependency-free 64-bit FNV-1a hash → hex string. */
|
|
20
|
+
function fnv1a(input: string): string {
|
|
21
|
+
// Two 32-bit lanes to approximate 64-bit without BigInt overhead concerns.
|
|
22
|
+
let h1 = 0x811c9dc5
|
|
23
|
+
let h2 = 0x811c9dc5
|
|
24
|
+
for (let i = 0; i < input.length; i++) {
|
|
25
|
+
const c = input.charCodeAt(i)
|
|
26
|
+
h1 ^= c & 0xff
|
|
27
|
+
h1 = Math.imul(h1, 0x01000193)
|
|
28
|
+
h2 ^= (c >> 8) & 0xff
|
|
29
|
+
h2 = Math.imul(h2, 0x01000193)
|
|
30
|
+
}
|
|
31
|
+
const hex = (n: number): string => (n >>> 0).toString(16).padStart(8, '0')
|
|
32
|
+
return hex(h1) + hex(h2)
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
/** Canonical, key-sorted JSON so logically-equal inputs hash identically. */
|
|
36
|
+
function canonical(value: unknown): string {
|
|
37
|
+
if (value === null || typeof value !== 'object') return JSON.stringify(value)
|
|
38
|
+
if (Array.isArray(value)) return `[${value.map(canonical).join(',')}]`
|
|
39
|
+
const keys = Object.keys(value).sort()
|
|
40
|
+
return `{${keys
|
|
41
|
+
.map(
|
|
42
|
+
(k) =>
|
|
43
|
+
`${JSON.stringify(k)}:${canonical((value as Record<string, unknown>)[k])}`,
|
|
44
|
+
)
|
|
45
|
+
.join(',')}}`
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
/**
|
|
49
|
+
* Hash of the parts of a workspace that change what the agent sees. Secrets are
|
|
50
|
+
* intentionally excluded (rotating a token must not orphan the sandbox).
|
|
51
|
+
*/
|
|
52
|
+
export function computeWorkspaceHash(
|
|
53
|
+
workspace: WorkspaceDefinition | undefined,
|
|
54
|
+
): string {
|
|
55
|
+
if (!workspace) return fnv1a('no-workspace')
|
|
56
|
+
const { secrets: _secrets, ...rest } = workspace
|
|
57
|
+
return fnv1a(canonical(rest))
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
/** Compute the compound sandbox instance key. */
|
|
61
|
+
export function computeSandboxKey(input: SandboxKeyInput): string {
|
|
62
|
+
const material = canonical({
|
|
63
|
+
threadId: input.threadId,
|
|
64
|
+
sandboxId: input.sandboxId,
|
|
65
|
+
providerName: input.providerName,
|
|
66
|
+
workspaceHash: computeWorkspaceHash(input.workspace),
|
|
67
|
+
tenant: input.tenant ?? null,
|
|
68
|
+
})
|
|
69
|
+
return fnv1a(material)
|
|
70
|
+
}
|
|
@@ -0,0 +1,233 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `withSandbox(definition)` — the middleware that PROVIDES the
|
|
3
|
+
* {@link SandboxCapability} a harness adapter requires.
|
|
4
|
+
*
|
|
5
|
+
* - `setup`: resume-or-create the sandbox (via the definition's ensure
|
|
6
|
+
* algorithm), provide the handle, using the optional SandboxStore/Locks
|
|
7
|
+
* capabilities when a persistence middleware supplied them (in-memory
|
|
8
|
+
* fallback otherwise). If `fileEvents` is not false, starts a watcher
|
|
9
|
+
* that dispatches to sandbox-scoped hooks and forwards to the runtime sink.
|
|
10
|
+
* - `onFinish`/`onAbort`/`onError`: stop the watcher, snapshot (`after-run`)
|
|
11
|
+
* and/or destroy per lifecycle.
|
|
12
|
+
*
|
|
13
|
+
* NOTE: streamed sandbox lifecycle events (sandbox.created, workspace.setup.*)
|
|
14
|
+
* are emitted by the harness adapter's chatStream (which can yield CUSTOM
|
|
15
|
+
* chunks), not from here — middleware setup runs before streaming begins.
|
|
16
|
+
*/
|
|
17
|
+
import { defineChatMiddleware } from '@tanstack/ai'
|
|
18
|
+
import { getSandboxRuntime } from '@tanstack/ai/adapter-internals'
|
|
19
|
+
import {
|
|
20
|
+
LocksCapability,
|
|
21
|
+
SandboxCapability,
|
|
22
|
+
SandboxStoreCapability,
|
|
23
|
+
provideSandbox,
|
|
24
|
+
provideSandboxPolicy,
|
|
25
|
+
} from './capabilities'
|
|
26
|
+
import { computeWorkspaceHash } from './key'
|
|
27
|
+
import { ProjectionCapability, provideWorkspaceProjection } from './projection'
|
|
28
|
+
import { resolveSecret } from './secrets'
|
|
29
|
+
import { watchWorkspace } from './watch'
|
|
30
|
+
import { DEFAULT_WORKSPACE_ROOT } from './bootstrap'
|
|
31
|
+
import type {
|
|
32
|
+
AbortInfo,
|
|
33
|
+
ChatMiddlewareContext,
|
|
34
|
+
DefinedChatMiddleware,
|
|
35
|
+
SandboxFileEvent,
|
|
36
|
+
} from '@tanstack/ai'
|
|
37
|
+
import type { SandboxHandle } from './contracts'
|
|
38
|
+
import type {
|
|
39
|
+
SandboxDefinition,
|
|
40
|
+
SandboxEnsureContext,
|
|
41
|
+
SandboxHooks,
|
|
42
|
+
} from './sandbox'
|
|
43
|
+
import type { SandboxWatchHandle } from './watch'
|
|
44
|
+
|
|
45
|
+
/** Per-request state we need to carry from `setup` to the terminal hooks. */
|
|
46
|
+
interface SandboxRunState {
|
|
47
|
+
handle: SandboxHandle
|
|
48
|
+
ensureCtx: SandboxEnsureContext
|
|
49
|
+
watcher?: SandboxWatchHandle
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
const runState = new WeakMap<object, SandboxRunState>()
|
|
53
|
+
|
|
54
|
+
/** Defensively pull tenant scoping out of the runtime context, if present. */
|
|
55
|
+
function tenantFrom(
|
|
56
|
+
context: unknown,
|
|
57
|
+
): { userId?: string; orgId?: string } | undefined {
|
|
58
|
+
if (context === null || typeof context !== 'object') return undefined
|
|
59
|
+
const c = context as Record<string, unknown>
|
|
60
|
+
const userId = typeof c.userId === 'string' ? c.userId : undefined
|
|
61
|
+
const orgId = typeof c.orgId === 'string' ? c.orgId : undefined
|
|
62
|
+
if (userId === undefined && orgId === undefined) return undefined
|
|
63
|
+
return { userId, orgId }
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
function buildEnsureCtx(ctx: ChatMiddlewareContext): SandboxEnsureContext {
|
|
67
|
+
return {
|
|
68
|
+
threadId: ctx.threadId,
|
|
69
|
+
runId: ctx.runId,
|
|
70
|
+
store: ctx.getOptional(SandboxStoreCapability),
|
|
71
|
+
locks: ctx.getOptional(LocksCapability),
|
|
72
|
+
tenant: tenantFrom(ctx.context),
|
|
73
|
+
signal: ctx.signal,
|
|
74
|
+
}
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
/**
|
|
78
|
+
* Dispatch a sandbox file event to the per-type hooks declared on the
|
|
79
|
+
* definition. Errors in individual hooks are swallowed so one bad hook
|
|
80
|
+
* cannot break the run.
|
|
81
|
+
*/
|
|
82
|
+
async function dispatchDefinitionHooks(
|
|
83
|
+
hooks: SandboxHooks | undefined,
|
|
84
|
+
event: SandboxFileEvent,
|
|
85
|
+
): Promise<void> {
|
|
86
|
+
if (!hooks) return
|
|
87
|
+
const typed = (
|
|
88
|
+
{
|
|
89
|
+
create: 'onFileCreate',
|
|
90
|
+
change: 'onFileChange',
|
|
91
|
+
delete: 'onFileDelete',
|
|
92
|
+
} as const
|
|
93
|
+
)[event.type]
|
|
94
|
+
for (const fn of [hooks.onFile, hooks[typed]]) {
|
|
95
|
+
if (!fn) continue
|
|
96
|
+
try {
|
|
97
|
+
await fn(event)
|
|
98
|
+
} catch {
|
|
99
|
+
// swallowed — one bad hook must not break the run
|
|
100
|
+
}
|
|
101
|
+
}
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
export function withSandbox(
|
|
105
|
+
definition: SandboxDefinition,
|
|
106
|
+
): DefinedChatMiddleware<
|
|
107
|
+
unknown,
|
|
108
|
+
readonly [],
|
|
109
|
+
readonly [typeof SandboxCapability, typeof ProjectionCapability]
|
|
110
|
+
> {
|
|
111
|
+
return defineChatMiddleware({
|
|
112
|
+
name: 'sandbox',
|
|
113
|
+
provides: [SandboxCapability, ProjectionCapability],
|
|
114
|
+
// SandboxPolicyCapability is provided conditionally (only when the
|
|
115
|
+
// definition has a policy), so it is intentionally NOT declared here —
|
|
116
|
+
// consumers read it via `getOptional`.
|
|
117
|
+
optionalRequires: [SandboxStoreCapability, LocksCapability],
|
|
118
|
+
|
|
119
|
+
async setup(ctx) {
|
|
120
|
+
const ensureCtx = buildEnsureCtx(ctx)
|
|
121
|
+
const handle = await definition.ensure(ensureCtx)
|
|
122
|
+
provideSandbox(ctx, handle)
|
|
123
|
+
if (definition.policy) provideSandboxPolicy(ctx, definition.policy)
|
|
124
|
+
|
|
125
|
+
const workspace = definition.workspace
|
|
126
|
+
if (workspace !== undefined) {
|
|
127
|
+
const root = workspace.root ?? DEFAULT_WORKSPACE_ROOT
|
|
128
|
+
const workspaceHash = computeWorkspaceHash(workspace)
|
|
129
|
+
const secrets = workspace.secrets
|
|
130
|
+
provideWorkspaceProjection(ctx, {
|
|
131
|
+
skills: workspace.skills ?? [],
|
|
132
|
+
plugins: workspace.plugins ?? [],
|
|
133
|
+
resolveSecret: (ref) => {
|
|
134
|
+
if (secrets === undefined) {
|
|
135
|
+
throw new Error(
|
|
136
|
+
`resolveSecret: no secrets defined on this workspace (ref: "${ref.__secretName}")`,
|
|
137
|
+
)
|
|
138
|
+
}
|
|
139
|
+
return resolveSecret(secrets, ref)
|
|
140
|
+
},
|
|
141
|
+
markerPath: `${root}/.tanstack-projected-${workspaceHash}`,
|
|
142
|
+
root,
|
|
143
|
+
...(workspace.scripts !== undefined
|
|
144
|
+
? { scripts: workspace.scripts }
|
|
145
|
+
: {}),
|
|
146
|
+
})
|
|
147
|
+
}
|
|
148
|
+
|
|
149
|
+
const hooks = definition.hooks
|
|
150
|
+
await hooks?.onReady?.(handle)
|
|
151
|
+
|
|
152
|
+
let watcher: SandboxWatchHandle | undefined
|
|
153
|
+
if (definition.fileEvents !== false) {
|
|
154
|
+
const runtime = getSandboxRuntime(ctx, { optional: true })
|
|
155
|
+
watcher = await watchWorkspace(handle, {
|
|
156
|
+
onEvent: (event: SandboxFileEvent) => {
|
|
157
|
+
void dispatchDefinitionHooks(hooks, event)
|
|
158
|
+
runtime?.emit(event)
|
|
159
|
+
},
|
|
160
|
+
...(ctx.signal !== undefined ? { signal: ctx.signal } : {}),
|
|
161
|
+
})
|
|
162
|
+
}
|
|
163
|
+
|
|
164
|
+
runState.set(ctx, { handle, ensureCtx, ...(watcher ? { watcher } : {}) })
|
|
165
|
+
},
|
|
166
|
+
|
|
167
|
+
async onFinish(ctx) {
|
|
168
|
+
const state = runState.get(ctx)
|
|
169
|
+
if (!state) return
|
|
170
|
+
const { handle, ensureCtx } = state
|
|
171
|
+
|
|
172
|
+
await state.watcher?.stop()
|
|
173
|
+
|
|
174
|
+
const lifecycle = definition.lifecycle
|
|
175
|
+
|
|
176
|
+
if (
|
|
177
|
+
lifecycle?.snapshot === 'after-run' &&
|
|
178
|
+
handle.capabilities.snapshots &&
|
|
179
|
+
handle.snapshot
|
|
180
|
+
) {
|
|
181
|
+
const snapshot = await handle.snapshot(`after-run-${ctx.runId}`)
|
|
182
|
+
const store = ensureCtx.store
|
|
183
|
+
if (store) {
|
|
184
|
+
const key = definition.key(ensureCtx)
|
|
185
|
+
const existing = await store.get(key)
|
|
186
|
+
if (existing) {
|
|
187
|
+
await store.upsert({
|
|
188
|
+
...existing,
|
|
189
|
+
latestSnapshotId: snapshot.id,
|
|
190
|
+
updatedAt: Date.now(),
|
|
191
|
+
})
|
|
192
|
+
}
|
|
193
|
+
}
|
|
194
|
+
}
|
|
195
|
+
|
|
196
|
+
if (lifecycle?.destroyOnComplete) {
|
|
197
|
+
await definition.destroy(ensureCtx)
|
|
198
|
+
await definition.hooks?.onDestroy?.()
|
|
199
|
+
}
|
|
200
|
+
},
|
|
201
|
+
|
|
202
|
+
async onAbort(ctx, _info: AbortInfo) {
|
|
203
|
+
const state = runState.get(ctx)
|
|
204
|
+
if (!state) return
|
|
205
|
+
|
|
206
|
+
await state.watcher?.stop()
|
|
207
|
+
|
|
208
|
+
// ALWAYS tear down on an explicit abort, regardless of `destroyOnComplete`.
|
|
209
|
+
// The in-sandbox agent process is not killed by closing its IO stream
|
|
210
|
+
// (e.g. a Docker exec survives client disconnect), so the only reliable way
|
|
211
|
+
// to stop it — and the token/cost drain of its ongoing API calls — is to
|
|
212
|
+
// destroy the sandbox (stop the container/VM). `keepAlive` /
|
|
213
|
+
// `destroyOnComplete:false` governs *successful completion*, never cancel.
|
|
214
|
+
await definition.destroy(state.ensureCtx)
|
|
215
|
+
await definition.hooks?.onDestroy?.()
|
|
216
|
+
},
|
|
217
|
+
|
|
218
|
+
async onError(ctx, info) {
|
|
219
|
+
const state = runState.get(ctx)
|
|
220
|
+
if (!state) return
|
|
221
|
+
|
|
222
|
+
await state.watcher?.stop()
|
|
223
|
+
await definition.hooks?.onError?.(info.error)
|
|
224
|
+
|
|
225
|
+
// On failure, only tear down when the lifecycle says so; otherwise leave
|
|
226
|
+
// the sandbox for a resumed retry.
|
|
227
|
+
if (definition.lifecycle?.destroyOnComplete) {
|
|
228
|
+
await definition.destroy(state.ensureCtx)
|
|
229
|
+
await definition.hooks?.onDestroy?.()
|
|
230
|
+
}
|
|
231
|
+
},
|
|
232
|
+
})
|
|
233
|
+
}
|
package/src/ngrok.ts
ADDED
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* ngrok-backed tool-bridge provisioner — make the host tool bridge reachable
|
|
3
|
+
* from REMOTE sandboxes (Daytona, Vercel, …) while developing locally.
|
|
4
|
+
*
|
|
5
|
+
* The default bridge ({@link nodeHttpBridgeProvisioner}) binds `localhost`, so
|
|
6
|
+
* only same-machine providers (local-process, Docker) can reach it. A cloud
|
|
7
|
+
* sandbox is a remote VM and can't dial your machine's loopback. This
|
|
8
|
+
* provisioner stands up the normal loopback bridge, opens an ngrok tunnel to its
|
|
9
|
+
* port, and advertises the public `https://…/mcp` URL (with the same per-run
|
|
10
|
+
* bearer token) to the sandbox — so bridged tools / code mode work there too.
|
|
11
|
+
*
|
|
12
|
+
* In PRODUCTION you don't need this: a deployed orchestrator already has a public
|
|
13
|
+
* URL, so a provisioner can advertise that directly (derived from the request).
|
|
14
|
+
* ngrok is the local-dev stand-in for "the orchestrator is reachable".
|
|
15
|
+
*
|
|
16
|
+
* `@ngrok/ngrok` is an OPTIONAL peer dependency — it's loaded lazily, so this
|
|
17
|
+
* subpath imports cleanly without it; only {@link withNgrokBridge} /
|
|
18
|
+
* {@link ngrokBridgeProvisioner} require it at run time. Set `NGROK_AUTHTOKEN`.
|
|
19
|
+
*/
|
|
20
|
+
import { defineChatMiddleware } from '@tanstack/ai'
|
|
21
|
+
import { provideToolBridgeProvisioner } from './capabilities'
|
|
22
|
+
import { startHostToolBridge } from './tool-bridge'
|
|
23
|
+
import type { ToolBridgeProvisioner } from './tool-bridge'
|
|
24
|
+
|
|
25
|
+
/** Whether ngrok tunnelling is configured (an authtoken is present). */
|
|
26
|
+
export function ngrokConfigured(): boolean {
|
|
27
|
+
return Boolean(process.env.NGROK_AUTHTOKEN)
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
/**
|
|
31
|
+
* A {@link ToolBridgeProvisioner} that tunnels the loopback bridge through ngrok
|
|
32
|
+
* (one ephemeral tunnel per run; both are torn down together). Requires the
|
|
33
|
+
* optional `@ngrok/ngrok` peer dependency and `NGROK_AUTHTOKEN`.
|
|
34
|
+
*/
|
|
35
|
+
export const ngrokBridgeProvisioner: ToolBridgeProvisioner = {
|
|
36
|
+
async provision(tools, options) {
|
|
37
|
+
// Lazy + optional: only needed when this provisioner actually runs.
|
|
38
|
+
const { default: ngrok } = await import('@ngrok/ngrok')
|
|
39
|
+
const { provider: _provider, ...core } = options
|
|
40
|
+
const bridge = await startHostToolBridge(tools, {
|
|
41
|
+
hostForSandbox: '127.0.0.1',
|
|
42
|
+
bindAddress: '127.0.0.1',
|
|
43
|
+
...core,
|
|
44
|
+
})
|
|
45
|
+
try {
|
|
46
|
+
const port = Number(new URL(bridge.url).port)
|
|
47
|
+
const listener = await ngrok.forward({
|
|
48
|
+
addr: port,
|
|
49
|
+
authtoken_from_env: true,
|
|
50
|
+
})
|
|
51
|
+
const publicUrl = listener.url()
|
|
52
|
+
if (!publicUrl) {
|
|
53
|
+
throw new Error('ngrok did not return a public URL')
|
|
54
|
+
}
|
|
55
|
+
return {
|
|
56
|
+
...bridge,
|
|
57
|
+
url: `${publicUrl}/mcp`,
|
|
58
|
+
close: async () => {
|
|
59
|
+
try {
|
|
60
|
+
await listener.close()
|
|
61
|
+
} finally {
|
|
62
|
+
await bridge.close()
|
|
63
|
+
}
|
|
64
|
+
},
|
|
65
|
+
}
|
|
66
|
+
} catch (error) {
|
|
67
|
+
// Don't leak the loopback bridge if the tunnel couldn't be opened.
|
|
68
|
+
await bridge.close()
|
|
69
|
+
throw error
|
|
70
|
+
}
|
|
71
|
+
},
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
/**
|
|
75
|
+
* Chat middleware that routes the tool bridge through ngrok. Add it AFTER
|
|
76
|
+
* `withSandbox(...)` for cloud providers so the in-sandbox harness can reach the
|
|
77
|
+
* host tools. Not needed for local-process / Docker (they reach the bridge
|
|
78
|
+
* directly) — just don't add it there.
|
|
79
|
+
*/
|
|
80
|
+
export const withNgrokBridge = defineChatMiddleware({
|
|
81
|
+
name: 'ngrok-bridge',
|
|
82
|
+
setup(ctx) {
|
|
83
|
+
provideToolBridgeProvisioner(ctx, ngrokBridgeProvisioner)
|
|
84
|
+
},
|
|
85
|
+
})
|
package/src/policy.ts
ADDED
|
@@ -0,0 +1,111 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Sandbox policy — a portable, harness-agnostic description of what the agent
|
|
3
|
+
* may do. Each harness adapter MAPS this onto its native permission system
|
|
4
|
+
* (Claude Code → canUseTool + allowedTools/disallowedTools/permissionMode).
|
|
5
|
+
*
|
|
6
|
+
* Command rules are matched as glob/prefix patterns against the command line.
|
|
7
|
+
* Precedence is deny > ask > allow; unmatched commands fall to `default`.
|
|
8
|
+
* `'ask'` surfaces the existing resume-based `approval-requested` flow.
|
|
9
|
+
*/
|
|
10
|
+
|
|
11
|
+
export type PolicyDecision = 'allow' | 'ask' | 'deny'
|
|
12
|
+
|
|
13
|
+
export interface CommandRules {
|
|
14
|
+
/** Glob/prefix patterns to allow outright (e.g. 'pnpm *', 'git diff'). */
|
|
15
|
+
allow?: Array<string>
|
|
16
|
+
/** Patterns that require approval before running. */
|
|
17
|
+
ask?: Array<string>
|
|
18
|
+
/** Patterns to refuse (e.g. 'sudo *', 'rm -rf *'). */
|
|
19
|
+
deny?: Array<string>
|
|
20
|
+
}
|
|
21
|
+
|
|
22
|
+
/** Coarse, non-command capability gates for tools like Write/Edit and network. */
|
|
23
|
+
export interface CapabilityRules {
|
|
24
|
+
/** File-modifying tools (Write/Edit). Defaults to the policy `default`. */
|
|
25
|
+
fileWrite?: PolicyDecision
|
|
26
|
+
/** Outbound network access. Defaults to the policy `default`. */
|
|
27
|
+
network?: PolicyDecision
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
export interface SandboxPolicy {
|
|
31
|
+
commands?: CommandRules
|
|
32
|
+
capabilities?: CapabilityRules
|
|
33
|
+
/** Decision for anything not matched by a rule. Defaults to `'ask'`. */
|
|
34
|
+
default?: PolicyDecision
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
export function defineSandboxPolicy(policy: SandboxPolicy): SandboxPolicy {
|
|
38
|
+
return policy
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
/** Convert a glob/prefix pattern to a RegExp anchored to the full command. */
|
|
42
|
+
function patternToRegExp(pattern: string): RegExp {
|
|
43
|
+
// Escape regex metacharacters except '*', then turn '*' into '.*'.
|
|
44
|
+
const escaped = pattern
|
|
45
|
+
.replace(/[.+?^${}()|[\]\\]/g, '\\$&')
|
|
46
|
+
.replace(/\*/g, '.*')
|
|
47
|
+
return new RegExp(`^${escaped}$`)
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
/**
|
|
51
|
+
* All equivalent forms of a command line when workspace scripts are defined:
|
|
52
|
+
* the literal command, its expanded script value, and any script name that
|
|
53
|
+
* expands to the same value.
|
|
54
|
+
*/
|
|
55
|
+
export function commandAliases(
|
|
56
|
+
command: string,
|
|
57
|
+
scripts: Record<string, string> | undefined,
|
|
58
|
+
): Array<string> {
|
|
59
|
+
const trimmed = command.trim()
|
|
60
|
+
const aliases = new Set<string>([trimmed])
|
|
61
|
+
if (scripts === undefined) return [...aliases]
|
|
62
|
+
|
|
63
|
+
const expanded = scripts[trimmed]
|
|
64
|
+
if (expanded !== undefined) aliases.add(expanded)
|
|
65
|
+
|
|
66
|
+
for (const [name, value] of Object.entries(scripts)) {
|
|
67
|
+
if (value === trimmed) aliases.add(name)
|
|
68
|
+
}
|
|
69
|
+
return [...aliases]
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
function patternMatchesCommand(
|
|
73
|
+
pattern: string,
|
|
74
|
+
command: string,
|
|
75
|
+
scripts: Record<string, string> | undefined,
|
|
76
|
+
): boolean {
|
|
77
|
+
const commandForms = commandAliases(command, scripts)
|
|
78
|
+
for (const patternForm of commandAliases(pattern, scripts)) {
|
|
79
|
+
const re = patternToRegExp(patternForm)
|
|
80
|
+
if (commandForms.some((form) => re.test(form))) return true
|
|
81
|
+
}
|
|
82
|
+
return false
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
/**
|
|
86
|
+
* Resolve a command line against the policy. Precedence: deny > ask > allow,
|
|
87
|
+
* then `default` (defaults to `'ask'`). Exported for adapter permission
|
|
88
|
+
* mappers and unit tests.
|
|
89
|
+
*
|
|
90
|
+
* When `scripts` is provided, policy patterns may match either a script name
|
|
91
|
+
* or its expanded command value (and vice versa for the command under test).
|
|
92
|
+
*/
|
|
93
|
+
export function evaluateCommand(
|
|
94
|
+
command: string,
|
|
95
|
+
policy: SandboxPolicy | undefined,
|
|
96
|
+
scripts?: Record<string, string>,
|
|
97
|
+
): PolicyDecision {
|
|
98
|
+
const fallback = policy?.default ?? 'ask'
|
|
99
|
+
const rules = policy?.commands
|
|
100
|
+
if (!rules) return fallback
|
|
101
|
+
|
|
102
|
+
const matches = (patterns: Array<string> | undefined): boolean =>
|
|
103
|
+
(patterns ?? []).some((pattern) =>
|
|
104
|
+
patternMatchesCommand(pattern, command, scripts),
|
|
105
|
+
)
|
|
106
|
+
|
|
107
|
+
if (matches(rules.deny)) return 'deny'
|
|
108
|
+
if (matches(rules.ask)) return 'ask'
|
|
109
|
+
if (matches(rules.allow)) return 'allow'
|
|
110
|
+
return fallback
|
|
111
|
+
}
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Workspace projection capability — provided by `withSandbox` and consumed by
|
|
3
|
+
* harness adapters (claude-code, codex, opencode) to idempotently
|
|
4
|
+
* project skills, plugins, and resolved secrets into the native harness format.
|
|
5
|
+
*
|
|
6
|
+
* The capability carries the raw provisioning inputs (skills, plugins, a
|
|
7
|
+
* resolve function for secret refs) together with a marker path that lets
|
|
8
|
+
* adapters guard the projection with a one-time idempotency file.
|
|
9
|
+
*/
|
|
10
|
+
import { createCapability } from '@tanstack/ai'
|
|
11
|
+
import type { SecretRef } from './secrets'
|
|
12
|
+
import type { WorkspaceSkill } from './workspace'
|
|
13
|
+
|
|
14
|
+
/**
|
|
15
|
+
* The shape provided to harness adapters via the sandbox projection capability.
|
|
16
|
+
* Harness adapters read this in their `chatStream` setup to project workspace
|
|
17
|
+
* inputs into their native format (MCP config, skills dirs, plugin installs).
|
|
18
|
+
*/
|
|
19
|
+
export interface WorkspaceProjection {
|
|
20
|
+
/** Skills declared on the workspace — MCP servers, file skills, git repos, etc. */
|
|
21
|
+
skills: Array<WorkspaceSkill>
|
|
22
|
+
/** Harness plugin identifiers to install idempotently. */
|
|
23
|
+
plugins: Array<string>
|
|
24
|
+
/**
|
|
25
|
+
* Resolve a SecretRef to its plaintext value. Bound to the workspace's
|
|
26
|
+
* secrets registry; throws when the ref is unknown.
|
|
27
|
+
*/
|
|
28
|
+
resolveSecret: (ref: SecretRef) => string
|
|
29
|
+
/**
|
|
30
|
+
* Absolute path to the idempotency marker file. Harness adapters write this
|
|
31
|
+
* file after a successful projection so subsequent runs skip re-projection.
|
|
32
|
+
* The file is NOT included in snapshots — absent on restore, triggering
|
|
33
|
+
* re-projection (which re-writes any secret-bearing config files).
|
|
34
|
+
*/
|
|
35
|
+
markerPath: string
|
|
36
|
+
/** Workspace root inside the sandbox (e.g. `/workspace`). */
|
|
37
|
+
root: string
|
|
38
|
+
/** Named commands declared on the workspace (e.g. `{ test: 'pnpm test' }`). */
|
|
39
|
+
scripts?: Record<string, string>
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
export const ProjectionCapability =
|
|
43
|
+
createCapability<WorkspaceProjection>()('sandbox-projection')
|
|
44
|
+
|
|
45
|
+
export const [getWorkspaceProjection, provideWorkspaceProjection] =
|
|
46
|
+
ProjectionCapability
|