castle-web-cli 0.4.183 → 0.4.185

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (36) hide show
  1. package/dist/agent-failures.d.ts +1 -0
  2. package/dist/agent-failures.js +31 -4
  3. package/dist/agent-prompts.d.ts +0 -3
  4. package/dist/agent-prompts.js +14 -28
  5. package/dist/agent.d.ts +1 -13
  6. package/dist/agent.js +191 -938
  7. package/dist/byo-accounts.d.ts +4 -4
  8. package/dist/byo-accounts.js +17 -38
  9. package/dist/byo-auth.d.ts +2 -5
  10. package/dist/byo-auth.js +4 -52
  11. package/dist/byo-login.d.ts +4 -5
  12. package/dist/byo-login.js +28 -56
  13. package/dist/mcpPlaytest.js +1 -2
  14. package/dist/metering.d.ts +35 -23
  15. package/dist/metering.js +57 -52
  16. package/dist/openrouter-catalog.d.ts +0 -1
  17. package/dist/openrouter-catalog.js +0 -11
  18. package/dist/platformDoc.d.ts +1 -1
  19. package/dist/platformDoc.js +9 -12
  20. package/dist/serve.js +3 -5
  21. package/dist/shell/assets/index-BHWfx7Tz.css +1 -0
  22. package/dist/shell/assets/index-BMu42DTS.js +449 -0
  23. package/dist/shell/index.html +2 -2
  24. package/kits/physics-2d/CLAUDE.md +5 -1
  25. package/kits/physics-2d/castle.json +1 -1
  26. package/kits/physics-2d/editors/SceneEditor.jsx +2 -1
  27. package/kits/physics-2d/engine/autoInspector.jsx +32 -3
  28. package/kits/physics-2d/engine/fields/fields.jsx +305 -5
  29. package/kits/physics-2d/engine/fields/fields.module.css +76 -0
  30. package/kits/physics-2d/engine/paletteField.jsx +8 -3
  31. package/kits/physics-2d/engine/spriteField.jsx +3 -3
  32. package/kits/physics-2d/engine/ui.jsx +5 -277
  33. package/kits/physics-2d/engine/ui.module.css +0 -69
  34. package/package.json +1 -1
  35. package/dist/shell/assets/index-DnNy-Z5u.js +0 -447
  36. package/dist/shell/assets/index-cdsH2lna.css +0 -1
package/dist/agent.js CHANGED
@@ -11,9 +11,8 @@
11
11
  // shows up as small log messages (started / ready for review / completed)
12
12
  // instead of extra router turns.
13
13
  //
14
- // Backend CLI: cursor-agent in headless print mode (stream-json). The router
15
- // runs with --mode ask (read-only at the CLI level); task agents run with
16
- // --force. Claude support can slot in later behind runAgentCli.
14
+ // Backend CLI: the claude CLI in headless print mode (stream-json), behind
15
+ // runAgentCli.
17
16
  import { execFileSync, spawn } from 'child_process';
18
17
  import { quickReferenceFor } from './platformDoc.js';
19
18
  import { IMPORTS_DIR } from './imports.js';
@@ -30,25 +29,18 @@ import { atomicWriteFileSync } from './atomicFile.js';
30
29
  import { AGENT_ATTACHMENT_PREFIX, AGENT_PLAYTEST_PREFIX, PLAN_FILE } from './localPaths.js';
31
30
  import { applyPlanOps, buildRouterPromptParts, buildTaskPrompt, parsePlanOps, planOpenQuestionLines, truncateToBytes, userTurnInstruction, CLAUDE_TASK_SYSTEM_REMINDER, } from './agent-prompts.js';
32
31
  import { readCastleJson } from './castleJson.js';
33
- import { checkOpenrouterKey, checkOpenrouterModel, openrouterCatalogEntry, primeOpenrouterCatalog, } from './openrouter-catalog.js';
32
+ import { checkOpenrouterKey, checkOpenrouterModel, primeOpenrouterCatalog, } from './openrouter-catalog.js';
34
33
  import { classifyProviderError, failureCopy, setReaderTimeZone, } from './agent-failures.js';
35
- import { castleCreditsExhausted, createRefreshQueue, fetchAiCredits, fetchBudget, meteringHeaders, newAgentSessionId, reportCursorRun, spendableMicros, withCustomHeaders, } from './metering.js';
36
- import { anthropicKeyHelperCommand, claudeHasSavedLogin, cursorAuthPath, cursorHasUserLogin, resolveAnthropicAuth, userKey, ANTHROPIC_CREDENTIAL_ENV, ANTHROPIC_PROXY_ENV, CASTLE_USER_KEYS_PATH, } from './byo-auth.js';
34
+ import { castleCreditsExhausted, createRefreshQueue, fetchAiCredits, fetchBudget, freeRolesFor, freeTierForRoles, meteringHeaders, modelIsFree, newAgentSessionId, spendableMicros, withCustomHeaders, } from './metering.js';
35
+ import { anthropicKeyHelperCommand, resolveAnthropicAuth, userKey, ANTHROPIC_CREDENTIAL_ENV, ANTHROPIC_PROXY_ENV, CASTLE_USER_KEYS_PATH, } from './byo-auth.js';
37
36
  import { accountsSnapshot, loginProviderFor, watchCredentials, writeCredential, } from './byo-accounts.js';
38
37
  import { cancelLogin, logout, startLogin, submitLoginCode } from './byo-login.js';
39
- import { runAgentNative } from './native/loop.js';
40
38
  import { createPlaytestBrowserManager, } from './native/playtest-browser.js';
41
39
  import { createPlaywrightPlaytestExecutor } from './native/playtest-executor.js';
42
40
  import { createRemotePlaytestExecutor } from './native/playtest-remote.js';
43
41
  import { runPlaytest, } from './native/playtest.js';
44
42
  export const AGENT_WS_PATH = '/__castle/agent';
45
- // Same-origin proxy for OpenRouter model capabilities (avoids browser CORS
46
- // against openrouter.ai). GET ?model=<slug> -> ModelCaps JSON. Powers the
47
- // settings popover's dynamic reasoning-effort / provider-tier pickers.
48
- export const AGENT_MODEL_CAPS_PREFIX = '/__castle/agent/model-caps';
49
43
  const DEFAULT_SETTINGS = {
50
- router: 'claude',
51
- tasks: 'claude',
52
44
  // Both roles run the claude CLI routed through OpenRouter at the slug below.
53
45
  // Note this takes the Anthropic credential out of play entirely: a role on
54
46
  // "openrouter" resolves OpenRouter auth, so a user's `claude /login` or
@@ -56,27 +48,13 @@ const DEFAULT_SETTINGS = {
56
48
  // exactly why a user who HAS one defaults to ANTHROPIC_DEFAULT_MODELS below.
57
49
  routerClaudeModel: 'openrouter',
58
50
  tasksClaudeModel: 'openrouter',
59
- // Free-form -- change to any OpenRouter slug.
60
- routerOpenrouterModel: 'openai/gpt-5.6-terra',
61
- tasksOpenrouterModel: 'openai/gpt-5.6-terra',
62
- // Both roles think at "medium": the operator stays snappy (the user waits
63
- // on every operator turn), and task agents' multi-turn tool loops don't pay
64
- // reasoning tax on mechanical read/edit/run turns. Deep decomposition
65
- // quality comes from the operator prompt, not a higher effort default.
66
- routerReasoningEffort: 'medium',
67
- tasksReasoningEffort: 'medium',
68
- // Routing splits by what each role optimizes for: the interactive operator
69
- // routes for speed (nitro = throughput-sorted endpoints), unattended task
70
- // agents route for correctness (exacto = benchmark-accurate endpoints,
71
- // which matters for tool-calling fidelity over long loops).
72
- routerRouting: 'nitro',
73
- tasksRouting: 'exacto',
74
- // Operator pins OpenAI's priority (low-latency SLA) tier; harmless with a
75
- // slug that lacks it since the pin falls back when the tag doesn't exist
76
- // (allow_fallbacks). Tasks stay on auto: high-volume background turns
77
- // should ride the cheapest available capacity.
78
- routerProviderTier: 'openai/priority',
79
- tasksProviderTier: '',
51
+ // Free-form -- change to any OpenRouter slug. Both roles default to the
52
+ // model Castle pays for (the proxy's freeModelPrefixes list), so an editor
53
+ // that has never touched settings costs the user nothing and never runs into
54
+ // a credit balance -- see freeTierForRoles / the free-model skip in
55
+ // castleSpendRefusal. Picking any other slug puts that role back on credits.
56
+ routerOpenrouterModel: 'meta/muse-spark-1.3-contributor',
57
+ tasksOpenrouterModel: 'meta/muse-spark-1.3-contributor',
80
58
  };
81
59
  // What the two claude-model defaults become once the user has an Anthropic
82
60
  // credential of their own -- a key or a claude.ai login, which
@@ -97,9 +75,6 @@ function defaultSettings() {
97
75
  ? { ...DEFAULT_SETTINGS }
98
76
  : { ...DEFAULT_SETTINGS, ...ANTHROPIC_DEFAULT_MODELS };
99
77
  }
100
- function normalizeBackend(value) {
101
- return value === 'cursor' || value === 'claude' || value === 'smith' ? value : null;
102
- }
103
78
  // Bump to force every deck onto the current DEFAULT_SETTINGS once, discarding
104
79
  // what users had chosen: a stored file below this epoch has ALL of its setting
105
80
  // fields dropped on load (see loadAgentSettings). Bump ONLY when a default
@@ -107,7 +82,13 @@ function normalizeBackend(value) {
107
82
  // bump, since sparse storage already reaches anyone who never set that field.
108
83
  // The wipe is also what retires older file shapes, so a migration for one is
109
84
  // only worth writing if it must survive the epoch that introduces it.
110
- const SETTINGS_EPOCH = 1;
85
+ //
86
+ // 2: the OpenRouter default moved to the model Castle pays for. Deliberately a
87
+ // bump rather than an ordinary default change -- decks that had already stored
88
+ // a slug (including the one that was the default when they stored it) would
89
+ // otherwise keep spending credits on a metered model while every user around
90
+ // them runs free.
91
+ const SETTINGS_EPOCH = 2;
111
92
  function normalizeClaudeModel(value) {
112
93
  return value === 'sonnet' || value === 'opus' || value === 'fable' || value === 'openrouter'
113
94
  ? value
@@ -139,62 +120,19 @@ function normalizeOpenrouterModel(value) {
139
120
  const trimmed = value.trim();
140
121
  return trimmed && trimmed.length <= OPENROUTER_MODEL_MAX_LEN ? trimmed : null;
141
122
  }
142
- // OpenRouter's full effort superset -- validated against the union rather than
143
- // a per-model list because supported efforts are model-specific (the client
144
- // fetches them from the model-caps endpoint to build the picker) and
145
- // OpenRouter maps an unsupported level to the nearest one anyway.
146
- const REASONING_EFFORTS = [
147
- 'none',
148
- 'minimal',
149
- 'low',
150
- 'medium',
151
- 'high',
152
- 'xhigh',
153
- 'max',
154
- ];
155
- function normalizeReasoningEffort(value) {
156
- return REASONING_EFFORTS.includes(value)
157
- ? value
158
- : null;
159
- }
160
- const ROUTING_MODES = ['balanced', 'nitro', 'exacto', 'floor'];
161
- function normalizeRoutingMode(value) {
162
- return ROUTING_MODES.includes(value) ? value : null;
163
- }
164
- // Provider tier is an OpenRouter endpoint `tag` ("openai/flex", "azure/eu",
165
- // ...) which is model-specific, so validation is loose like the model slug.
166
- // Unlike the slug, empty string is VALID and meaningful: "auto" (no pin), so
167
- // this returns "" rather than null for the clear case -- callers must treat
168
- // null (invalid) and "" (clear) differently.
169
- function normalizeProviderTier(value) {
170
- if (typeof value !== 'string')
171
- return null;
172
- const trimmed = value.trim();
173
- if (trimmed.length > OPENROUTER_MODEL_MAX_LEN)
174
- return null;
175
- return trimmed;
176
- }
177
123
  // Every setting's validator, keyed the same as AgentSettings, so load and save
178
124
  // iterate one list rather than repeating the field names per direction.
179
125
  const SETTING_NORMALIZERS = {
180
- router: normalizeBackend,
181
- tasks: normalizeBackend,
182
126
  routerClaudeModel: normalizeClaudeModel,
183
127
  tasksClaudeModel: normalizeClaudeModel,
184
128
  routerOpenrouterModel: normalizeOpenrouterModel,
185
129
  tasksOpenrouterModel: normalizeOpenrouterModel,
186
- routerReasoningEffort: normalizeReasoningEffort,
187
- tasksReasoningEffort: normalizeReasoningEffort,
188
- routerRouting: normalizeRoutingMode,
189
- tasksRouting: normalizeRoutingMode,
190
- routerProviderTier: normalizeProviderTier,
191
- tasksProviderTier: normalizeProviderTier,
192
130
  };
193
131
  const SETTING_KEYS = Object.keys(SETTING_NORMALIZERS);
194
132
  // Sparse ON PURPOSE: only fields that DIFFER from the current defaults are
195
133
  // stored, so "absent" means "follow the default" and a later default change
196
134
  // reaches every deck that never set that field. Writing the full object
197
- // instead (as this once did) froze all 12 settings the first time a user
135
+ // instead (as this once did) froze every setting the first time a user
198
136
  // touched any one of them, pinning models nobody had chosen.
199
137
  //
200
138
  // The flip side, accepted: re-selecting the value that is currently the
@@ -294,9 +232,6 @@ function openrouterAnthropicBase() {
294
232
  // (the CLI appends /v1/messages), bypassing the proxy. The proxy branch instead
295
233
  // uses openrouterAnthropicBase() (the injected OPENROUTER_BASE_URL origin).
296
234
  const OPENROUTER_DIRECT_ANTHROPIC_BASE = 'https://openrouter.ai/api';
297
- // OpenAI-shaped chat-completions ORIGIN for the smith native loop on a user's
298
- // own OpenRouter key (native/openrouter.ts appends /chat/completions).
299
- const OPENROUTER_DIRECT_CHAT_BASE = 'https://openrouter.ai/api/v1';
300
235
  function envForOpenrouterSpawn(auth) {
301
236
  if (!auth.key) {
302
237
  // Unreachable via runAgentTurn (pre-flight rejects a keyless run before
@@ -320,121 +255,16 @@ function envForOpenrouterSpawn(auth) {
320
255
  env.ANTHROPIC_AUTH_TOKEN = auth.key;
321
256
  return env;
322
257
  }
323
- // The one OpenRouter credential, shared by BOTH OpenRouter paths: smith's
324
- // native loop passes it as runAgentNative's apiKey (Bearer header), and the
325
- // claude-via-OpenRouter spawn sends it as ANTHROPIC_AUTH_TOKEN (see
326
- // envForOpenrouterSpawn). Castle-provided key file first, env fallback --
327
- // same sourcing order as envForAgentSpawn.
258
+ // The OpenRouter credential the claude-via-OpenRouter spawn sends as
259
+ // ANTHROPIC_AUTH_TOKEN (see envForOpenrouterSpawn). Castle-provided key file
260
+ // first, env fallback.
328
261
  const OPENROUTER_KEY_NAME = 'OPENROUTER_API_KEY';
329
262
  function openrouterApiKey() {
330
263
  return castleKeys()[OPENROUTER_KEY_NAME] ?? process.env[OPENROUTER_KEY_NAME] ?? '';
331
264
  }
332
- const MODEL_CAPS_TTL_MS = 10 * 60_000;
333
- const OPENROUTER_API_BASE = 'https://openrouter.ai/api/v1';
334
- const modelCapsCache = new Map();
335
- function asRecord(v) {
336
- return v && typeof v === 'object' ? v : null;
337
- }
338
- async function openrouterProviderTiers(slug) {
339
- const res = await fetch(`${OPENROUTER_API_BASE}/models/${slug}/endpoints`);
340
- if (!res.ok)
341
- return [];
342
- const json = asRecord(await res.json());
343
- const data = asRecord(json?.data);
344
- const endpoints = data && Array.isArray(data.endpoints) ? data.endpoints : [];
345
- const tags = [];
346
- for (const ep of endpoints) {
347
- const rec = asRecord(ep);
348
- const tag = rec && typeof rec.tag === 'string' ? rec.tag : null;
349
- if (tag && !tags.includes(tag))
350
- tags.push(tag);
351
- }
352
- return tags;
353
- }
354
- async function fetchModelCaps(slug) {
355
- const cached = modelCapsCache.get(slug);
356
- if (cached && Date.now() - cached.at < MODEL_CAPS_TTL_MS)
357
- return cached.caps;
358
- // Best-effort per source: a failure in either leaves that half empty rather
359
- // than failing the whole lookup, so a bad slug still yields a usable (empty)
360
- // caps object the client can render as "no dynamic options".
361
- let reasoningEfforts = null;
362
- let defaultEffort = null;
363
- try {
364
- // Shared catalog (openrouter-catalog.ts) rather than a second /models
365
- // fetch: it's the same list the pre-flight slug check reads, and it brings
366
- // a disk-backed stale-while-revalidate cache with it.
367
- const entry = await openrouterCatalogEntry(slug);
368
- const reasoning = asRecord(entry?.reasoning);
369
- const efforts = reasoning?.supported_efforts;
370
- const supportedParams = entry?.supportedParameters;
371
- const acceptsEffort = Array.isArray(supportedParams) &&
372
- (supportedParams.includes('reasoning_effort') || supportedParams.includes('reasoning'));
373
- if (acceptsEffort && Array.isArray(efforts) && efforts.length > 0) {
374
- reasoningEfforts = efforts.filter((e) => typeof e === 'string');
375
- defaultEffort =
376
- typeof reasoning?.default_effort === 'string' ? reasoning.default_effort : null;
377
- }
378
- }
379
- catch {
380
- // leave reasoning fields null
381
- }
382
- let providerTiers = [];
383
- try {
384
- providerTiers = await openrouterProviderTiers(slug);
385
- }
386
- catch {
387
- // leave providerTiers empty
388
- }
389
- const caps = {
390
- model: slug,
391
- reasoningEfforts,
392
- defaultEffort,
393
- providerTiers,
394
- };
395
- modelCapsCache.set(slug, { caps, at: Date.now() });
396
- return caps;
397
- }
398
- // GET AGENT_MODEL_CAPS_PREFIX?model=<slug>. Returns 400 for a missing/oversized
399
- // slug, 200 ModelCaps otherwise (empty caps on upstream failure -- see
400
- // fetchModelCaps). reqPath is already query-stripped; parse req.url for it.
401
- function handleModelCaps(req, res) {
402
- const send = (status, body) => {
403
- res.writeHead(status, {
404
- 'content-type': 'application/json',
405
- 'cache-control': 'no-store',
406
- });
407
- res.end(JSON.stringify(body));
408
- return true;
409
- };
410
- let slug = '';
411
- try {
412
- slug = (new URL(req.url ?? '', 'http://localhost').searchParams.get('model') ?? '').trim();
413
- }
414
- catch {
415
- slug = '';
416
- }
417
- if (!slug || slug.length > OPENROUTER_MODEL_MAX_LEN) {
418
- return send(400, { error: 'missing or invalid model' });
419
- }
420
- // Strip any routing suffix the client may have on the displayed slug so the
421
- // OpenRouter lookup hits the base model id.
422
- const baseSlug = slug.replace(/:(nitro|exacto|floor)$/, '');
423
- fetchModelCaps(baseSlug)
424
- .then((caps) => send(200, caps))
425
- .catch(() => send(200, {
426
- model: baseSlug,
427
- reasoningEfforts: null,
428
- defaultEffort: null,
429
- providerTiers: [],
430
- }));
431
- return true;
432
- }
433
- // Build the headless CLI invocation for a spawning backend/role (smith never
434
- // comes through here -- it has no CLI process; see runAgentSmith). Cursor's
435
- // router runs in read-only ask mode; hosted claude runs bypassPermissions for
436
- // both roles because the container is already the permission boundary. Local
437
- // claude stays on auto (NOT plan mode -- that makes it emit plan tool calls)
265
+ // Build the headless CLI invocation for a role. Hosted claude runs
266
+ // bypassPermissions for both roles because the container is already the
267
+ // permission boundary. Local claude stays on auto (NOT plan mode -- that makes it emit plan tool calls)
438
268
  // at medium effort. claudeModel "openrouter" keeps the SAME claude binary and
439
269
  // swaps only the model slug and env (see envForOpenrouterSpawn).
440
270
  // claude CLI 2.1.198 SILENTLY resolves `--permission-mode auto` to "default"
@@ -453,8 +283,8 @@ function handleModelCaps(req, res) {
453
283
  const OPENROUTER_ALLOWED_TOOLS = ['Edit', 'Write', 'NotebookEdit', 'Bash'];
454
284
  // The tool id claude assigns our MCP server's `playtest` tool: mcp__<server>__<tool>,
455
285
  // where <server> is the key under mcpServers in --mcp-config (see mcpConfigArg).
456
- // Task-only, matching smith (native/tools.ts) -- a router turn has no deck to
457
- // play and no frames dir to write into.
286
+ // Task-only -- a router turn has no deck to play and no frames dir to write
287
+ // into.
458
288
  const CLAUDE_PLAYTEST_TOOL = 'mcp__castle__playtest';
459
289
  // Single `=` token because --allowedTools is variadic and would otherwise
460
290
  // swallow the trailing prompt positional.
@@ -489,9 +319,6 @@ function mcpConfigArg(deckDir, taskId) {
489
319
  });
490
320
  return `--mcp-config=${config}`;
491
321
  }
492
- // Cursor's proprietary model. Also the slug reported to the metering ledger, so
493
- // the two can never drift into disagreeing about what a cursor row ran on.
494
- const CURSOR_MODEL = 'composer-2.5-fast';
495
322
  // Keep runs independent of the machine's user config: no user plugins (LSP
496
323
  // servers etc.), no user MCP servers. CLAUDE.md auto-discovery and OAuth still
497
324
  // work. On a user's own Anthropic KEY this also carries the apiKeyHelper that
@@ -517,96 +344,73 @@ const ROUTER_CLAUDE_ENV = {
517
344
  CLAUDE_CODE_DISABLE_GIT_INSTRUCTIONS: '1',
518
345
  CLAUDE_CODE_DISABLE_AUTO_MEMORY: '1',
519
346
  };
520
- function buildAgentInvocation(backend, role, prompt, claudeModel,
347
+ function buildAgentInvocation(role, prompt, claudeModel,
521
348
  // Already resolved for this role by the caller (router turns pass
522
349
  // settings.routerOpenrouterModel, task spawns settings.tasksOpenrouterModel).
523
350
  openrouterModel,
524
351
  // This run's ledger identity, and the deck dir its castle.json is read from.
525
- // Only the claude branch can carry it: cursor-agent runs on its own key and
526
- // never traverses the llm-proxy.
527
352
  metering,
528
353
  // The task this spawn belongs to, so its playtest frames land under
529
354
  // tasks/<id>/playtest/ and count against that task's call budget. Absent for
530
355
  // router turns, which get no playtest tool at all.
531
356
  taskId,
532
- // The stable part of a router prompt. claude takes it as appended system
533
- // text, which its own cache breakpoints cover; cursor has no such flag and
534
- // gets it ahead of the prompt.
357
+ // The stable part of a router prompt, passed as appended system text so the
358
+ // CLI's own cache breakpoints cover it.
535
359
  systemPrompt) {
536
360
  const systemText = [systemPrompt, role === 'task' ? CLAUDE_TASK_SYSTEM_REMINDER : undefined]
537
361
  .filter((text) => Boolean(text))
538
362
  .join('\n\n');
539
- if (backend === 'claude') {
540
- const hosted = Boolean(process.env.CASTLE_SANDBOX_ID);
541
- const viaOpenrouter = claudeModel === 'openrouter';
542
- const orAuth = viaOpenrouter ? resolveOpenrouterAuth() : null;
543
- const anAuth = viaOpenrouter ? null : resolveAnthropicAuth();
544
- // Direct = the user's own credential/login is in play, so this run bypasses
545
- // the proxy and must NOT carry metering headers (metering.ts would otherwise
546
- // attach them off process.env and the direct spawn would leak them upstream).
547
- const direct = viaOpenrouter ? orAuth.mode === 'user-key' : anAuth.mode !== 'proxy';
548
- return {
549
- command: 'claude',
550
- args: [
551
- '-p',
552
- '--verbose',
553
- '--output-format',
554
- 'stream-json',
555
- '--include-partial-messages',
556
- '--permission-mode',
557
- hosted ? 'bypassPermissions' : 'auto',
558
- ...(!hosted && viaOpenrouter ? [openrouterAllowedToolsArg(role)] : []),
559
- '--model',
560
- viaOpenrouter ? openrouterModel : claudeModel,
561
- '--effort',
562
- 'medium',
563
- // Newer claude models default thinking display to "omitted" (empty
564
- // thinking_delta text, signature only); "summarized" restores actual
565
- // summary text so the shell's expandable thinking transcript has
566
- // content. Slight time-to-first-text cost (the API streams the
567
- // summary before prose). Undocumented in --help but honored.
568
- '--thinking-display',
569
- 'summarized',
570
- '--settings',
571
- claudeSettingsArg(anAuth),
572
- '--strict-mcp-config',
573
- ...(role === 'task' ? [mcpConfigArg(metering.deckDir, taskId)] : []),
574
- ...(systemText ? ['--append-system-prompt', systemText] : []),
575
- prompt,
576
- ],
577
- env: {
578
- ...withCustomHeaders(viaOpenrouter ? envForOpenrouterSpawn(orAuth) : envForClaudeSpawn(anAuth), meteringHeaders({
579
- deckDir: metering.deckDir,
580
- sessionId: metering.sessionId,
581
- route: viaOpenrouter ? 'openrouter' : 'anthropic',
582
- direct,
583
- })),
584
- ...(role === 'router' ? ROUTER_CLAUDE_ENV : {}),
585
- // Claude Code 2.1.258 refuses bypassPermissions as root unless the
586
- // child explicitly identifies itself as running in a sandbox.
587
- ...(hosted ? { IS_SANDBOX: '1' } : {}),
588
- },
589
- };
590
- }
363
+ const hosted = Boolean(process.env.CASTLE_SANDBOX_ID);
364
+ const viaOpenrouter = claudeModel === 'openrouter';
365
+ const orAuth = viaOpenrouter ? resolveOpenrouterAuth() : null;
366
+ const anAuth = viaOpenrouter ? null : resolveAnthropicAuth();
367
+ // Direct = the user's own credential/login is in play, so this run bypasses
368
+ // the proxy and must NOT carry metering headers (metering.ts would otherwise
369
+ // attach them off process.env and the direct spawn would leak them upstream).
370
+ const direct = viaOpenrouter ? orAuth.mode === 'user-key' : anAuth.mode !== 'proxy';
591
371
  return {
592
- command: 'cursor-agent',
372
+ command: 'claude',
593
373
  args: [
594
374
  '-p',
375
+ '--verbose',
595
376
  '--output-format',
596
377
  'stream-json',
597
- '--stream-partial-output',
598
- '--trust',
378
+ '--include-partial-messages',
379
+ '--permission-mode',
380
+ hosted ? 'bypassPermissions' : 'auto',
381
+ ...(!hosted && viaOpenrouter ? [openrouterAllowedToolsArg(role)] : []),
599
382
  '--model',
600
- CURSOR_MODEL,
601
- ...(role === 'router' ? ['--mode', 'ask'] : ['--force']),
602
- systemPrompt ? `${systemPrompt}\n\n${prompt}` : prompt,
383
+ viaOpenrouter ? openrouterModel : claudeModel,
384
+ '--effort',
385
+ 'medium',
386
+ // Newer claude models default thinking display to "omitted" (empty
387
+ // thinking_delta text, signature only); "summarized" restores actual
388
+ // summary text so the shell's expandable thinking transcript has
389
+ // content. Slight time-to-first-text cost (the API streams the
390
+ // summary before prose). Undocumented in --help but honored.
391
+ '--thinking-display',
392
+ 'summarized',
393
+ '--settings',
394
+ claudeSettingsArg(anAuth),
395
+ '--strict-mcp-config',
396
+ ...(role === 'task' ? [mcpConfigArg(metering.deckDir, taskId)] : []),
397
+ ...(systemText ? ['--append-system-prompt', systemText] : []),
398
+ prompt,
603
399
  ],
604
- env: envForAgentSpawn(backend),
400
+ env: {
401
+ ...withCustomHeaders(viaOpenrouter ? envForOpenrouterSpawn(orAuth) : envForClaudeSpawn(anAuth), meteringHeaders({
402
+ deckDir: metering.deckDir,
403
+ sessionId: metering.sessionId,
404
+ route: viaOpenrouter ? 'openrouter' : 'anthropic',
405
+ direct,
406
+ })),
407
+ ...(role === 'router' ? ROUTER_CLAUDE_ENV : {}),
408
+ // Claude Code 2.1.258 refuses bypassPermissions as root unless the
409
+ // child explicitly identifies itself as running in a sandbox.
410
+ ...(hosted ? { IS_SANDBOX: '1' } : {}),
411
+ },
605
412
  };
606
413
  }
607
- function parserForBackend(backend) {
608
- return backend === 'cursor' ? 'cursor' : 'claude';
609
- }
610
414
  const ROUTER_TIMEOUT_MS = 3 * 60_000;
611
415
  const TASK_TIMEOUT_MS = 30 * 60_000;
612
416
  const MAX_TASK_ATTEMPTS = 3;
@@ -616,10 +420,10 @@ const MAX_TASK_ATTEMPTS = 3;
616
420
  const MAX_CONCURRENT_TASKS = Number(process.env.CASTLE_MAX_CONCURRENT_TASKS) || 4;
617
421
  // Base backoff (ms) before a crashed task agent is relaunched, scaled by
618
422
  // attempt number and jittered (see waitBeforeTaskRetry). Incident (2026-07):
619
- // the router spawned 4 parallel tasks whose cursor-agent processes all
423
+ // the router spawned 4 parallel tasks whose agent CLI processes all
620
424
  // crashed at startup within ~1.3s with a macOS keychain error ("Security
621
425
  // command failed: Security process exited with code: 44" -- credential
622
- // lookup contention when many cursor-agent processes start at once). The old
426
+ // lookup contention when many agent CLI processes start at once). The old
623
427
  // instant retry burned all 3 attempts of every task inside that same ~6s
624
428
  // contention window (11 agent processes launched in ~7s); the one retry that
625
429
  // happened to land ~2s later succeeded, so growing + jittered spacing
@@ -774,11 +578,8 @@ function planDefault() {
774
578
  // The playtest steering: one router rule plus the task prompt's carve-out
775
579
  // (PLAYTEST_ROUTER_RULE in agent-prompts.ts). On unless the serve was told
776
580
  // `CASTLE_AGENT_PLAYTEST=off` (the only env value that does anything; no
777
- // per-deck switch), and never for cursor tasks, which are the one backend
778
- // spawned without a playtest tool.
779
- function playtestSteering(tasksBackend) {
780
- if (tasksBackend === 'cursor')
781
- return false;
581
+ // per-deck switch).
582
+ function playtestSteering() {
782
583
  return process.env.CASTLE_AGENT_PLAYTEST !== 'off';
783
584
  }
784
585
  function planSession(deckDir) {
@@ -1019,22 +820,6 @@ function humanizeAskBlocks(text) {
1019
820
  }
1020
821
  });
1021
822
  }
1022
- // Monotonic source of smith pseudo-pids (see AgentRunHandle.pid).
1023
- let smithRunSeq = 0;
1024
- function makeSmithRunHandle(controller) {
1025
- smithRunSeq += 1;
1026
- return {
1027
- pid: -smithRunSeq,
1028
- // Reads as "still running" for symmetry, but nothing consumes it for
1029
- // smith runs: the registry already excludes them by pid sign.
1030
- exitCode: null,
1031
- spawnfile: 'castle-smith',
1032
- kill: () => {
1033
- controller.abort();
1034
- return true;
1035
- },
1036
- };
1037
- }
1038
823
  function baseName(p) {
1039
824
  const parts = p.split(/[\\/]/).filter(Boolean);
1040
825
  return parts[parts.length - 1] || p;
@@ -1082,7 +867,6 @@ function claudeToolFeedLabel(name, input) {
1082
867
  // content_block_start, before the input JSON streams in). Router-only fallback
1083
868
  // so the streaming reply's activity line reflects EVERY tool -- including the
1084
869
  // commands/searches claudeToolFeedLabel intentionally drops from task feeds.
1085
- // Mirrors the cursor toolActivityLabel verbs so the two backends read alike.
1086
870
  function genericClaudeToolLabel(name) {
1087
871
  const kind = name.toLowerCase();
1088
872
  if (kind === CLAUDE_PLAYTEST_TOOL)
@@ -1099,21 +883,6 @@ function genericClaudeToolLabel(name) {
1099
883
  return 'Searching the web';
1100
884
  return 'Working';
1101
885
  }
1102
- // Human-readable label for a tool_call event, e.g. readToolCall -> "reading
1103
- // the deck". Shown as the streaming message's activity line.
1104
- function toolActivityLabel(ev) {
1105
- const call = ev.tool_call;
1106
- const key = call ? Object.keys(call).find((k) => k.endsWith('ToolCall')) : undefined;
1107
- const kind = (key ?? '').slice(0, -'ToolCall'.length).toLowerCase();
1108
- if (['read', 'glob', 'grep', 'ls', 'list'].some((p) => kind.startsWith(p))) {
1109
- return 'Reading the deck';
1110
- }
1111
- if (['write', 'edit', 'delete', 'mv'].some((p) => kind.startsWith(p)))
1112
- return 'Editing files';
1113
- if (['shell', 'bash', 'terminal'].some((p) => kind.startsWith(p)))
1114
- return 'Running a command';
1115
- return 'Working';
1116
- }
1117
886
  // Castle's agent CLI keys, delivered to the sandbox as a file
1118
887
  // (~/.castle/keys.json) rather than sandbox-wide env -- so an ambient key can't
1119
888
  // override a user's own subscription login. Falls back to process.env for
@@ -1139,41 +908,6 @@ function resolveOpenrouterAuth() {
1139
908
  return { mode: 'user-key', key: k };
1140
909
  return { mode: 'proxy', key: openrouterApiKey() };
1141
910
  }
1142
- // Keys for the SPAWNING backends' env injection (envForAgentSpawn). Smith is
1143
- // absent by design: it never spawns a CLI -- its OpenRouter key flows through
1144
- // openrouterApiKey() into runAgentNative's Authorization header instead.
1145
- const BACKEND_KEY_ENV = {
1146
- claude: 'ANTHROPIC_API_KEY',
1147
- cursor: 'CURSOR_API_KEY',
1148
- };
1149
- // When we inject Castle's key, any auth.json cursor cached from a DIFFERENT key
1150
- // -- a rotated-out old key, or a tester's own key we've chosen to override -- is
1151
- // dead weight: cursor re-auths from the injected env key and ignores it. Leaving
1152
- // it means a stale API key sits on the sandbox fs, so drop it and keep only the
1153
- // live key cached. Best-effort; an absent file is the normal case. An OAuth login
1154
- // (no apiKey field) is never reached here -- that path withholds the key instead.
1155
- function purgeStaleCursorAuth(home, injectedKey) {
1156
- try {
1157
- const authPath = cursorAuthPath(home);
1158
- const auth = JSON.parse(fs.readFileSync(authPath, 'utf8'));
1159
- if (auth.apiKey && auth.apiKey !== injectedKey) {
1160
- fs.rmSync(authPath, { force: true });
1161
- }
1162
- }
1163
- catch {
1164
- // no auth.json, or unreadable/unparseable -- nothing to purge
1165
- }
1166
- }
1167
- // True when the user has their OWN saved auth for this backend -- a login we
1168
- // route to directly (and bill to them) instead of Castle's proxy / key. The
1169
- // claude side lives in byo-auth.ts, which the editor terminal shares.
1170
- function backendHasSavedAuth(backend) {
1171
- if (backend === 'claude')
1172
- return claudeHasSavedLogin();
1173
- if (backend === 'cursor')
1174
- return cursorHasUserLogin(os.homedir());
1175
- return false;
1176
- }
1177
911
  // Env for the plain claude CLI path (NOT claude-via-OpenRouter -- that's
1178
912
  // envForOpenrouterSpawn). resolveAnthropicAuth decides the routing:
1179
913
  // - proxy: inherit the host-injected ANTHROPIC_BASE_URL + ANTHROPIC_AUTH_TOKEN
@@ -1195,32 +929,6 @@ function envForClaudeSpawn(auth) {
1195
929
  delete env[name];
1196
930
  return env;
1197
931
  }
1198
- // Env for a cursor-agent spawn: inject Castle's key ONLY when the backend has no
1199
- // saved auth of the user's own. This is what lets internal testers run on their
1200
- // own subscription (log in once in the terminal) instead of Castle's key. The
1201
- // claude path no longer routes through here -- see envForClaudeSpawn /
1202
- // resolveAnthropicAuth (this stays generic but is only ever called for cursor).
1203
- function envForAgentSpawn(backend) {
1204
- const env = { ...process.env };
1205
- const keyName = BACKEND_KEY_ENV[backend];
1206
- if (!keyName)
1207
- return env;
1208
- if (backendHasSavedAuth(backend)) {
1209
- delete env[keyName];
1210
- }
1211
- else {
1212
- const val = castleKeys()[keyName] ?? process.env[keyName];
1213
- if (val) {
1214
- env[keyName] = val;
1215
- if (backend === 'cursor')
1216
- purgeStaleCursorAuth(os.homedir(), val);
1217
- }
1218
- else {
1219
- delete env[keyName];
1220
- }
1221
- }
1222
- return env;
1223
- }
1224
932
  const DECK_TREE_EXCLUDE = new Set(['node_modules', '.castle', '.git', 'dist', '.DS_Store']);
1225
933
  const DECK_TREE_MAX_ENTRIES = 200;
1226
934
  // Per-directory listing cap. A successful deck accumulates hundreds of
@@ -1229,22 +937,11 @@ const DECK_TREE_MAX_ENTRIES = 200;
1229
937
  // from the snapshot entirely. Summarizing the overflow as "(+N more .pxart)"
1230
938
  // keeps every directory visible and turns the count itself into signal.
1231
939
  const DECK_TREE_PER_DIR = 15;
1232
- // Smith-only slimmed caps for the tree section on a turn where
1233
- // buildDeckContents (below) already inlined the deck's actual source -- the
1234
- // tree's job shrinks from "the listing" to "orientation for whatever isn't
1235
- // inlined" (art/binaries, and any overflow buildDeckContents had to list
1236
- // instead of inline), so it doesn't need the full 200-entry/15-per-dir detail
1237
- // the CLI backends still get (they have no contents section to lean on).
1238
- const DECK_TREE_SLIM_MAX_ENTRIES = 60;
1239
- const DECK_TREE_SLIM_PER_DIR = 8;
1240
940
  // Shallow orientation snapshot for router/task prompts. Best-effort by design:
1241
- // filesystem hiccups should cost context, not fail an agent turn. `caps` lets
1242
- // a smith turn ask for the slimmed version above once deck contents are
1243
- // already inlined; omitted (the CLI backends) keeps today's byte-identical
1244
- // output.
1245
- function buildDeckTree(deckDir, caps) {
1246
- const maxEntries = caps?.maxEntries ?? DECK_TREE_MAX_ENTRIES;
1247
- const perDir = caps?.perDir ?? DECK_TREE_PER_DIR;
941
+ // filesystem hiccups should cost context, not fail an agent turn.
942
+ function buildDeckTree(deckDir) {
943
+ const maxEntries = DECK_TREE_MAX_ENTRIES;
944
+ const perDir = DECK_TREE_PER_DIR;
1248
945
  const lines = [];
1249
946
  const walk = (dir, prefix, depth) => {
1250
947
  let entries;
@@ -1284,153 +981,6 @@ function buildDeckTree(deckDir, caps) {
1284
981
  walk(deckDir, '', 0);
1285
982
  return lines.join('\n');
1286
983
  }
1287
- // Text source extensions buildDeckContents will inline. Deliberately an
1288
- // allowlist (not "everything readable") -- decks are small web projects, so
1289
- // this covers what an agent actually hand-edits. SVG is included: it's a text
1290
- // format read_file already treats as source (view_image explicitly excludes
1291
- // it for the same reason). .scene is included too, and deliberately not
1292
- // filtered out despite looking like data: scenes/*.scene and
1293
- // blueprints/*.scene are JSON (actors, behaviors, layout), and they are
1294
- // exactly what a router/task agent orients on first when it opens a kit deck
1295
- // -- leaving them out just forces the hand-read this inlining exists to
1296
- // avoid (castle.json needs no separate entry -- it's already covered by
1297
- // `.json`). .pxart is deliberately NOT added even though it's also
1298
- // JSON-ish text: pixel-art drawings are bulky per-pixel grid data, not
1299
- // orientation material an agent reads to understand the deck, and they're
1300
- // meant to be produced by a tool/command, not hand-typed -- inlining them
1301
- // would just burn budget for content the agent isn't meant to
1302
- // read-then-copy-edit. True binary/art formats (images, audio, fonts) and
1303
- // anything else outside this list are skipped for the more obvious reason
1304
- // that they're unreadable as text.
1305
- const DECK_CONTENTS_TEXT_EXTS = new Set([
1306
- '.ts',
1307
- '.tsx',
1308
- '.js',
1309
- '.jsx',
1310
- '.mjs',
1311
- '.cjs',
1312
- '.mts',
1313
- '.cts',
1314
- '.json',
1315
- '.css',
1316
- '.html',
1317
- '.md',
1318
- '.txt',
1319
- '.svg',
1320
- '.yml',
1321
- '.yaml',
1322
- '.scene',
1323
- ]);
1324
- // Generated/lockfiles that happen to match the extension allowlist but are
1325
- // never hand-edited -- inlining a lockfile would just spend budget other
1326
- // files need for zero benefit (nobody reads a lockfile to plan an edit).
1327
- const DECK_CONTENTS_SKIP_NAMES = new Set(['package-lock.json', 'pnpm-lock.yaml', 'yarn.lock']);
1328
- // Per-file cap: keeps one unusually large source file (a generated bundle
1329
- // accidentally left in the tree, a huge scene) from eating the whole budget
1330
- // by itself. 16KB comfortably covers real hand-written deck files (kit engine
1331
- // modules run a few KB each) while still being small next to the total budget
1332
- // below.
1333
- const DECK_CONTENTS_PER_FILE_CAP = 16 * 1024;
1334
- // Safety valve against a pathological tree (mirrors tools.ts's
1335
- // MAX_WALK_FILES) -- decks are modest-sized web projects, this is not a
1336
- // normal limit.
1337
- const DECK_CONTENTS_MAX_WALK_FILES = 5_000;
1338
- // Smith-only total budgets (bytes), passed by the call sites below. The
1339
- // router's prompt is already the largest (rules + full transcript + the
1340
- // whole task board), so it gets a tighter ceiling than a task's -- a task
1341
- // prompt is comparatively lean and can afford to carry more of the deck.
1342
- const TASK_DECK_CONTENTS_BUDGET = 80 * 1024;
1343
- const ROUTER_DECK_CONTENTS_BUDGET = 40 * 1024;
1344
- // Full-content inlining for smith prompts (buildTaskPrompt/buildRouterPrompt's
1345
- // `deckContents`): unlike buildDeckTree (shallow, name-only, same for every
1346
- // backend), this walks the ENTIRE file set the tree covers -- no depth limit
1347
- // -- because the point is to hand smith the deck's actual code, not just its
1348
- // shape. Only smith needs this: cursor/claude read files with their own tools
1349
- // turn over turn, but smith's native loop pays a full round-trip (a tool call
1350
- // + its result re-entering context on every later iteration) for every read,
1351
- // and Castle decks are tiny enough that this rarely costs more than one
1352
- // inlining would anyway -- so front-loading it removes those round trips
1353
- // entirely for the common small/greenfield deck.
1354
- //
1355
- // `budget` (bytes) is the caller's total ceiling (tasks get more room than
1356
- // the router -- see the call sites in runTaskAgentIn / routerTurnPrompt).
1357
- // Degrades gracefully under both caps: a file over DECK_CONTENTS_PER_FILE_CAP,
1358
- // or one that would push the running total over `budget`, is listed by name
1359
- // instead of inlined (deterministic order -- once the budget is spent every
1360
- // later file lists too, it never back-fills a smaller one out of order).
1361
- function buildDeckContents(deckDir, budget) {
1362
- const files = [];
1363
- const walk = (dir) => {
1364
- if (files.length >= DECK_CONTENTS_MAX_WALK_FILES)
1365
- return;
1366
- let entries;
1367
- try {
1368
- entries = fs.readdirSync(dir, { withFileTypes: true });
1369
- }
1370
- catch {
1371
- return;
1372
- }
1373
- entries.sort((a, b) => a.name.localeCompare(b.name));
1374
- for (const entry of entries) {
1375
- if (DECK_TREE_EXCLUDE.has(entry.name))
1376
- continue;
1377
- const abs = path.join(dir, entry.name);
1378
- if (entry.isDirectory()) {
1379
- walk(abs);
1380
- }
1381
- else if (entry.isFile()) {
1382
- files.push(abs);
1383
- if (files.length >= DECK_CONTENTS_MAX_WALK_FILES)
1384
- return;
1385
- }
1386
- }
1387
- };
1388
- walk(deckDir);
1389
- const inlined = [];
1390
- const listed = [];
1391
- let used = 0;
1392
- for (const abs of files) {
1393
- const rel = path.relative(deckDir, abs).split(path.sep).join('/');
1394
- if (DECK_CONTENTS_SKIP_NAMES.has(path.basename(rel)))
1395
- continue;
1396
- if (!DECK_CONTENTS_TEXT_EXTS.has(path.extname(rel).toLowerCase()))
1397
- continue;
1398
- let stat;
1399
- try {
1400
- stat = fs.statSync(abs);
1401
- }
1402
- catch {
1403
- continue;
1404
- }
1405
- if (stat.size > DECK_CONTENTS_PER_FILE_CAP || used + stat.size > budget) {
1406
- listed.push(`${rel} (${stat.size}b, not inlined -- use read tool)`);
1407
- continue;
1408
- }
1409
- let content;
1410
- try {
1411
- content = fs.readFileSync(abs, 'utf8');
1412
- }
1413
- catch {
1414
- listed.push(`${rel} (not inlined -- use read tool)`);
1415
- continue;
1416
- }
1417
- if (content.includes('\u0000')) {
1418
- // Looks binary despite the extension allowlist (e.g. a mislabeled
1419
- // asset) -- skip it exactly like read_file's own NUL sniff would.
1420
- listed.push(`${rel} (not inlined -- use read tool)`);
1421
- continue;
1422
- }
1423
- inlined.push(`--- ${rel} ---\n${content}`);
1424
- used += stat.size;
1425
- }
1426
- const parts = [];
1427
- if (inlined.length > 0)
1428
- parts.push(inlined.join('\n\n'));
1429
- if (listed.length > 0) {
1430
- parts.push(`(not inlined -- binary/generated, over the per-file cap, or past the total budget; read these yourself if you need them):\n${listed.join('\n')}`);
1431
- }
1432
- return parts.join('\n\n');
1433
- }
1434
984
  const DEFAULT_WELCOME_MESSAGE = "Welcome to an early test of Castle's new engine! You're starting with a blank deck, without the official art or scene editors, but if HTML and JavaScript can do it then I can help you build it.\n\nDo you already know what you want to make, or do you want to figure it out together?";
1435
985
  function readClaudeSection(deckDir, section) {
1436
986
  const headingRe = new RegExp(`^## ${section}\\s*$`, 'im');
@@ -1525,7 +1075,6 @@ function createAgentStreamState() {
1525
1075
  resultIsError: false,
1526
1076
  usage: undefined,
1527
1077
  sawResult: false,
1528
- segmentText: '',
1529
1078
  needsGap: false,
1530
1079
  pendingTools: new Map(),
1531
1080
  };
@@ -1554,32 +1103,24 @@ function formatTokenCount(value) {
1554
1103
  return `${(value / 1000).toFixed(1)}k`;
1555
1104
  return String(value);
1556
1105
  }
1557
- // One usage line per finished agent run, labeled per role/backend so the
1558
- // serve's stderr is a running cost ledger for BOTH roles: `label` is
1559
- // "router" or "task <id>". Task runs log once per attempt (each attempt is
1560
- // its own CLI process / native run with its own usage).
1561
- function logAgentUsage(label, backend, usage) {
1106
+ // One usage line per finished agent run so the serve's stderr is a running
1107
+ // cost ledger for BOTH roles: `label` is "router" or "task <id>". Task runs
1108
+ // log once per attempt (each attempt is its own CLI process with its own
1109
+ // usage).
1110
+ function logAgentUsage(label, usage) {
1562
1111
  if (!usage)
1563
1112
  return;
1564
1113
  const input = formatTokenCount(usage.input_tokens);
1565
1114
  const read = formatTokenCount(usage.cache_read_input_tokens);
1566
1115
  const created = formatTokenCount(usage.cache_creation_input_tokens);
1567
1116
  const output = formatTokenCount(usage.output_tokens);
1568
- console.error(`[agent usage] ${label} ${backend}: input=${input} cache_read=${read} cache_created=${created} output=${output}`);
1117
+ console.error(`[agent usage] ${label}: input=${input} cache_read=${read} cache_created=${created} output=${output}`);
1569
1118
  }
1570
1119
  // Build the per-run stdout event handler over a shared mutable parser state.
1571
- // Splitting the cursor + claude stream decoding out of runAgentCli keeps each
1572
- // within the max-lines budget; behavior is identical (same delta/activity/
1573
- // result hooks and dedupe/segment-gap handling).
1574
- //
1575
- // Cursor closes each text segment (e.g. right before a tool call) by
1576
- // re-emitting the segment's full text as one more delta-shaped event; track the
1577
- // current segment so those re-emissions are dropped instead of duplicating
1578
- // lines. Segment boundaries also need a paragraph gap -- cursor starts the next
1579
- // segment without one, which glues "Checking the deck..." lines onto the
1580
- // previous paragraph. Claude tool_use blocks stream their input JSON by block
1581
- // index, accumulated in pendingTools so content_block_stop can label them with
1582
- // the real file / command (and drop progress-file writes).
1120
+ // Splitting the stream decoding out of runAgentCli keeps each within the
1121
+ // max-lines budget. A tool_use block streams its input JSON by block index,
1122
+ // accumulated in pendingTools so content_block_stop can label it with the
1123
+ // real file / command (and drop progress-file writes).
1583
1124
  function makeAgentEventHandler(opts, state) {
1584
1125
  const emitDelta = (rawDelta) => {
1585
1126
  let delta = rawDelta;
@@ -1589,7 +1130,6 @@ function makeAgentEventHandler(opts, state) {
1589
1130
  delta = (state.accumulated.endsWith('\n') ? '\n' : '\n\n') + delta;
1590
1131
  }
1591
1132
  }
1592
- state.segmentText += delta;
1593
1133
  state.accumulated += delta;
1594
1134
  opts.onDelta?.(delta);
1595
1135
  opts.onActivity?.(null);
@@ -1612,9 +1152,8 @@ function makeAgentEventHandler(opts, state) {
1612
1152
  }
1613
1153
  else if (e.content_block?.type === 'thinking') {
1614
1154
  state.needsGap = true;
1615
- // Surface extended thinking as the activity line (mirrors the cursor
1616
- // path's `thinking` signal). The next text delta clears it via
1617
- // emitDelta's onActivity(null); a tool block relabels it.
1155
+ // Surface extended thinking as the activity line. The next text delta
1156
+ // clears it via emitDelta's onActivity(null); a tool block relabels it.
1618
1157
  opts.onActivity?.('Thinking');
1619
1158
  }
1620
1159
  }
@@ -1662,51 +1201,11 @@ function makeAgentEventHandler(opts, state) {
1662
1201
  state.usage = parseCliUsage(ev.usage);
1663
1202
  }
1664
1203
  };
1665
- return (ev) => {
1666
- if (opts.parser === 'claude') {
1667
- handleClaudeEvent(ev);
1668
- return;
1669
- }
1670
- if (ev.type === 'assistant' && typeof ev.timestamp_ms === 'number') {
1671
- const message = ev.message;
1672
- const delta = (message?.content ?? [])
1673
- .map((c) => (typeof c?.text === 'string' ? c.text : ''))
1674
- .join('');
1675
- if (!delta)
1676
- return;
1677
- const trimmed = delta.trim();
1678
- if (trimmed.length >= 16 && state.segmentText.trim().endsWith(trimmed))
1679
- return;
1680
- emitDelta(delta);
1681
- }
1682
- else if (ev.type === 'tool_call') {
1683
- state.segmentText = '';
1684
- state.needsGap = true;
1685
- if (ev.subtype === 'started')
1686
- opts.onActivity?.(toolActivityLabel(ev));
1687
- }
1688
- else if (ev.type === 'thinking') {
1689
- state.segmentText = '';
1690
- state.needsGap = true;
1691
- opts.onActivity?.('Thinking');
1692
- }
1693
- else if (ev.type === 'result') {
1694
- state.sawResult = true;
1695
- // An EMPTY string result falls back to the streamed text. Models that
1696
- // return their blocks as `text, thinking` (gemini through OpenRouter's
1697
- // anthropic-compatible endpoint) end the turn on an empty thinking block,
1698
- // and the CLI reports `result: ""` even though the answer streamed fine.
1699
- state.finalText = typeof ev.result === 'string' && ev.result ? ev.result : state.accumulated;
1700
- state.resultIsError = ev.is_error === true;
1701
- state.usage = parseCliUsage(ev.usage);
1702
- }
1703
- };
1204
+ return handleClaudeEvent;
1704
1205
  }
1705
- // One headless agent CLI run (cursor or claude), normalized to the same
1706
- // delta/activity/result hooks. Cursor: assistant events carrying timestamp_ms
1707
- // are text deltas; the trailing assistant event without one repeats the whole
1708
- // message (skipped). Claude: stream_event wraps anthropic SSE deltas. Both
1709
- // end with a result event carrying the canonical final text.
1206
+ // One headless claude CLI run, normalized to delta/activity/result hooks:
1207
+ // stream_event wraps anthropic SSE deltas, and a result event carries the
1208
+ // canonical final text.
1710
1209
  function runAgentCli(opts) {
1711
1210
  return new Promise((resolve) => {
1712
1211
  // spawn() throws SYNCHRONOUSLY for the failures the OS rejects at exec
@@ -1792,9 +1291,7 @@ function runAgentCli(opts) {
1792
1291
  settle({
1793
1292
  ok: false,
1794
1293
  finalText: state.accumulated,
1795
- // The binary itself wouldn't start (ENOENT, EACCES). Names the command
1796
- // actually being spawned -- this used to say "cursor-agent" for every
1797
- // backend, so a missing `claude` reported the wrong tool.
1294
+ // The binary itself wouldn't start (ENOENT, EACCES).
1798
1295
  error: `could not run ${opts.command}: ${err.message}`,
1799
1296
  failure: { kind: 'spawn', detail: `${opts.command}: ${err.message}` },
1800
1297
  });
@@ -1808,8 +1305,9 @@ function runAgentCli(opts) {
1808
1305
  // do write there. Both are read BEFORE the 300-char truncation below.
1809
1306
  //
1810
1307
  // ONLY for OpenRouter-routed runs (opts.openrouterModel is set only when
1811
- // roleUsesOpenrouter): the classifier's copy is OpenRouter-branded, so a
1812
- // plain claude/cursor failure must NOT run through it -- an Anthropic 529
1308
+ // the role's model is "openrouter"): the classifier's copy is
1309
+ // OpenRouter-branded, so a
1310
+ // plain claude failure must NOT run through it -- an Anthropic 529
1813
1311
  // would otherwise read as "OpenRouter is busy", naming the wrong
1814
1312
  // provider. Plain runs keep their provider-neutral "exit" copy, exactly
1815
1313
  // as before this feature.
@@ -1837,94 +1335,6 @@ function runAgentCli(opts) {
1837
1335
  });
1838
1336
  });
1839
1337
  }
1840
- // One smith (native castle agent) run, adapted to runAgentCli's contract so
1841
- // every caller downstream of runAgentTurn is backend-agnostic:
1842
- // - NativeRunResult.text -> finalText; error/usage/crashed pass through by
1843
- // name. `ok` is derived as !error && !crashed -- there is no process exit
1844
- // code; those two fields are the whole story.
1845
- // - Cancellation: one AbortController per run, registered in the same
1846
- // `children` set the CLI runs use, via a handle whose kill() aborts it
1847
- // (see AgentRunHandle). Interrupts (killRouterChildren), task halts
1848
- // (haltTask), and serve shutdown all just work. An aborted run resolves
1849
- // with error "agent run stopped" and NO crashed flag -- same downstream
1850
- // behavior as a SIGKILLed CLI child (router: the epoch check routes to
1851
- // settleInterruptedTurn; tasks: stopRequested returns before the retry
1852
- // check).
1853
- // - Timeout is enforced INSIDE runAgentNative (same "agent run timed out"
1854
- // error string runAgentCli produces, so classifyRouterFailure's buckets
1855
- // hold); connect failures surface as "could not run openrouter: ..."
1856
- // matching the "could not run" spawn bucket.
1857
- // - logPath appends structured JSONL run events to the SAME files the CLI
1858
- // backends stream their raw log into (tasks/<id>/log.jsonl,
1859
- // .castle/agent/router-log.jsonl) -- see createRunLogger in native/loop.ts
1860
- // for the line shapes.
1861
- // - attachments (router turns): deck-relative image paths that become
1862
- // image_url content parts in the run's initial user message -- the native
1863
- // tool set can't read image bytes (read_file rejects binaries), unlike the
1864
- // CLI agents which re-read attachment files themselves.
1865
- async function runAgentSmith(opts) {
1866
- const controller = new AbortController();
1867
- const handle = makeSmithRunHandle(controller);
1868
- opts.children.add(handle);
1869
- // The pseudo-pid lands in task.pid so haltTask can find this run's handle
1870
- // in the shared children set, exactly like a CLI child's real pid.
1871
- opts.onSpawn?.(handle.pid);
1872
- try {
1873
- const result = await runAgentNative({
1874
- cwd: opts.cwd,
1875
- role: opts.role,
1876
- extraHeaders: opts.extraHeaders,
1877
- model: opts.model,
1878
- apiKey: opts.apiKey,
1879
- baseUrl: opts.baseUrl,
1880
- reasoningEffort: opts.openrouterTuning?.reasoningEffort,
1881
- routing: opts.openrouterTuning?.routing,
1882
- // "" (auto) becomes undefined so no provider.order is sent.
1883
- providerTier: opts.openrouterTuning?.providerTier || undefined,
1884
- prompt: opts.prompt,
1885
- system: opts.system,
1886
- systemReminder: opts.systemReminder,
1887
- attachments: opts.attachments,
1888
- timeoutMs: opts.timeoutMs,
1889
- logPath: opts.logPath,
1890
- playtest: opts.playtest,
1891
- restart: opts.restart,
1892
- signal: controller.signal,
1893
- onDelta: opts.onDelta,
1894
- onActivity: opts.onActivity,
1895
- onThinking: opts.onThinking,
1896
- });
1897
- return {
1898
- ok: !result.error && !result.crashed,
1899
- finalText: result.text,
1900
- error: result.error,
1901
- failure: result.failure,
1902
- usage: result.usage,
1903
- playtestFrames: result.playtestFrames,
1904
- crashed: result.crashed,
1905
- };
1906
- }
1907
- catch (err) {
1908
- // runAgentNative resolves (never rejects) by contract; this is pure
1909
- // belt-and-braces so an unexpected throw degrades like a spawn failure
1910
- // instead of rejecting through startTask's catch.
1911
- return {
1912
- ok: false,
1913
- finalText: '',
1914
- error: `could not run openrouter: ${err instanceof Error ? err.message : String(err)}`,
1915
- };
1916
- }
1917
- finally {
1918
- opts.children.delete(handle);
1919
- }
1920
- }
1921
- // True when a role's path goes through OpenRouter: the smith native loop, or
1922
- // the claude CLI routed at OpenRouter. The client has a twin of this in
1923
- // conductor.tsx (it decides whether to show the slug field) -- keep them in
1924
- // step; this one decides whether the slug is worth validating at all.
1925
- export function roleUsesOpenrouter(backend, claudeModel) {
1926
- return backend === 'smith' || (backend === 'claude' && claudeModel === 'openrouter');
1927
- }
1928
1338
  function configFailure(reason, detail, extra) {
1929
1339
  return { kind: 'config', reason, detail, ...extra };
1930
1340
  }
@@ -1938,7 +1348,7 @@ function configFailure(reason, detail, extra) {
1938
1348
  // resolve "unavailable" on any network trouble, and we allow the run rather
1939
1349
  // than invent a verdict.
1940
1350
  async function preflightOpenrouterRun(opts) {
1941
- if (!roleUsesOpenrouter(opts.backend, opts.claudeModel))
1351
+ if (opts.claudeModel !== 'openrouter')
1942
1352
  return null;
1943
1353
  const auth = opts.orAuth ?? resolveOpenrouterAuth();
1944
1354
  const apiKey = auth.key;
@@ -1989,37 +1399,37 @@ function notifyAgentRunFinished() {
1989
1399
  listener();
1990
1400
  }
1991
1401
  // True when THIS run spends Castle's credential rather than the user's own.
1992
- // Resolves exactly as buildAgentInvocation / runAgentSmith will, one step
1993
- // earlier, so the gate and the run can never disagree about who is paying.
1994
- //
1995
- // Cursor asks the same question as the others, just from a different file:
1996
- // envForAgentSpawn withholds Castle's key when the user has an OAuth login, so
1997
- // a run with one is already going to their account and the daily limit is not
1998
- // Castle's to enforce over it. It used to answer `true` unconditionally, which
1999
- // refused a user on their own subscription once Castle's budget ran out --
2000
- // while the editor offered them that sign-in and promised it bypassed the
2001
- // limit. Deferring only to a LOGIN is what keeps this from being a hole: a
2002
- // cursor run on Castle's key stays gated, and cursor is unmetered (its traffic
2003
- // never reaches the proxy), so an ungated one would be a free ride.
2004
- function runIsCastlePaid(backend, claudeModel, orAuth) {
2005
- if (backend === 'cursor')
2006
- return !cursorHasUserLogin(os.homedir());
2007
- return roleUsesOpenrouter(backend, claudeModel)
1402
+ // Resolves exactly as buildAgentInvocation will, one step earlier, so the
1403
+ // gate and the run can never disagree about who is paying.
1404
+ function runIsCastlePaid(claudeModel, orAuth) {
1405
+ return claudeModel === 'openrouter'
2008
1406
  ? (orAuth ?? resolveOpenrouterAuth()).mode === 'proxy'
2009
1407
  : resolveAnthropicAuth().mode === 'proxy';
2010
1408
  }
2011
1409
  // Whether Castle's budget is this editor's to spend at all, which is what makes
2012
1410
  // the usage bar worth drawing. Both roles count, and they can disagree: an
2013
- // Anthropic login covers the claude roles while a cursor role on Castle's key
2014
- // still spends Castle's. Re-read per refresh, so adding or removing a
2015
- // credential (or switching a role's backend) moves the bar on the next one.
2016
- //
2017
- // A cursor role on Castle's key does draw a bar its own runs never move --
2018
- // cursor is unmetered by contract. That is a display question, not a payment
2019
- // one, and it is not settled here.
1411
+ // Anthropic login covers a role on a claude alias while a role routed through
1412
+ // OpenRouter still spends Castle's key. Re-read per refresh, so adding or
1413
+ // removing a credential (or switching a role's model) moves the bar on the
1414
+ // next one.
2020
1415
  function anyRoleIsCastlePaid(settings) {
2021
- return (runIsCastlePaid(settings.router, settings.routerClaudeModel, null) ||
2022
- runIsCastlePaid(settings.tasks, settings.tasksClaudeModel, null));
1416
+ return roleSpends(settings).some((role) => role.castlePaid);
1417
+ }
1418
+ // The two roles as the spend questions see them: who pays, and -- when the
1419
+ // path goes through OpenRouter at all -- which slug the payment is for. Both
1420
+ // halves are answered the way the RUN answers them (runIsCastlePaid), so "this
1421
+ // editor is free" can never disagree with what the proxy actually bills. A role
1422
+ // that doesn't reach OpenRouter carries no slug: the fixed claude aliases are
1423
+ // metered whatever is free.
1424
+ function roleSpends(settings) {
1425
+ return ['router', 'tasks'].map((role) => {
1426
+ const claudeModel = role === 'router' ? settings.routerClaudeModel : settings.tasksClaudeModel;
1427
+ const slug = role === 'router' ? settings.routerOpenrouterModel : settings.tasksOpenrouterModel;
1428
+ return {
1429
+ castlePaid: runIsCastlePaid(claudeModel, null),
1430
+ slug: claudeModel === 'openrouter' ? slug : null,
1431
+ };
1432
+ });
2023
1433
  }
2024
1434
  // The proxy 403s a spent-out user -- or one who asked for a model they don't
2025
1435
  // have -- mid-stream, which a CLI surfaces as a generic provider error after a
@@ -2029,13 +1439,16 @@ function anyRoleIsCastlePaid(settings) {
2029
1439
  // Only the claude aliases are checked, which is exactly what the picker offers.
2030
1440
  // A free-form OpenRouter slug naming a restricted model is left to the proxy:
2031
1441
  // resolving an arbitrary slug to what it bills as is its job, not the editor's.
2032
- async function castleSpendRefusal(backend, claudeModel, orAuth) {
2033
- if (!runIsCastlePaid(backend, claudeModel, orAuth))
1442
+ async function castleSpendRefusal(claudeModel, openrouterModel, orAuth) {
1443
+ if (!runIsCastlePaid(claudeModel, orAuth))
2034
1444
  return null;
2035
1445
  const budget = await fetchBudget();
2036
1446
  if (!budget)
2037
1447
  return null;
2038
- if (claudeModelBlocked(claudeModel, budget)) {
1448
+ // Only a run on a fixed alias reaches the proxy as that alias; on
1449
+ // claude-via-OpenRouter the stored alias is an inactive leftover, and
1450
+ // refusing on it would block a run that never asks for it.
1451
+ if (claudeModel !== 'openrouter' && claudeModelBlocked(claudeModel, budget)) {
2039
1452
  return {
2040
1453
  kind: 'config',
2041
1454
  reason: 'model-not-allowed',
@@ -2043,6 +1456,15 @@ async function castleSpendRefusal(backend, claudeModel, orAuth) {
2043
1456
  model: claudeModel,
2044
1457
  };
2045
1458
  }
1459
+ // Credit state cannot pre-empt a model Castle pays for: an empty balance and
1460
+ // a reached daily cap both describe budget this run will not touch, and
1461
+ // refusing here would strand a user on the free default with no way to keep
1462
+ // working. The free model has its own (invisible) daily cap, enforced where
1463
+ // every other spend rule is -- at the proxy, whose 403 carries copy of its
1464
+ // own (castleFreeModelLimit in agent-failures).
1465
+ if (claudeModel === 'openrouter' && modelIsFree(openrouterModel, budget.freeModelPrefixes)) {
1466
+ return null;
1467
+ }
2046
1468
  if (!budget.blocked)
2047
1469
  return null;
2048
1470
  const credits = await fetchAiCredits();
@@ -2068,8 +1490,8 @@ const PICKER_CLAUDE_MODELS = ['sonnet', 'opus', 'fable'];
2068
1490
  * Which of the picker's claude models this editor can't use. Gated on the
2069
1491
  * ANTHROPIC credential specifically, not on `anyRoleIsCastlePaid` (which draws
2070
1492
  * the usage bar): those two disagree exactly when one role runs on Castle's
2071
- * OpenRouter key -- or on Castle's cursor key -- while the user's own Anthropic
2072
- * key or login covers every claude run. Those runs never reach the proxy, so
1493
+ * OpenRouter key while the user's own Anthropic key or login covers every
1494
+ * claude run. Those runs never reach the proxy, so
2073
1495
  * nothing about them is Castle's to restrict, and the picker must keep offering
2074
1496
  * the model. A claude run on a fixed alias always resolves through
2075
1497
  * resolveAnthropicAuth, so no role's settings enter into this.
@@ -2079,7 +1501,7 @@ function blockedClaudeModels(budget) {
2079
1501
  return [];
2080
1502
  return PICKER_CLAUDE_MODELS.filter((m) => claudeModelBlocked(m, budget));
2081
1503
  }
2082
- function usageFrame(budget, credits) {
1504
+ function usageFrame(budget, credits, settings) {
2083
1505
  if (!budget)
2084
1506
  return null;
2085
1507
  return {
@@ -2091,6 +1513,11 @@ function usageFrame(budget, credits) {
2091
1513
  credits,
2092
1514
  creditsExhausted: castleCreditsExhausted(budget, credits),
2093
1515
  blockedClaudeModels: blockedClaudeModels(budget),
1516
+ freeTier: freeTierForRoles(roleSpends(settings), budget.freeModelPrefixes),
1517
+ freeModel: runIsCastlePaid('openrouter', null)
1518
+ ? (budget.freeModelPrefixes[0] ?? null)
1519
+ : null,
1520
+ freeRoles: freeRolesFor(roleSpends(settings), budget.freeModelPrefixes),
2094
1521
  };
2095
1522
  }
2096
1523
  /**
@@ -2111,7 +1538,8 @@ function createUsageFeed(opts) {
2111
1538
  let credits = null;
2112
1539
  let budgetPolls = 0;
2113
1540
  async function readUsage(refreshCredits = false) {
2114
- if (!opts.castlePaid()) {
1541
+ const settings = opts.settings();
1542
+ if (!anyRoleIsCastlePaid(settings)) {
2115
1543
  credits = null;
2116
1544
  return null;
2117
1545
  }
@@ -2127,7 +1555,7 @@ function createUsageFeed(opts) {
2127
1555
  if (nextCredits)
2128
1556
  credits = nextCredits;
2129
1557
  }
2130
- return usageFrame(budget, credits);
1558
+ return usageFrame(budget, credits, settings);
2131
1559
  }
2132
1560
  function publishUsage(next) {
2133
1561
  if (JSON.stringify(next ?? null) === JSON.stringify(latest ?? null))
@@ -2159,17 +1587,13 @@ function createUsageFeed(opts) {
2159
1587
  },
2160
1588
  };
2161
1589
  }
2162
- // The one backend dispatch point for running an agent turn: smith runs
2163
- // in-process (runAgentSmith -> runAgentNative); cursor/claude spawn a CLI
2164
- // (buildAgentInvocation -> runAgentCli). Everything downstream consumes the
2165
- // same CliRunResult contract either way.
1590
+ // One agent turn: buildAgentInvocation -> runAgentCli, with the pre-flight
1591
+ // refusals that stop a run before it spawns.
2166
1592
  async function runAgentTurn(opts) {
2167
1593
  // Resolve the OpenRouter credential ONCE and reuse it for pre-flight, the run,
2168
1594
  // and metering, so validation, spend, and the direct/proxy routing decision
2169
1595
  // can never diverge. Null on a non-OpenRouter path.
2170
- const orAuth = roleUsesOpenrouter(opts.backend, opts.claudeModel)
2171
- ? resolveOpenrouterAuth()
2172
- : null;
1596
+ const orAuth = opts.claudeModel === 'openrouter' ? resolveOpenrouterAuth() : null;
2173
1597
  // Deterministic config errors stop here: nothing spawned, no request issued,
2174
1598
  // nothing billed. Returned (not thrown) because the callers' catch paths
2175
1599
  // emit generic "something went wrong" copy, which would bury the specific
@@ -2177,7 +1601,7 @@ async function runAgentTurn(opts) {
2177
1601
  // refuses is the same shape of answer, and comes second so a misconfigured
2178
1602
  // run is still reported as misconfigured.
2179
1603
  const failure = (await preflightOpenrouterRun({ ...opts, orAuth })) ??
2180
- (await castleSpendRefusal(opts.backend, opts.claudeModel, orAuth));
1604
+ (await castleSpendRefusal(opts.claudeModel, opts.openrouterModel, orAuth));
2181
1605
  if (failure) {
2182
1606
  return {
2183
1607
  ok: false,
@@ -2194,46 +1618,11 @@ async function runAgentTurn(opts) {
2194
1618
  // conversation -- nothing is resumed).
2195
1619
  const sessionId = newAgentSessionId(opts.role);
2196
1620
  const settled = (run) => run.finally(() => notifyAgentRunFinished());
2197
- if (opts.backend === 'smith') {
2198
- // roleUsesOpenrouter is true for smith, so orAuth is non-null here.
2199
- const direct = orAuth.mode === 'user-key';
2200
- return settled(runAgentSmith({
2201
- cwd: opts.cwd,
2202
- role: opts.role,
2203
- apiKey: orAuth.key,
2204
- baseUrl: direct ? OPENROUTER_DIRECT_CHAT_BASE : undefined,
2205
- extraHeaders: meteringHeaders({
2206
- deckDir: opts.cwd,
2207
- sessionId,
2208
- route: 'openrouter',
2209
- direct,
2210
- }),
2211
- model: opts.openrouterModel,
2212
- prompt: opts.prompt,
2213
- system: opts.systemPrompt,
2214
- // Mirrors claude's --append-system-prompt for tasks (the native loop
2215
- // appends it to its own system framing).
2216
- systemReminder: opts.role === 'task' ? CLAUDE_TASK_SYSTEM_REMINDER : undefined,
2217
- attachments: opts.attachments,
2218
- openrouterTuning: opts.openrouterTuning,
2219
- timeoutMs: opts.timeoutMs,
2220
- logPath: opts.logPath,
2221
- playtest: opts.playtest,
2222
- restart: opts.restart,
2223
- children: opts.children,
2224
- onDelta: opts.onDelta,
2225
- onActivity: opts.onActivity,
2226
- onThinking: opts.onThinking,
2227
- onSpawn: opts.onSpawn,
2228
- }));
2229
- }
2230
- const invocation = buildAgentInvocation(opts.backend, opts.role, opts.prompt, opts.claudeModel, opts.openrouterModel, { sessionId, deckDir: opts.cwd }, opts.taskId, opts.systemPrompt);
2231
- const startedMs = Date.now();
1621
+ const invocation = buildAgentInvocation(opts.role, opts.prompt, opts.claudeModel, opts.openrouterModel, { sessionId, deckDir: opts.cwd }, opts.taskId, opts.systemPrompt);
2232
1622
  const run = runAgentCli({
2233
1623
  cwd: opts.cwd,
2234
1624
  command: invocation.command,
2235
1625
  args: invocation.args,
2236
- parser: parserForBackend(opts.backend),
2237
1626
  env: invocation.env,
2238
1627
  timeoutMs: opts.timeoutMs,
2239
1628
  logPath: opts.logPath,
@@ -2243,25 +1632,9 @@ async function runAgentTurn(opts) {
2243
1632
  onThinking: opts.onThinking,
2244
1633
  onSpawn: opts.onSpawn,
2245
1634
  labelUnknownTools: opts.labelUnknownTools,
2246
- openrouterModel: roleUsesOpenrouter(opts.backend, opts.claudeModel)
2247
- ? opts.openrouterModel
2248
- : undefined,
1635
+ openrouterModel: opts.claudeModel === 'openrouter' ? opts.openrouterModel : undefined,
2249
1636
  });
2250
- // Cursor is the only backend whose traffic never reaches the llm-proxy, so its
2251
- // run is reported from here. Every other backend is already recorded upstream,
2252
- // and reporting them here too would double-count them in the same table.
2253
- if (opts.backend !== 'cursor')
2254
- return settled(run);
2255
- return settled(run.then((result) => {
2256
- reportCursorRun({
2257
- deckDir: opts.cwd,
2258
- sessionId,
2259
- model: CURSOR_MODEL,
2260
- durationMs: Date.now() - startedMs,
2261
- ok: result.ok,
2262
- });
2263
- return result;
2264
- }));
1637
+ return settled(run);
2265
1638
  }
2266
1639
  // -- task store ---------------------------------------------------------------
2267
1640
  function persistTaskFile(tasksDir, task) {
@@ -2513,7 +1886,7 @@ function sleep(ms) {
2513
1886
  // Chained-promise gate spacing out task-agent PROCESS LAUNCHES -- deliberately
2514
1887
  // NOT a queue class, just a promise each launch chains onto. Why: the
2515
1888
  // 2026-07 keychain-contention incident (see TASK_SPAWN_STAGGER_MS above) was
2516
- // triggered by several cursor-agent processes starting at the exact same
1889
+ // triggered by several agent CLI processes starting at the exact same
2517
1890
  // instant; spacing consecutive launches out by TASK_SPAWN_STAGGER_MS avoids
2518
1891
  // that without limiting how many can run concurrently once they're up.
2519
1892
  let taskSpawnGate = Promise.resolve();
@@ -2536,13 +1909,6 @@ async function waitBeforeTaskRetry(attempt, stopRequested, taskId) {
2536
1909
  async function runTaskAgentIn(ctx, task) {
2537
1910
  const dir = path.join(ctx.tasksDir, task.id);
2538
1911
  const relDir = path.relative(ctx.deckDir, dir);
2539
- // Deck source inlining is smith-only (see buildDeckContents): cursor/claude
2540
- // read files themselves turn over turn, so computing and sending this
2541
- // would be pure prompt bloat for them with no offsetting benefit.
2542
- const isSmith = ctx.backend === 'smith';
2543
- const deckContents = isSmith
2544
- ? buildDeckContents(ctx.deckDir, TASK_DECK_CONTENTS_BUDGET)
2545
- : undefined;
2546
1912
  const taskPrompt = buildTaskPrompt({
2547
1913
  deckLabel: ctx.deckLabel,
2548
1914
  taskId: task.id,
@@ -2552,14 +1918,9 @@ async function runTaskAgentIn(ctx, task) {
2552
1918
  notesPath: path.join(relDir, 'notes.md'),
2553
1919
  handoffPath: path.join(relDir, 'handoff'),
2554
1920
  depsSummary: ctx.depsSummary,
2555
- backend: ctx.backend,
2556
- playtest: playtestSteering(ctx.backend),
2557
- // Slimmed once contents are inlined -- see DECK_TREE_SLIM_* above.
2558
- deckTree: buildDeckTree(ctx.deckDir, isSmith
2559
- ? { maxEntries: DECK_TREE_SLIM_MAX_ENTRIES, perDir: DECK_TREE_SLIM_PER_DIR }
2560
- : undefined),
2561
- deckContents,
2562
- quickReference: quickReferenceFor(ctx.deckDir, ctx.quickReference, ctx.backend),
1921
+ playtest: playtestSteering(),
1922
+ deckTree: buildDeckTree(ctx.deckDir),
1923
+ quickReference: quickReferenceFor(ctx.deckDir, ctx.quickReference),
2563
1924
  plan: taskPlanOpts(ctx.deckDir, task.item, ctx.finishedWork),
2564
1925
  siblings: ctx.siblings,
2565
1926
  });
@@ -2569,14 +1930,6 @@ async function runTaskAgentIn(ctx, task) {
2569
1930
  // can't-verify-live-play loop. A single message + the autonomy reminder
2570
1931
  // appended to the system prompt (CLAUDE_TASK_SYSTEM_REMINDER) is enough to
2571
1932
  // get the work done; the agent stops when the code is in, the user verifies.
2572
- // Pre-warm the playtest browser download as soon as a smith task starts
2573
- // (fire-and-forget; single-flight shared with any real playtest call), so
2574
- // the ~250MB one-time install usually completes during the agent's early
2575
- // read/plan turns instead of stalling its first playtest. Smith-only:
2576
- // playtest is a native-loop tool, so CLI-backend tasks would download a
2577
- // browser they can never use.
2578
- if (ctx.backend === 'smith')
2579
- ctx.playtest?.prewarm?.();
2580
1933
  let result = { ok: false, finalText: '', error: 'not run' };
2581
1934
  let lineBuf = '';
2582
1935
  // ```signal blocks span multiple lines and must NOT show in the live feed:
@@ -2616,24 +1969,14 @@ async function runTaskAgentIn(ctx, task) {
2616
1969
  for (let attempt = 1; attempt <= MAX_TASK_ATTEMPTS; attempt++) {
2617
1970
  await staggerTaskSpawn();
2618
1971
  result = await runAgentTurn({
2619
- backend: ctx.backend,
2620
1972
  role: 'task',
2621
1973
  prompt: taskPrompt,
2622
1974
  claudeModel: ctx.claudeModel,
2623
1975
  openrouterModel: ctx.openrouterModel,
2624
- openrouterTuning: ctx.openrouterTuning,
2625
1976
  cwd: ctx.deckDir,
2626
1977
  timeoutMs: TASK_TIMEOUT_MS,
2627
1978
  logPath: path.join(dir, 'log.jsonl'),
2628
- playtest: ctx.playtest
2629
- ? {
2630
- executor: ctx.playtest.executor,
2631
- serveUrl: ctx.playtest.serveUrl,
2632
- framesDir: path.join(dir, 'playtest'),
2633
- }
2634
- : undefined,
2635
1979
  taskId: task.id,
2636
- restart: ctx.restart,
2637
1980
  children: ctx.children,
2638
1981
  onSpawn: (pid) => {
2639
1982
  task.pid = pid;
@@ -2645,7 +1988,7 @@ async function runTaskAgentIn(ctx, task) {
2645
1988
  ctx.onFeed(`[${activity}]`);
2646
1989
  },
2647
1990
  });
2648
- logAgentUsage(`task ${task.id}`, ctx.backend, result.usage);
1991
+ logAgentUsage(`task ${task.id}`, result.usage);
2649
1992
  if (ctx.stopRequested.has(task.id))
2650
1993
  return result;
2651
1994
  if (!result.crashed)
@@ -2684,14 +2027,11 @@ function startTask(ctx, task) {
2684
2027
  deckLabel: ctx.deckLabel,
2685
2028
  tasksDir: ctx.tasksDir,
2686
2029
  children: ctx.children,
2687
- backend: ctx.backend(),
2688
2030
  claudeModel: ctx.claudeModel(),
2689
2031
  openrouterModel: ctx.openrouterModel(),
2690
- openrouterTuning: ctx.openrouterTuning(),
2691
2032
  stopRequested: ctx.stopRequested,
2692
2033
  quickReference: ctx.quickReference,
2693
2034
  playtest: ctx.playtest,
2694
- restart: ctx.restart,
2695
2035
  depsSummary: depsSummaryFor(ctx.tasks, task),
2696
2036
  // Same visibility rule as the router's board (hide acked+finished rows),
2697
2037
  // minus this task itself. Snapshot at start -- consistent with the deck
@@ -2737,12 +2077,6 @@ function startTask(ctx, task) {
2737
2077
  task.acknowledged = true;
2738
2078
  if (result.ok && !wasStopped)
2739
2079
  task.progress = 100;
2740
- // Smith reports its frames on the run result; a claude agent's reach
2741
- // the record through runPlaytestFor as each call lands. Overwriting
2742
- // with `?? []` would erase the latter, since a CLI backend's result
2743
- // never carries them.
2744
- if (result.playtestFrames)
2745
- task.playtestFrames = result.playtestFrames;
2746
2080
  task.finishedAt = nowIso();
2747
2081
  task.resultSummary = wasStopped
2748
2082
  ? 'stopped by the router'
@@ -2900,12 +2234,9 @@ function createTaskStore(opts) {
2900
2234
  tasks,
2901
2235
  stopRequested,
2902
2236
  quickReference: opts.quickReference,
2903
- backend: opts.backend,
2904
2237
  claudeModel: opts.claudeModel,
2905
2238
  openrouterModel: opts.openrouterModel,
2906
- openrouterTuning: opts.openrouterTuning,
2907
2239
  playtest: opts.playtest,
2908
- restart: opts.restart,
2909
2240
  onStarted: opts.onStarted,
2910
2241
  onFinished: opts.onFinished,
2911
2242
  onRetry: opts.onRetry,
@@ -3160,7 +2491,7 @@ const ERROR_PREVIEW_CHARS = 200;
3160
2491
  // tracing to '/var/folders/.../cursor_retrieval....log'" while the real
3161
2492
  // "Error: Security command failed: Security process exited with code: 44"
3162
2493
  // sat right below it. Only these two wrapper shapes get this treatment --
3163
- // "agent run timed out" and "could not run cursor-agent: <message>" (the
2494
+ // "agent run timed out" and "could not run claude: <message>" (the
3164
2495
  // other two runAgentCli error strings) are already the whole story on line 1
3165
2496
  // and are left alone, same as before.
3166
2497
  const EXIT_WRAPPER_RE = /^agent (exited \d+|process kept dying)/;
@@ -3398,18 +2729,11 @@ function resolveFailure(result) {
3398
2729
  // context + transcript replay (minus log lines and the in-flight reply) +
3399
2730
  // the live board + this turn's instruction.
3400
2731
  function routerTurnPrompt(ctx, instruction, selfMessageId, plan) {
3401
- // Smith-only, smaller budget than a task's -- see TASK_DECK_CONTENTS_BUDGET/
3402
- // ROUTER_DECK_CONTENTS_BUDGET's comment (the router prompt is already the
3403
- // largest one this serve builds).
3404
- const isSmith = ctx.backend() === 'smith';
3405
2732
  return buildRouterPromptParts({
3406
2733
  deckLabel: ctx.deckLabel,
3407
- quickReference: quickReferenceFor(ctx.deckDir, ctx.quickReference, ctx.backend()),
3408
- deckTree: buildDeckTree(ctx.deckDir, isSmith
3409
- ? { maxEntries: DECK_TREE_SLIM_MAX_ENTRIES, perDir: DECK_TREE_SLIM_PER_DIR }
3410
- : undefined),
3411
- deckContents: isSmith ? buildDeckContents(ctx.deckDir, ROUTER_DECK_CONTENTS_BUDGET) : undefined,
3412
- playtest: playtestSteering(ctx.tasksBackend()),
2734
+ quickReference: quickReferenceFor(ctx.deckDir, ctx.quickReference),
2735
+ deckTree: buildDeckTree(ctx.deckDir),
2736
+ playtest: playtestSteering(),
3413
2737
  messages: ctx.log.messages
3414
2738
  .filter((m) => m.role !== 'log' && m.id !== selfMessageId && m.status !== 'streaming')
3415
2739
  .map((m) => ({
@@ -3508,10 +2832,7 @@ function spawnCompletedTaskFences(ctx, message, midStream, raw) {
3508
2832
  }
3509
2833
  // One router turn: stream a reply message, then spawn the directives it
3510
2834
  // emitted (unless a newer user message superseded this turn).
3511
- // `attachments` = deck-relative image paths from the drained sends; only
3512
- // smith consumes them (image content blocks), the CLI backends read the
3513
- // files themselves via the paths already named in the instruction text.
3514
- function runRouterTurnIn(ctx, instruction, attachments = []) {
2835
+ function runRouterTurnIn(ctx, instruction) {
3515
2836
  const epoch = ctx.currentEpoch();
3516
2837
  const message = {
3517
2838
  id: nanoid(8),
@@ -3534,20 +2855,14 @@ function runRouterTurnIn(ctx, instruction, attachments = []) {
3534
2855
  let lastActivity = 'Thinking';
3535
2856
  ctx.broadcast({ type: 'message-activity', id: message.id, activity: 'Thinking' });
3536
2857
  const prompt = routerTurnPrompt(ctx, instruction, message.id, readPlanSnapshot(ctx.deckDir));
3537
- const backend = ctx.backend();
3538
2858
  void runAgentTurn({
3539
- backend,
3540
2859
  role: 'router',
3541
2860
  prompt: prompt.user,
3542
2861
  systemPrompt: prompt.system,
3543
2862
  claudeModel: ctx.claudeModel(),
3544
2863
  openrouterModel: ctx.openrouterModel(),
3545
- openrouterTuning: ctx.openrouterTuning(),
3546
- attachments,
3547
2864
  cwd: ctx.deckDir,
3548
2865
  timeoutMs: ROUTER_TIMEOUT_MS,
3549
- // CLI backends append raw stream-json; smith appends structured run
3550
- // events (see createRunLogger in native/loop.ts). Same file either way.
3551
2866
  logPath: path.join(ctx.agentDir, 'router-log.jsonl'),
3552
2867
  children: ctx.children,
3553
2868
  labelUnknownTools: true,
@@ -3575,7 +2890,7 @@ function runRouterTurnIn(ctx, instruction, attachments = []) {
3575
2890
  },
3576
2891
  })
3577
2892
  .then((result) => {
3578
- logAgentUsage('router', backend, result.usage);
2893
+ logAgentUsage('router', result.usage);
3579
2894
  // Signals the finally -> onSettled(retryable): the turn failed cleanly
3580
2895
  // enough (transient, nothing salvaged) that the queue may re-run it.
3581
2896
  let retryable = false;
@@ -3726,13 +3041,6 @@ function runRouterTurnIn(ctx, instruction, attachments = []) {
3726
3041
  function applyAgentSettings(incoming, ctx) {
3727
3042
  const { settings } = ctx;
3728
3043
  const changes = [];
3729
- for (const key of ['router', 'tasks']) {
3730
- const value = normalizeBackend(incoming[key]);
3731
- if (value && value !== settings[key]) {
3732
- settings[key] = value;
3733
- changes.push(`${key} agent -> ${value}`);
3734
- }
3735
- }
3736
3044
  for (const key of ['routerClaudeModel', 'tasksClaudeModel']) {
3737
3045
  const value = normalizeClaudeModel(incoming[key]);
3738
3046
  if (value && value !== settings[key]) {
@@ -3747,29 +3055,6 @@ function applyAgentSettings(incoming, ctx) {
3747
3055
  changes.push(`${key === 'routerOpenrouterModel' ? 'operator' : 'tasks'} openrouter model -> ${value}`);
3748
3056
  }
3749
3057
  }
3750
- for (const key of ['routerReasoningEffort', 'tasksReasoningEffort']) {
3751
- const value = normalizeReasoningEffort(incoming[key]);
3752
- if (value && value !== settings[key]) {
3753
- settings[key] = value;
3754
- changes.push(`${key === 'routerReasoningEffort' ? 'operator' : 'tasks'} reasoning effort -> ${value}`);
3755
- }
3756
- }
3757
- for (const key of ['routerRouting', 'tasksRouting']) {
3758
- const value = normalizeRoutingMode(incoming[key]);
3759
- if (value && value !== settings[key]) {
3760
- settings[key] = value;
3761
- changes.push(`${key === 'routerRouting' ? 'operator' : 'tasks'} routing -> ${value}`);
3762
- }
3763
- }
3764
- for (const key of ['routerProviderTier', 'tasksProviderTier']) {
3765
- // "" is a valid value (auto), so check for null (invalid) explicitly
3766
- // rather than truthiness -- otherwise the tier could never be cleared.
3767
- const value = normalizeProviderTier(incoming[key]);
3768
- if (value !== null && value !== settings[key]) {
3769
- settings[key] = value;
3770
- changes.push(`${key === 'routerProviderTier' ? 'operator' : 'tasks'} provider tier -> ${value || 'auto'}`);
3771
- }
3772
- }
3773
3058
  if (changes.length === 0)
3774
3059
  return;
3775
3060
  atomicWriteFileSync(ctx.settingsPath, serializeAgentSettings(settings));
@@ -3784,8 +3069,8 @@ function applyAgentSettings(incoming, ctx) {
3784
3069
  const ACCOUNTS_POLL_MS = 15_000;
3785
3070
  // The editor's view of which credentials exist, kept live. Unlike settings --
3786
3071
  // which nothing outside this process writes -- a credential changes behind the
3787
- // editor's back all the time: `claude /login` or `cursor-agent logout` in the
3788
- // terminal panel, a hand-edited user-keys.json, an agent run that refreshes an
3072
+ // editor's back all the time: `claude /login` in the terminal panel, a
3073
+ // hand-edited user-keys.json, an agent run that refreshes an
3789
3074
  // OAuth token. Push those, or the popover reports whatever was true when the
3790
3075
  // page loaded and a terminal sign-in reads as having done nothing.
3791
3076
  //
@@ -3853,12 +3138,10 @@ function applyCredentialChange(msg, ctx) {
3853
3138
  // set at all.
3854
3139
  function slugKeysToValidate(settings) {
3855
3140
  const keys = [];
3856
- if (roleUsesOpenrouter(settings.router ?? 'claude', settings.routerClaudeModel ?? 'opus') &&
3857
- settings.routerOpenrouterModel) {
3141
+ if (settings.routerClaudeModel === 'openrouter' && settings.routerOpenrouterModel) {
3858
3142
  keys.push('routerOpenrouterModel');
3859
3143
  }
3860
- if (roleUsesOpenrouter(settings.tasks ?? 'claude', settings.tasksClaudeModel ?? 'sonnet') &&
3861
- settings.tasksOpenrouterModel) {
3144
+ if (settings.tasksClaudeModel === 'openrouter' && settings.tasksOpenrouterModel) {
3862
3145
  keys.push('tasksOpenrouterModel');
3863
3146
  }
3864
3147
  return keys;
@@ -3908,7 +3191,7 @@ function killOrphanAgents(registryPath) {
3908
3191
  const cmd = execFileSync('ps', ['-p', String(entry.pid), '-o', 'command='], {
3909
3192
  encoding: 'utf8',
3910
3193
  }).trim();
3911
- if (cmd.includes('cursor-agent') || cmd.includes('claude')) {
3194
+ if (cmd.includes('claude')) {
3912
3195
  process.kill(entry.pid, 'SIGKILL');
3913
3196
  }
3914
3197
  }
@@ -3929,9 +3212,7 @@ function startChildRegistry(registryPath, groups) {
3929
3212
  const live = [];
3930
3213
  for (const group of groups) {
3931
3214
  for (const child of group) {
3932
- // pid > 0 excludes smith runs (negative pseudo-pids, no OS process
3933
- // for the orphan sweep to kill -- see AgentRunHandle).
3934
- if (typeof child.pid === 'number' && child.pid > 0 && child.exitCode === null) {
3215
+ if (typeof child.pid === 'number' && child.exitCode === null) {
3935
3216
  live.push({ pid: child.pid, command: child.spawnfile });
3936
3217
  }
3937
3218
  }
@@ -4042,9 +3323,8 @@ function killRouterChildren(ctx) {
4042
3323
  }
4043
3324
  return drafts.join('\n\n');
4044
3325
  }
4045
- function startRouterTurn(ctx, instruction, attachments = []) {
3326
+ function startRouterTurn(ctx, instruction) {
4046
3327
  ctx.state.lastInstruction = instruction;
4047
- ctx.state.lastAttachments = attachments;
4048
3328
  runRouterTurnIn({
4049
3329
  deckDir: ctx.deckDir,
4050
3330
  deckLabel: ctx.deckLabel,
@@ -4056,18 +3336,11 @@ function startRouterTurn(ctx, instruction, attachments = []) {
4056
3336
  taskStore: ctx.taskStore,
4057
3337
  messageThinking: ctx.messageThinking,
4058
3338
  currentEpoch: () => ctx.state.userEpoch,
4059
- backend: () => ctx.settings.router,
4060
- tasksBackend: () => ctx.settings.tasks,
4061
3339
  claudeModel: () => ctx.settings.routerClaudeModel,
4062
3340
  openrouterModel: () => ctx.settings.routerOpenrouterModel,
4063
- openrouterTuning: () => ({
4064
- reasoningEffort: ctx.settings.routerReasoningEffort,
4065
- routing: ctx.settings.routerRouting,
4066
- providerTier: ctx.settings.routerProviderTier,
4067
- }),
4068
3341
  canAutoRetry: () => !ctx.state.autoRetryUsed && ctx.state.pendingSends.length === 0,
4069
3342
  onSettled: (retryable) => onRouterQueueSettled(ctx, retryable),
4070
- }, instruction, attachments);
3343
+ }, instruction);
4071
3344
  }
4072
3345
  // Commit each drained send to the message log (skipping ones already logged
4073
3346
  // -- see PendingSend.logged -- so a re-carried send doesn't double its user
@@ -4127,7 +3400,7 @@ function maybeStartRouterQueueTurn(ctx) {
4127
3400
  interruptedDraft: draft || undefined,
4128
3401
  attachments: attachmentPaths,
4129
3402
  refs,
4130
- }), attachmentPaths);
3403
+ }));
4131
3404
  }
4132
3405
  // The turn settled: clear the busy flag, broadcast it, then flush anything
4133
3406
  // that queued mid-turn (a clean end and an interrupt take the same path). A
@@ -4144,7 +3417,7 @@ function onRouterQueueSettled(ctx, retryable) {
4144
3417
  state.autoRetryUsed = true;
4145
3418
  state.routerRunning = true;
4146
3419
  broadcastQueueState(ctx);
4147
- startRouterTurn(ctx, state.lastInstruction, state.lastAttachments);
3420
+ startRouterTurn(ctx, state.lastInstruction);
4148
3421
  return;
4149
3422
  }
4150
3423
  broadcastQueueState(ctx);
@@ -4279,7 +3552,6 @@ function createRouterQueue(deps) {
4279
3552
  autoRetryUsed: false,
4280
3553
  autoFoldUsed: false,
4281
3554
  lastInstruction: '',
4282
- lastAttachments: [],
4283
3555
  },
4284
3556
  };
4285
3557
  // Restart recovery: re-enqueue sends that never reached the message log,
@@ -4298,12 +3570,10 @@ function createRouterQueue(deps) {
4298
3570
  // The serve-side half of the claude backend's playtest tool: resolve which
4299
3571
  // task's frames dir and call budget a request belongs to, run it on the shared
4300
3572
  // warm Chromium, and record the frames on the task so its card shows them.
4301
- // Smith never comes through here -- native/loop.ts builds its own ctx.
4302
3573
  function makePlaytestRunner(deps) {
4303
- // Per-TASK playtest budget, so PLAYTEST_MAX_CALLS_PER_RUN means the same
4304
- // thing to a claude agent as to smith. Stricter than smith's, which resets
4305
- // per attempt (one runAgentNative invocation): a retried claude task keeps
4306
- // spending the same four calls.
3574
+ // Per-TASK playtest budget: a retried task keeps spending the same
3575
+ // PLAYTEST_MAX_CALLS_PER_RUN calls rather than getting a fresh set per
3576
+ // attempt.
4307
3577
  const callCounts = new Map();
4308
3578
  return async function runPlaytestFor(req) {
4309
3579
  const task = req.taskId ? deps.taskStore.get(req.taskId) : undefined;
@@ -4374,12 +3644,6 @@ export function createAgentServer(opts) {
4374
3644
  // with a navigation error. `localhost` resolves correctly under
4375
3645
  // every binding mode (default ::1, --host 0.0.0.0, explicit IPs).
4376
3646
  serveUrl: opts.playtestServeUrl ?? `http://localhost:${opts.port}`,
4377
- // Kicks off the one-time Chromium download in the background at
4378
- // smith-task start (see runTaskAgentIn) -- single-flight with any
4379
- // real playtest call, silent on every failure path.
4380
- prewarm: playtestBrowserManager
4381
- ? () => playtestBrowserManager.prewarm()
4382
- : undefined,
4383
3647
  }
4384
3648
  : undefined;
4385
3649
  const taskChildren = new Set();
@@ -4407,7 +3671,7 @@ export function createAgentServer(opts) {
4407
3671
  const usageFeed = createUsageFeed({
4408
3672
  broadcast,
4409
3673
  hasClients: () => clients.size > 0,
4410
- castlePaid: () => anyRoleIsCastlePaid(settings),
3674
+ settings: () => settings,
4411
3675
  });
4412
3676
  const accountsFeed = createAccountsFeed({
4413
3677
  broadcast,
@@ -4424,8 +3688,8 @@ export function createAgentServer(opts) {
4424
3688
  });
4425
3689
  const applySettings = (incoming) => {
4426
3690
  applyAgentSettings(incoming, { settings, settingsPath, broadcast });
4427
- // A backend switch can change who pays (a cursor role always spends
4428
- // Castle's), so the bar follows the change instead of the poll.
3691
+ // A model switch can change who pays, and whether that spend is free at
3692
+ // all (usage.freeTier), so the bar follows the change instead of the poll.
4429
3693
  usageFeed.refresh();
4430
3694
  };
4431
3695
  const taskFeeds = createTaskFeeds(broadcast, tasksDir);
@@ -4436,16 +3700,9 @@ export function createAgentServer(opts) {
4436
3700
  tasksDir,
4437
3701
  children: taskChildren,
4438
3702
  quickReference,
4439
- backend: () => settings.tasks,
4440
3703
  openrouterModel: () => settings.tasksOpenrouterModel,
4441
3704
  claudeModel: () => settings.tasksClaudeModel,
4442
- openrouterTuning: () => ({
4443
- reasoningEffort: settings.tasksReasoningEffort,
4444
- routing: settings.tasksRouting,
4445
- providerTier: settings.tasksProviderTier,
4446
- }),
4447
3705
  playtest,
4448
- restart: opts.restart,
4449
3706
  // Task lifecycle stays on the board only -- log lines for it were spam.
4450
3707
  onUpdate: (task) => broadcast({ type: 'task-update', task: asClientTask(task, opts.signPath) }),
4451
3708
  onStarted: () => undefined,
@@ -4600,9 +3857,8 @@ export function createAgentServer(opts) {
4600
3857
  applyCredentialChange(msg, { pushAccounts: accountsFeed.push });
4601
3858
  }
4602
3859
  else if (msg.type === 'account-login') {
4603
- const provider = loginProviderFor(msg.id);
4604
- if (provider)
4605
- startLogin(provider, () => accountsFeed.push());
3860
+ if (loginProviderFor(msg.id))
3861
+ startLogin(() => accountsFeed.push());
4606
3862
  }
4607
3863
  else if (msg.type === 'account-login-code' && typeof msg.value === 'string') {
4608
3864
  submitLoginCode(msg.value);
@@ -4611,9 +3867,8 @@ export function createAgentServer(opts) {
4611
3867
  cancelLogin();
4612
3868
  }
4613
3869
  else if (msg.type === 'account-logout') {
4614
- const provider = loginProviderFor(msg.id);
4615
- if (provider)
4616
- logout(provider, () => accountsFeed.push());
3870
+ if (loginProviderFor(msg.id))
3871
+ logout(() => accountsFeed.push());
4617
3872
  }
4618
3873
  else if (msg.type === 'client-timezone' && typeof msg.timeZone === 'string') {
4619
3874
  setReaderTimeZone(msg.timeZone);
@@ -4642,8 +3897,6 @@ export function createAgentServer(opts) {
4642
3897
  const handleAttachment = makeAttachmentHandler(attachmentsDir);
4643
3898
  const handlePlaytestFrame = makePlaytestFrameHandler(tasksDir);
4644
3899
  function handleHttpRequest(req, res, reqPath) {
4645
- if (reqPath === AGENT_MODEL_CAPS_PREFIX)
4646
- return handleModelCaps(req, res);
4647
3900
  return handleAttachment(req, res, reqPath) || handlePlaytestFrame(req, res, reqPath);
4648
3901
  }
4649
3902
  function shutdown() {