@coreplane/switchboard 1.254.1 → 1.256.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (66) hide show
  1. package/dist/assets/.dockerignore +3 -0
  2. package/dist/assets/Dockerfile +12 -1
  3. package/dist/assets/config/config.example.yaml +6 -1
  4. package/dist/assets/deploy/cloudflare/worker.ts +39 -23
  5. package/dist/assets/deploy/cloudflare-memory/worker.ts +552 -5
  6. package/dist/assets/deploy/cloudflare-memory/wrangler.template.jsonc +5 -0
  7. package/dist/assets/deploy/cloudflare-resident/Dockerfile +13 -1
  8. package/dist/assets/deploy/cloudflare-resident/drain.ts +55 -1
  9. package/dist/assets/deploy/cloudflare-resident/levels.ts +84 -0
  10. package/dist/assets/deploy/cloudflare-resident/prepare-commit-msg +17 -0
  11. package/dist/assets/deploy/cloudflare-resident/worker.ts +358 -42
  12. package/dist/assets/deploy/cloudflare-sandbox/Dockerfile +11 -1
  13. package/dist/assets/deploy/cloudflare-sandbox/prepare-commit-msg +17 -0
  14. package/dist/assets/deploy/cloudflare-sandbox/worker.ts +21 -4
  15. package/dist/assets/deploy/hooks/prepare-commit-msg +17 -0
  16. package/dist/assets/deploy/secrets.manifest.json +12 -0
  17. package/dist/assets/package-lock.json +3 -3
  18. package/dist/assets/package.json +1 -1
  19. package/dist/assets/source.json +3 -3
  20. package/dist/assets/src/agents/registry.ts +52 -0
  21. package/dist/assets/src/core/budgets.ts +35 -2
  22. package/dist/assets/src/core/coordinator/contract.ts +6 -0
  23. package/dist/assets/src/core/coordinator/driver.ts +59 -6
  24. package/dist/assets/src/core/costs.ts +39 -16
  25. package/dist/assets/src/core/pipelineStanding.ts +62 -0
  26. package/dist/assets/src/core/plane/decide.ts +389 -22
  27. package/dist/assets/src/core/refusal.ts +3 -0
  28. package/dist/assets/src/core/reviewVerdict.ts +64 -2
  29. package/dist/assets/src/core/runEvents.ts +31 -12
  30. package/dist/assets/src/core/runLedger/sessionLog.ts +128 -0
  31. package/dist/assets/src/core/runLedger/types.ts +3 -0
  32. package/dist/assets/src/core/runRecord.ts +24 -7
  33. package/dist/assets/src/core/ship/contract.ts +20 -30
  34. package/dist/assets/src/core/ship/coordinator.ts +519 -126
  35. package/dist/assets/src/core/trace/workerTrace.ts +3 -0
  36. package/dist/assets/src/execution/sandboxErrors.ts +77 -6
  37. package/dist/assets/web/dist/.vite/manifest.json +61 -60
  38. package/dist/assets/web/dist/assets/CostsPage-BaeWnm-o.js +1 -0
  39. package/dist/assets/web/dist/assets/{DeliveryPage-ngPsO2to.js → DeliveryPage-BBAyLwPq.js} +1 -1
  40. package/dist/assets/web/dist/assets/{HomePage-DvxTHzPx.js → HomePage-BEiStFwW.js} +1 -1
  41. package/dist/assets/web/dist/assets/PendingTurnRow-DGINv9XT.js +1 -0
  42. package/dist/assets/web/dist/assets/{PlanePage-DpWfiX4C.js → PlanePage-JEj-lqgz.js} +1 -1
  43. package/dist/assets/web/dist/assets/{ResidentDetailPage-DG86v39Y.js → ResidentDetailPage-jYhEeeyu.js} +1 -1
  44. package/dist/assets/web/dist/assets/{ResidentsIndexPage-x6p689VH.js → ResidentsIndexPage-DvFmlV4U.js} +1 -1
  45. package/dist/assets/web/dist/assets/RunFoldRow-DvWzR1JQ.js +1 -0
  46. package/dist/assets/web/dist/assets/RunRoutePage-CvZ-TOT3.js +9 -0
  47. package/dist/assets/web/dist/assets/{RunsIndexPage-CuzFchcn.js → RunsIndexPage-JqVSDOok.js} +1 -1
  48. package/dist/assets/web/dist/assets/{ScheduledPage-BuLmfcbG.js → ScheduledPage-Viim-bus.js} +1 -1
  49. package/dist/assets/web/dist/assets/{SettingsPage-BujWkdU_.js → SettingsPage-CYUy8McC.js} +1 -1
  50. package/dist/assets/web/dist/assets/{StatusDot-C8Bc0pTX.js → StatusDot-Dv6UMaPy.js} +1 -1
  51. package/dist/assets/web/dist/assets/{Tooltip-BWwJx27K.js → Tooltip-Bd-5Rypv.js} +1 -1
  52. package/dist/assets/web/dist/assets/UnitRoutePage-B81wEnKN.js +1 -0
  53. package/dist/assets/web/dist/assets/budgets-c1eumrqD.js +1 -0
  54. package/dist/assets/web/dist/assets/{dist-CpnyQGOb.js → dist-Nl3uaxrP.js} +1 -1
  55. package/dist/assets/web/dist/assets/indexRow-DborJPFp.js +1 -0
  56. package/dist/assets/web/dist/assets/{main-Comxmwi4.js → main-DRxWlffc.js} +2 -2
  57. package/dist/assets/web/dist/assets/{sseReplay-IzTdD4-3.js → sseReplay-DE6wv1Ua.js} +6 -6
  58. package/dist/cli.js +4400 -3079
  59. package/package.json +1 -1
  60. package/dist/assets/web/dist/assets/CostsPage-BuKjw3nv.js +0 -1
  61. package/dist/assets/web/dist/assets/PendingTurnRow-ZYIRCCZ2.js +0 -1
  62. package/dist/assets/web/dist/assets/RunFoldRow-DG29LOTs.js +0 -1
  63. package/dist/assets/web/dist/assets/RunRoutePage-ysJBY8xQ.js +0 -9
  64. package/dist/assets/web/dist/assets/UnitRoutePage-B9kjA1AT.js +0 -1
  65. package/dist/assets/web/dist/assets/budgets-CbIyPAER.js +0 -1
  66. package/dist/assets/web/dist/assets/indexRow-DABQtONT.js +0 -1
@@ -0,0 +1,17 @@
1
+ #!/bin/sh
2
+ # The agent trailer (docs/reference/specs/execution.md): every commit made in a
3
+ # Switchboard image carries `Co-Authored-By: <bot pair>`, so the agent's hand
4
+ # stays visible even when the author is the requester. The pair is read from
5
+ # GIT_COMMITTER_NAME/GIT_COMMITTER_EMAIL at commit time — the bot fills them
6
+ # per exec — so the image stays installation-agnostic; without them the
7
+ # image's own git identity stands (its fallback address is off the GitHub
8
+ # domain, so it never renders as a GitHub account). Idempotent: a message that
9
+ # already carries this exact trailer is left unchanged; a foreign
10
+ # Co-Authored-By does not stop it (the identity rewrite scrubs those).
11
+ set -e
12
+ msg="$1"
13
+ name="${GIT_COMMITTER_NAME:-$(git config user.name || true)}"
14
+ email="${GIT_COMMITTER_EMAIL:-$(git config user.email || true)}"
15
+ [ -n "$name" ] && [ -n "$email" ] || exit 0
16
+ git interpret-trailers --in-place --if-exists addIfDifferent \
17
+ --trailer "Co-Authored-By: $name <$email>" "$msg"
@@ -34,6 +34,18 @@
34
34
  "optional": true,
35
35
  "note": "Anthropic Admin API key (sk-ant-admin…) for the LLM-spend layer of GET /costs. Absent, /costs shows the Cloudflare side only."
36
36
  },
37
+ {
38
+ "name": "OPENROUTER_MANAGEMENT_KEY",
39
+ "workers": ["bot"],
40
+ "optional": true,
41
+ "note": "OpenRouter management key for the openrouter biller's invoice tie-out on GET /costs (providers.openrouter.invoiceKeyEnv). Never the inference key. Absent, that biller shows no invoice."
42
+ },
43
+ {
44
+ "name": "OPENAI_ADMIN_KEY",
45
+ "workers": ["bot"],
46
+ "optional": true,
47
+ "note": "OpenAI admin key for the openai biller's invoice tie-out on GET /costs (providers.openai.invoiceKeyEnv). Never the inference key. Absent, that biller shows no invoice."
48
+ },
37
49
  {
38
50
  "name": "BRAVE_SEARCH_API_KEY",
39
51
  "workers": ["bot"],
@@ -1,12 +1,12 @@
1
1
  {
2
2
  "name": "switchboard",
3
- "version": "1.254.1",
3
+ "version": "1.256.0",
4
4
  "lockfileVersion": 3,
5
5
  "requires": true,
6
6
  "packages": {
7
7
  "": {
8
8
  "name": "switchboard",
9
- "version": "1.254.1",
9
+ "version": "1.256.0",
10
10
  "license": "Apache-2.0",
11
11
  "workspaces": [
12
12
  "web",
@@ -20445,7 +20445,7 @@
20445
20445
  },
20446
20446
  "packages/switchboard": {
20447
20447
  "name": "@coreplane/switchboard",
20448
- "version": "1.254.1",
20448
+ "version": "1.256.0",
20449
20449
  "license": "Apache-2.0",
20450
20450
  "dependencies": {
20451
20451
  "@earendil-works/pi-ai": "0.85.1",
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "switchboard",
3
- "version": "1.254.1",
3
+ "version": "1.256.0",
4
4
  "private": true,
5
5
  "description": "Mention it in Slack and an agent reviews the PR, ships the fix, or answers the question — on the model you choose, with its tools running where you decide.",
6
6
  "license": "Apache-2.0",
@@ -1,5 +1,5 @@
1
1
  {
2
- "version": "1.254.1",
3
- "commit": "48a9a480354e503b8fea59dc926f94132cf79756",
4
- "builtAt": "2026-09-19T21:28:07.753Z"
2
+ "version": "1.256.0",
3
+ "commit": "fd814ee9ad2421f38178b15103d49ff4d9c5eb7c",
4
+ "builtAt": "2026-09-20T04:37:22.536Z"
5
5
  }
@@ -47,6 +47,14 @@ export function machineNeedsRepo(machine: MachineClass): boolean {
47
47
  export const IDENTITIES = ["none", "read", "write"] as const;
48
48
  export type Identity = (typeof IDENTITIES)[number];
49
49
 
50
+ /** The model tiers a preset may run on (the one-door plan's tiers rule): the
51
+ * `fast` tier is the router's own model (`routing.model`), everything else is
52
+ * `strong`. Each preset declares its allowed set below (`AgentDef.tiers`);
53
+ * a parent choosing a child's model at spawn is held to the child preset's
54
+ * set, and escalation is a new run — a run's tier is fixed at dispatch. */
55
+ export const MODEL_TIERS = ["fast", "strong"] as const;
56
+ export type ModelTier = (typeof MODEL_TIERS)[number];
57
+
50
58
  /** The wall clocks live in `src/core/budgets.ts` (docs/decisions/0046): a
51
59
  * preset's ask, the turn cap derived from it and every allowance are rows
52
60
  * there, and this registry reads them. Re-exported for the readers that
@@ -89,6 +97,12 @@ export interface AgentDef {
89
97
  * worktree flag and the token the sandbox env and the `repo-cold` vet mint
90
98
  * read this, through the run's effective profile. */
91
99
  identity: Identity;
100
+ /** The model tiers this preset may run on (`MODEL_TIERS`): what a parent's
101
+ * spawn — and the operator's bind — may put in the child's request slot.
102
+ * A preset that writes code (`coding`, `ship`, `review`) never includes
103
+ * `fast`: a wrong approval or a wrong edit costs more than the tokens
104
+ * saved; `explore` and `research` read, and may run fast. */
105
+ tiers: readonly ModelTier[];
92
106
  /** Whether the request router (docs/reference/specs/routing-and-config.md
93
107
  * item 21) may pick this preset for a plain message. Absent means yes: the
94
108
  * router's table is rendered from this registry. `false` keeps a preset
@@ -240,6 +254,31 @@ Keep notes with the \`notes\` tool: one short document, replaced whole each time
240
254
  // classes are by duration, the project's own scripts and CI say which is which.
241
255
  export const CHECKS_BY_COST = `CHECKS BY COST — push before the expensive ones. Every check you might run has a cost class: seconds (a formatter or a linter on the files you touched, one test file, a docs, link or spec check, the typecheck of one package) or minutes (the whole test suite, a build, a dependency install, an end-to-end or full verification). Know a command's class before you run it — from the project's own scripts and CI configuration, from how long it took last time, or by the class above when you have nothing better. Prove each change with the cheapest check that can prove it, matched to the change's scope and scoped to the changed set — the tests nearest your change, the touched project's typecheck, the changed files' formatting, never the whole tree: a documentation change gets the documentation checks, one module gets its own tests, a shared type gets the typecheck. Every CI pipeline runs the tests, the types, the formatting and the full verification on your push, so you never run them again: you validate and fix your own change before pushing, at the changed-set scope. Passing the full test suite and the full typecheck is NOT part of your criteria: CI is that gate and the only place they run — on a shared machine they cost minutes that every other run pays for. As soon as the change exists and those checks pass, commit and push — the pushed branch is the deliverable, and an unpushed tree does not survive the run's end. Beyond the changed set, use judgement about what this change needs rather than a checklist, fixing forward with further commits and pushes. Never start an operation whose expected duration does not fit the time you have left minus what a commit, a push and the description need: push what there is and say plainly what is unverified instead. At the wind-down note, commit and push what compiles, say what does not, then answer. ${TIMEOUT_ON_LONG_COMMANDS} The description's validation names exactly what ran; what did not run is CI's to gate, and you say so.`;
242
256
 
257
+ // Every coding prompt carries this verbatim, right after the checks-by-cost
258
+ // rule (docs/reference/specs/agent-coding.md item 13; issue 1796): the fast
259
+ // gates before every push, each the changed-set form with its command named,
260
+ // and the full verification named as CI's gate. The paragraph lived in the
261
+ // ship contract's first instruction alone, so every ask that was not a plan
262
+ // unit had to repeat it by hand or watch the run spend most of its budget on
263
+ // the whole suite or the full verification before its first push (issue 1909
264
+ // measured three such command shapes at 85–95 % of a child's life). One
265
+ // constant, one text: the three coding variants carry the same bytes, the
266
+ // contract's first instruction points here instead of re-stating it
267
+ // (src/core/ship/contract.ts), and no requester repeats it. Unlike the
268
+ // checks-by-cost rule above, this paragraph names its commands on purpose —
269
+ // children handed only the classes chose wrong in both directions — and a
270
+ // repository on another stack maps each gate by its class.
271
+ export const TOUCHED_TESTS_COMMAND = "`npx vitest run` on the test files you touched, by name,";
272
+
273
+ export const FAST_GATES_BEFORE_PUSH =
274
+ "THE FAST GATES, before every push — each scoped to the changed set, never the whole project: " +
275
+ `${TOUCHED_TESTS_COMMAND} once (never \`--changed\`, never a directory: on a moving base that is most of the suite), ` +
276
+ "`tsc --noEmit -p` the touched tsconfig under `NODE_OPTIONS=--max-old-space-size=6144`, " +
277
+ "`npx prettier --check` on the changed files, `npm run hygiene:check` and `npm run specs:check` — " +
278
+ "then your judgement on what else this change needs, not a longer checklist. The full verification is " +
279
+ "CI's gate — `npm run verify` runs there on your push, never here: push a head early and let CI judge it, " +
280
+ "fixing forward with further commits and pushes.";
281
+
243
282
  const CODING_SYSTEM = `You are Switchboard's coding agent, operating from a Slack request.
244
283
 
245
284
  You work inside a dedicated workspace directory with bash, read_file, and write_file tools. ${SANDBOX_TOOLCHAIN}
@@ -262,6 +301,8 @@ Workflow for shipping a PR:
262
301
 
263
302
  ${CHECKS_BY_COST}
264
303
 
304
+ ${FAST_GATES_BEFORE_PUSH}
305
+
265
306
  ${NEVER_MERGE}
266
307
 
267
308
  ${UNIT_CONTRACT}
@@ -309,6 +350,8 @@ Workflow for shipping a change:
309
350
 
310
351
  ${CHECKS_BY_COST}
311
352
 
353
+ ${FAST_GATES_BEFORE_PUSH}
354
+
312
355
  ${NEVER_MERGE}
313
356
 
314
357
  ${UNIT_CONTRACT}
@@ -352,6 +395,8 @@ Workflow for shipping a change:
352
395
 
353
396
  ${CHECKS_BY_COST}
354
397
 
398
+ ${FAST_GATES_BEFORE_PUSH}
399
+
355
400
  ${NEVER_MERGE}
356
401
 
357
402
  ${UNIT_CONTRACT}
@@ -664,6 +709,7 @@ const WORK_PRESETS = {
664
709
  identity: "none",
665
710
  maxTokens: 16000,
666
711
  ...loopBudget("general"),
712
+ tiers: ["fast", "strong"],
667
713
  },
668
714
  coding: {
669
715
  name: "coding",
@@ -678,6 +724,7 @@ const WORK_PRESETS = {
678
724
  // `config set channel efforts.coding=…`, or `effort:` per request).
679
725
  machine: "repo-resident",
680
726
  identity: "write", // pushes branches and opens pull requests
727
+ tiers: ["strong"], // code-writing never runs fast (the one-door plan's tiers rule)
681
728
  // Never routed: a plain write ask deserves the coding → review loop, so
682
729
  // the router's table offers `ship` in coding's seat — a routed ship runs
683
730
  // a generated one-unit plan whose merge is a person's, never the runner's.
@@ -693,6 +740,7 @@ const WORK_PRESETS = {
693
740
  toolset: "readonly",
694
741
  machine: "repo-resident",
695
742
  identity: "read", // a read-scoped token and a read-only worktree: it cannot post or push from inside
743
+ tiers: ["strong"], // a wrong finding costs a merge decision: reviews never run fast
696
744
  maxTokens: 64000,
697
745
  ...loopBudget("review"), // a safety net — typical reviews land in ~5 minutes
698
746
  effort: "medium", // fast turns; one big-context pass does the deep work
@@ -717,6 +765,7 @@ const WORK_PRESETS = {
717
765
  toolset: "full",
718
766
  machine: "repo-resident",
719
767
  identity: "write",
768
+ tiers: ["strong"], // the pipeline's children write and review code: never fast
720
769
  // Routable: a routed ship runs a generated one-unit plan whose merge is a
721
770
  // person's (`merge: person`) and never a seeded plan (the hand-off refuses
722
771
  // a routed `plan <path>.md` naming `agent:ship`), so a wrong route costs a
@@ -733,6 +782,7 @@ const WORK_PRESETS = {
733
782
  toolset: "web",
734
783
  machine: "none", // web I/O only; no workspace is provisioned
735
784
  identity: "none",
785
+ tiers: ["fast", "strong"], // reads and reports: may run fast
736
786
  maxTokens: 24000,
737
787
  ...loopBudget("research"),
738
788
  effort: "medium",
@@ -747,6 +797,7 @@ const WORK_PRESETS = {
747
797
  // review depends on: a two-hour job shares no container with anyone.
748
798
  machine: "repo-cold",
749
799
  identity: "read", // a read-scoped token: it can clone and read, never push — whatever the caller holds
800
+ tiers: ["fast", "strong"], // reads and reports: may run fast
750
801
  maxTokens: 64000,
751
802
  ...loopBudget("explore"),
752
803
  // No built-in effort: the deployment decides, as for coding.
@@ -761,6 +812,7 @@ export const AGENTS: Record<string, AgentDef> = {
761
812
  "Coordinates other runs: spawns child runs as the requester — each in a thread of its own, under their permissions — follows them, and reports. No workspace or shell.",
762
813
  system: conductorSystem(Object.values(WORK_PRESETS)),
763
814
  toolset: "conductor",
815
+ tiers: ["fast", "strong"], // reads and coordinates: may run fast
764
816
  // Nothing is provisioned and no credential minted: the run tools call the
765
817
  // dispatcher, the GitHub reads are REST in the bot process.
766
818
  machine: "none",
@@ -46,6 +46,20 @@ export const CONFIRMATION_TTL_MS = 10 * MINUTE_MS;
46
46
  * and only loses answers. People answer a question in minutes to hours. */
47
47
  export const QUESTION_TTL_MS = DAY_MS;
48
48
 
49
+ /** How long a resolved author binding is reused before the stored `{ login,
50
+ * id }` pair is re-read from GitHub (docs/decisions/0062;
51
+ * docs/reference/specs/authorization.md item 18). The commit identity pairs
52
+ * are resolved per EXEC (`gitIdentityEnvs`), and the pi harness's log polls,
53
+ * FIFO sends and file writes ride the same executor — without a cache a
54
+ * write run costs one `GET /user/<id>` per command, drawing down the
55
+ * installation's shared rate limit for a freshness nothing needs: the
56
+ * identity rewrite re-reads the binding fresh before the PR opens, so a
57
+ * rename is caught there whatever this window holds. Read by `bindingOf`
58
+ * (`src/execution/authorBinding.ts`) for the resolved pair and the rename
59
+ * refusal alike (one `[identity]` line per window, not one per exec); a read
60
+ * that FAILED is never cached — the next read asks again. */
61
+ export const AUTHOR_BINDING_TTL_MS = 15 * MINUTE_MS;
62
+
49
63
  /** How long a dispatch's FIRST attach to a resident — a fresh run's, or a
50
64
  * resumed run's re-attach to its recorded worktree — waits for the resident
51
65
  * to wake when the Worker typed its refusal as the platform's transient
@@ -172,6 +186,13 @@ export const FLOORS: Readonly<Record<RoundKind, number>> = {
172
186
  * when the remainder allows it. */
173
187
  export const MERGE_WAIT_ASK_MINUTES = 60;
174
188
 
189
+ /** The provider retry ladder (issue 1932): the backoff before each retry of a
190
+ * transient model-call failure — a gateway 5xx, a stream cut before
191
+ * `message_stop`, a gateway timeout. Three attempts with growing waits,
192
+ * charged to the run's lease; the waits stay small next to the run's minutes
193
+ * because the observed blips are edge transients of seconds. */
194
+ export const PROVIDER_RETRY_BACKOFFS_MS = [5_000, 15_000, 45_000] as const;
195
+
175
196
  /** A hosted ship parent's deadline margin past the pipeline's wall clock
176
197
  * (record 0060): the row's `state.hosting.until` is the hand-off time plus
177
198
  * the instance's `caps.maxMinutes` plus this hour, absorbing the runner's own
@@ -183,8 +204,11 @@ export const HOSTED_DEADLINE_MARGIN_MINUTES = 60;
183
204
  * and the pause before a busy spawn is asked again. */
184
205
  export const SHIP_WAIT = { marginMinutes: 5, chunkMinutes: 5, mergeChunkMinutes: 5, busyRetryMinutes: 2 } as const;
185
206
 
186
- /** The plane's table (docs/decisions/0064): how long a finished run stays on it. */
187
- export const PLANE = { recentMinutes: 60 } as const;
207
+ /** The plane's table (docs/decisions/0064): how long a finished run stays on
208
+ * it, a reservation's window, and the default re-ask cadence — how often a
209
+ * silent resident something waits on is probed (`plane.reaskMinutes`
210
+ * overrides it per deployment). */
211
+ export const PLANE = { recentMinutes: 60, reservationMinutes: 2, reaskMinutes: 2 } as const;
188
212
 
189
213
  /** The named amounts a lease holds back, in minutes. Each stands for a step
190
214
  * every run or round pays: `provision` is attach and restore before the
@@ -473,6 +497,15 @@ export const DEFAULT_GRANT: Grant = { renewals: 0 };
473
497
  * in one message before a plan should carry it. */
474
498
  export const GRANT_RENEWALS_MAX = 12;
475
499
 
500
+ // ---- the structured-answer seam (record 0067: a violation is re-asked) ----
501
+
502
+ /** The most re-asks one structured ask may make (`askStructured`,
503
+ * src/core/dispatch/structured.ts): a named violation is quoted back to the
504
+ * same model at most this many times; the next violation is the caller's
505
+ * declared floor. A count, not a duration — the caller's one timeout covers
506
+ * the whole loop. */
507
+ export const STRUCTURED_RETRIES_MAX = 2;
508
+
476
509
  // ---- the idle unit (record 0051: a pipeline idles instead of ending) ----
477
510
 
478
511
  /** `ship.idleDays`' default: zero — today's endings byte for byte. It moves to
@@ -347,6 +347,11 @@ export interface CoordinatorInstance {
347
347
  * attempt ended reruns the units not merged under `plan-<plan-id>-<attempt>`);
348
348
  * absent for the first. */
349
349
  attempt?: number;
350
+ /** The hard stop's mark (record 0060; issue 1924): written when the hosted
351
+ * parent is sealed, read by the runner before every unit start and before
352
+ * every child spawn — it honours the mark by ending the remaining units
353
+ * `stopped` and running nothing more. */
354
+ stop?: { at: number };
350
355
  }
351
356
 
352
357
  export interface UnitSegment {
@@ -491,6 +496,7 @@ export function isCoordinatorInstance(v: unknown): v is CoordinatorInstance {
491
496
  if (!isOptionalText(r.runId) || !isOptionalText(r.label)) return false;
492
497
  if (r.verbosity !== undefined && !isVerbosity(r.verbosity)) return false;
493
498
  if (r.attempt !== undefined && !(Number.isInteger(r.attempt) && (r.attempt as number) >= 2)) return false;
499
+ if (r.stop !== undefined && !(isObject(r.stop) && isFinite(r.stop.at))) return false;
494
500
  return true;
495
501
  }
496
502
 
@@ -209,6 +209,9 @@ interface PlanFacts {
209
209
  generated: boolean;
210
210
  /** The runs page base the bot answered: the report links a child's write-up to its run page with it. */
211
211
  runPageBase?: string;
212
+ /** The hard stop's mark (record 0060; issue 1924): the hosted parent was
213
+ * sealed, so the walk ends the remaining units stopped and runs nothing more. */
214
+ stopped: boolean;
212
215
  repo: string;
213
216
  base: string;
214
217
  caps: ShipCaps;
@@ -263,6 +266,7 @@ function readPlan(a: BotAnswer): PlanFacts {
263
266
  idleDays: readIdleDays(b.idleDays),
264
267
  generated: b.generated === true,
265
268
  ...(typeof b.runPageBase === "string" && b.runPageBase.length > 0 ? { runPageBase: b.runPageBase } : {}),
269
+ stopped: b.stopped === true,
266
270
  repo: b.repo,
267
271
  base: b.base,
268
272
  caps: { maxRounds: b.caps.maxRounds, maxMinutes: b.caps.maxMinutes },
@@ -288,6 +292,9 @@ function spawnReturn(step: string, a: BotAnswer): StepReturn {
288
292
  return { type: "spawn", step, outcome: alreadySpawned === true ? "alreadySpawned" : "spawned", runId, at };
289
293
  if (a.status === 409 && error === "busy")
290
294
  return { type: "spawn", step, outcome: "busy", ...(typeof runId === "string" ? { runId } : {}), at };
295
+ // The hosted parent's hard stop (record 0060; issue 1924): the spawn is
296
+ // refused over the instance row's stop mark, and the unit ends stopped.
297
+ if (a.status === 409 && error === "stopped") return { type: "spawn", step, outcome: "stopped", at };
291
298
  if (a.status === 403 && typeof error === "string")
292
299
  return {
293
300
  type: "spawn",
@@ -306,7 +313,14 @@ function readRecordReturn(step: string, a: BotAnswer): StepReturn {
306
313
  const run = a.body.run;
307
314
  if (a.body.ok !== true || !isRecord(run) || typeof run.finished !== "boolean")
308
315
  throw new UnreadableAnswer("read-record", a, "run");
309
- if (!run.finished) return { type: "read-record", step, run: { finished: false }, at: a.body.at };
316
+ // The hard stop's mark, as the bot's answer carries it (record 0060; issue
317
+ // 1924): a finished child's unit ends stopped on it.
318
+ const stopped = a.body.stopped === true ? { stopped: true as const } : {};
319
+ // The interrupted child restarted from its request (issue 1903): the bot
320
+ // answers the live successor's id, and the machine keeps the wait on it.
321
+ const restarted = typeof a.body.restartedAs === "string" ? { restartedAs: a.body.restartedAs } : {};
322
+ if (!run.finished)
323
+ return { type: "read-record", step, run: { finished: false }, ...stopped, ...restarted, at: a.body.at };
310
324
  if (typeof run.status !== "string") throw new UnreadableAnswer("read-record", a, "status");
311
325
  // The typed artifacts as the bot's record carries them — shape-checked where
312
326
  // they were written (the run record's validator), read here as they are.
@@ -326,10 +340,13 @@ function readRecordReturn(step: string, a: BotAnswer): StepReturn {
326
340
  leaseStartedAt,
327
341
  costUsd,
328
342
  handoffLists,
343
+ failure,
344
+ interruption,
329
345
  } = facts;
330
346
  return {
331
347
  type: "read-record",
332
348
  step,
349
+ ...stopped,
333
350
  run: {
334
351
  finished: true,
335
352
  status: run.status as Extract<ChildFacts, { finished: true }>["status"],
@@ -348,6 +365,14 @@ function readRecordReturn(step: string, a: BotAnswer): StepReturn {
348
365
  ...(typeof leaseStartedAt === "number" ? { leaseStartedAt } : {}),
349
366
  ...(typeof costUsd === "number" || costUsd === null ? { costUsd } : {}),
350
367
  ...(handoffLists !== undefined ? { handoffLists } : {}),
368
+ // The failure by name (run-history item 57), shape-checked: a
369
+ // `provider_transient` drives the round-0 re-run (agent-ship item 9).
370
+ ...(isRecord(failure) && typeof failure.kind === "string" ? { failure: { kind: failure.kind } } : {}),
371
+ // What ended an interrupted child (issue 1876): the ending's sentence
372
+ // names the cause instead of claiming a bot restart for every one.
373
+ ...(interruption === "bot_restart" || interruption === "container_replaced" || interruption === "sandbox_fault"
374
+ ? { interruption }
375
+ : {}),
351
376
  },
352
377
  at: a.body.at,
353
378
  };
@@ -453,7 +478,11 @@ function mergeReturn(step: string, a: BotAnswer): StepReturn {
453
478
  return { type: "merge", step, outcome: "merged", by: "other", sha, mergedAt, at };
454
479
  if (ok === true && outcome === "merged" && typeof sha === "string")
455
480
  return { type: "merge", step, outcome: "merged", sha, at };
456
- if (ok === true && (outcome === "pending" || outcome === "refused") && typeof reason === "string")
481
+ if (
482
+ ok === true &&
483
+ (outcome === "pending" || outcome === "refused" || outcome === "enqueued" || outcome === "removed") &&
484
+ typeof reason === "string"
485
+ )
457
486
  return { type: "merge", step, outcome, reason, at };
458
487
  throw new UnreadableAnswer("merge", a, "outcome");
459
488
  }
@@ -642,7 +671,12 @@ async function perform(
642
671
  answerOf(
643
672
  "merge",
644
673
  await step.do(action.step, STEP_CONFIG, () =>
645
- call(bot, "merge", { ...tag, prNumber: action.prNumber, headSha: action.headSha }),
674
+ call(bot, "merge", {
675
+ ...tag,
676
+ prNumber: action.prNumber,
677
+ headSha: action.headSha,
678
+ ...(action.queued === true ? { queued: true } : {}),
679
+ }),
646
680
  ),
647
681
  ),
648
682
  );
@@ -807,6 +841,11 @@ async function runUnit(
807
841
  }
808
842
  }
809
843
 
844
+ /** The report of a unit the hard stop ended before it ran (record 0060; issue 1924). */
845
+ function stoppedReport(unit: string): string {
846
+ return `⏹ Stopped: the pipeline's hosted parent was hard-stopped, so ${unit} was ended without running. Re-issue the plan naming the remaining units to run them.`;
847
+ }
848
+
810
849
  function blockedReport(unit: string, dep: string, depEnding: string): string {
811
850
  if (depEnding === "blocked")
812
851
  return `⛔ Blocked: ${unit} waits on ${dep}, which is blocked itself. Re-issue the plan naming the remaining units once it is resolved.`;
@@ -877,6 +916,18 @@ async function walk(step: StepRunner, bot: CoordinatorBot, instanceId: string):
877
916
  let cursor = openPlanCursor(graph);
878
917
  const endings: Record<string, string> = {};
879
918
  for (;;) {
919
+ // The hard stop's mark (record 0060; issue 1924), read before every unit
920
+ // start: the walk ends every unit not yet ended `stopped` — the rows say
921
+ // why they never ran — and starts nothing more. A stop that lands once
922
+ // every unit has ended changes nothing: there is nothing left to end.
923
+ if (plan.stopped) {
924
+ for (const id of cursor.order.filter((u) => endings[u] === undefined)) {
925
+ endings[id] = "stopped";
926
+ const body = { parentInstanceId: instanceId, unit: id, ending: { kind: "stopped", report: stoppedReport(id) } };
927
+ await step.do(`${id}/end`, STEP_CONFIG, () => call(bot, "unit-end", body));
928
+ }
929
+ break;
930
+ }
880
931
  const [next] = readyUnits(graph, cursor);
881
932
  if (next === undefined) break;
882
933
  cursor = startUnit(graph, cursor, next);
@@ -914,8 +965,9 @@ async function walk(step: StepRunner, bot: CoordinatorBot, instanceId: string):
914
965
  // Blocked units, in the plan's order: each told its own ending, so the rows
915
966
  // and the summary say why it never ran. Every blocked unit's ending is known
916
967
  // before any report is rendered — a plan may list a dependent before the
917
- // dependency that blocks it.
918
- const blocked = cursor.order.filter((id) => cursor.status[id] === "blocked");
968
+ // dependency that blocks it. A stopped walk skips this: every unended unit
969
+ // was already ended `stopped` above, and its cursor never finishes.
970
+ const blocked = plan.stopped ? [] : cursor.order.filter((id) => cursor.status[id] === "blocked");
919
971
  for (const id of blocked) endings[id] = "blocked";
920
972
  for (const id of blocked) {
921
973
  const node = graph.units.find((u) => u.id === id)!;
@@ -927,7 +979,8 @@ async function walk(step: StepRunner, bot: CoordinatorBot, instanceId: string):
927
979
  };
928
980
  await step.do(`${id}/end`, STEP_CONFIG, () => call(bot, "unit-end", body));
929
981
  }
930
- if (!cursorFinished(cursor)) throw new Error(`the plan's cursor did not finish: ${JSON.stringify(cursor.status)}`);
982
+ if (!plan.stopped && !cursorFinished(cursor))
983
+ throw new Error(`the plan's cursor did not finish: ${JSON.stringify(cursor.status)}`);
931
984
  return {
932
985
  instance: instanceId,
933
986
  ...(plan.planId !== undefined ? { planId: plan.planId } : {}),
@@ -762,12 +762,19 @@ export interface BillerTieOutDay {
762
762
  invoiceUsd?: number;
763
763
  invoiceFeeUsd?: number;
764
764
  invoiceByokUsd?: number;
765
- /** The summed `model.turn` dollars whose ref names this block. */
765
+ /** The summed `model.turn` dollars whose ref names this block — the meter's
766
+ * charged figure from each turn's own `usd` (the provider's, the operator's
767
+ * or the registry's price at the time), never the price table's repricing
768
+ * the dimension views show (`buildCostsByReport`). */
766
769
  attributedUsd: number;
767
770
  /** The summed BYOK fees — the half an aggregator's `usage` ties against. */
768
771
  attributedFeeUsd?: number;
769
772
  /** `attributedUsd − attributedFeeUsd` — the half `byok_usage_inference` ties against. */
770
773
  attributedUpstreamUsd?: number;
774
+ /** The day's tokens whose turns carried no `usd` (null or absent) — spend the
775
+ * attributed side could not count, so a gap against the invoice is explained
776
+ * rather than silent. Absent when every turn was priced. */
777
+ unpricedTokens?: number;
771
778
  }
772
779
 
773
780
  /** One biller's daily tie-out over the range. */
@@ -784,15 +791,20 @@ const billerOfRef = (ref: string): string | undefined => (ref.includes("/") ? pa
784
791
 
785
792
  /**
786
793
  * Pure: each biller's invoice compared with the summed `model.turn` rows whose
787
- * ref names that block, per UTC day of the range. An aggregator's invoice
788
- * halves (`feeUsd`, `byokUsd`) sit beside the summed fees and the remainder. A
789
- * model whose `usd` is null or absent (unpriced, or from before the meter row)
790
- * contributes nothing to `attributedUsd` — an understated side is honest, an
791
- * invented one is not, so a day the source returned no row for keeps its
794
+ * ref names that block, per UTC day of the range. The attributed side is the
795
+ * meter's charged figure — each turn's own `usd`, never the price table's
796
+ * repricing. An aggregator's invoice halves (`feeUsd`, `byokUsd`) sit beside
797
+ * the summed fees and the remainder. A model whose `usd` is null or absent
798
+ * (unpriced, or from before the meter row) contributes nothing to
799
+ * `attributedUsd` — its tokens are counted as the day's `unpricedTokens`
800
+ * instead, so the understated side is visible: honest, never invented — and
801
+ * a day the source returned no row for keeps its
792
802
  * invoice side absent (the page renders —): the source holds closed days only,
793
803
  * and a trailing uninvoiced day beside real spend is not a $0 invoice. A day
794
- * with neither an invoice row nor a nonzero attributed figure is dropped; a
795
- * biller without a source (`invoice: false`) ties out against nothing.
804
+ * with no invoice row, no nonzero attributed figure and no unpriced tokens is
805
+ * dropped — but unpriced spend alone keeps its day, so the token count renders
806
+ * instead of vanishing; a biller without a source (`invoice: false`) ties out
807
+ * against nothing.
796
808
  */
797
809
  export function buildBillerTieOuts(
798
810
  billers: ReadonlyArray<{ name: string; invoice: boolean }>,
@@ -801,16 +813,22 @@ export function buildBillerTieOuts(
801
813
  range: DateRange,
802
814
  ): BillerTieOut[] {
803
815
  return billers.map(({ name, invoice }) => {
804
- const attributed = new Map<string, { usd: number; fee: number; hasFee: boolean }>();
816
+ const attributed = new Map<string, { usd: number; fee: number; hasFee: boolean; unpricedTokens: number }>();
805
817
  for (const cell of cells) {
806
818
  if (cell.day < range.from || cell.day > range.to) continue;
807
819
  for (const [ref, m] of Object.entries(cell.usage.byModel)) {
808
820
  if (billerOfRef(ref) !== name) continue;
809
- const day = attributed.get(cell.day) ?? { usd: 0, fee: 0, hasFee: false };
810
- if (typeof m.usd === "number") day.usd += m.usd;
811
- if (m.feeUsd !== undefined) {
812
- day.fee += m.feeUsd;
813
- day.hasFee = true;
821
+ const day = attributed.get(cell.day) ?? { usd: 0, fee: 0, hasFee: false, unpricedTokens: 0 };
822
+ // An unpriced model's feeUsd is skipped with its usd: adding the fee
823
+ // alone would drive attributedUpstreamUsd (usd − fee) negative.
824
+ if (typeof m.usd === "number") {
825
+ day.usd += m.usd;
826
+ if (m.feeUsd !== undefined) {
827
+ day.fee += m.feeUsd;
828
+ day.hasFee = true;
829
+ }
830
+ } else {
831
+ day.unpricedTokens += m.inputTokens + m.outputTokens + m.cacheReadTokens + m.cacheWriteTokens;
814
832
  }
815
833
  attributed.set(cell.day, day);
816
834
  }
@@ -820,7 +838,7 @@ export function buildBillerTieOuts(
820
838
  for (const d of invoices[name] ?? []) if (d.date >= range.from && d.date <= range.to) invoiced.set(d.date, d);
821
839
  const dates = [...new Set([...attributed.keys(), ...invoiced.keys()])].sort().filter((date) => {
822
840
  const a = attributed.get(date);
823
- return invoiced.has(date) || (a !== undefined && (a.usd !== 0 || a.hasFee));
841
+ return invoiced.has(date) || (a !== undefined && (a.usd !== 0 || a.hasFee || a.unpricedTokens > 0));
824
842
  });
825
843
  const days: BillerTieOutDay[] = dates.map((date) => {
826
844
  const a = attributed.get(date);
@@ -832,6 +850,7 @@ export function buildBillerTieOuts(
832
850
  ...(inv?.byokUsd !== undefined ? { invoiceByokUsd: inv.byokUsd } : {}),
833
851
  attributedUsd: a?.usd ?? 0,
834
852
  ...(a?.hasFee ? { attributedFeeUsd: a.fee, attributedUpstreamUsd: a.usd - a.fee } : {}),
853
+ ...(a !== undefined && a.unpricedTokens > 0 ? { unpricedTokens: a.unpricedTokens } : {}),
835
854
  };
836
855
  });
837
856
  return {
@@ -922,7 +941,11 @@ export class OpenAICostsSource implements LlmInvoiceSource {
922
941
  next_page?: string | null;
923
942
  };
924
943
  for (const bucket of body.data ?? []) {
925
- const date = new Date(num(bucket.start_time) * 1000).toISOString().slice(0, 10);
944
+ // A bucket without a finite start_time is skipped, never dated to epoch 0;
945
+ // out-of-range buckets are dropped like the OpenRouter source's rows.
946
+ if (typeof bucket.start_time !== "number" || !Number.isFinite(bucket.start_time)) continue;
947
+ const date = new Date(bucket.start_time * 1000).toISOString().slice(0, 10);
948
+ if (date < range.from || date > range.to) continue;
926
949
  for (const r of bucket.results ?? []) {
927
950
  const currency = str(r.amount?.currency).toLowerCase();
928
951
  if (currency !== "usd") throw new Error(`openai organization costs: unexpected currency ${currency || "?"}`);