@coreplane/switchboard 1.242.0 → 1.243.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. package/dist/assets/config/config.example.yaml +20 -1
  2. package/dist/assets/deploy/cloudflare-memory/worker.ts +86 -29
  3. package/dist/assets/deploy/cloudflare-resident/worker.ts +105 -13
  4. package/dist/assets/package-lock.json +3 -3
  5. package/dist/assets/package.json +1 -1
  6. package/dist/assets/project.json +3 -3
  7. package/dist/assets/source.json +3 -3
  8. package/dist/assets/src/agents/registry.ts +30 -34
  9. package/dist/assets/src/config/profile.ts +68 -3
  10. package/dist/assets/src/core/authz/policy.ts +4 -0
  11. package/dist/assets/src/core/budgets.ts +313 -0
  12. package/dist/assets/src/core/coordinator/driver.ts +0 -10
  13. package/dist/assets/src/core/costs.ts +9 -76
  14. package/dist/assets/src/core/modelPricing.ts +212 -0
  15. package/dist/assets/src/core/prDescriptionTypes.ts +29 -24
  16. package/dist/assets/src/core/reviewVerdict.ts +7 -0
  17. package/dist/assets/src/core/runEvents.ts +28 -4
  18. package/dist/assets/src/core/runFriction.ts +2 -1
  19. package/dist/assets/src/core/runRecord.ts +47 -0
  20. package/dist/assets/src/core/runUsage.ts +158 -47
  21. package/dist/assets/src/core/schedules.ts +3 -0
  22. package/dist/assets/src/core/ship/coordinator.ts +101 -84
  23. package/dist/assets/src/core/ship/handoff.ts +9 -0
  24. package/dist/assets/src/execution/bashTimeout.ts +8 -5
  25. package/dist/assets/src/execution/residentRefresh.ts +28 -3
  26. package/dist/assets/src/execution/residentSteps.ts +1 -1
  27. package/dist/assets/web/dist/.vite/manifest.json +67 -64
  28. package/dist/assets/web/dist/assets/CostsPage-CwXOmkeQ.js +2 -0
  29. package/dist/assets/web/dist/assets/HomePage-PRxjQiGG.js +2 -0
  30. package/dist/assets/web/dist/assets/PendingTurnRow-BT9RhFZ7.js +1 -0
  31. package/dist/assets/web/dist/assets/{ResidentDetailPage-DTMBgnIW.js → ResidentDetailPage-DACalNVF.js} +1 -1
  32. package/dist/assets/web/dist/assets/ResidentsIndexPage-SHPdu6uZ.js +1 -0
  33. package/dist/assets/web/dist/assets/RunFoldRow-0SdOmOr5.js +1 -0
  34. package/dist/assets/web/dist/assets/RunRoutePage-BaFS2p8I.js +9 -0
  35. package/dist/assets/web/dist/assets/RunsIndexPage-DYI-iALj.js +1 -0
  36. package/dist/assets/web/dist/assets/ScheduledPage-DJ8HiCPt.js +1 -0
  37. package/dist/assets/web/dist/assets/{SettingsPage-BIGio8Y0.js → SettingsPage-DLiN5IgY.js} +1 -1
  38. package/dist/assets/web/dist/assets/{StatusDot-ELoXHlFt.js → StatusDot-Dw0T1M-P.js} +1 -1
  39. package/dist/assets/web/dist/assets/{Tooltip-BoeFwYP2.js → Tooltip-BbLuIAiS.js} +1 -1
  40. package/dist/assets/web/dist/assets/UnitRoutePage-DicUG96U.js +1 -0
  41. package/dist/assets/web/dist/assets/{dist-BU5UivXC.js → dist-twkFmSUY.js} +1 -1
  42. package/dist/assets/web/dist/assets/format-BldUwl_R.js +1 -0
  43. package/dist/assets/web/dist/assets/indexRow-B_s5tKyq.js +1 -0
  44. package/dist/assets/web/dist/assets/{main-B2fX10aW.css → main-B4kEF3Sg.css} +1 -1
  45. package/dist/assets/web/dist/assets/{main-D4EA1g6n.js → main-d-w-tIKt.js} +2 -2
  46. package/dist/assets/web/dist/assets/{sseReplay-g7ml86LM.js → sseReplay-C9m_EB8J.js} +4 -4
  47. package/dist/cli.js +2758 -1445
  48. package/package.json +1 -1
  49. package/dist/assets/web/dist/assets/CostsPage-CtmKhhOF.js +0 -2
  50. package/dist/assets/web/dist/assets/HomePage-QtH1EwYF.js +0 -2
  51. package/dist/assets/web/dist/assets/PendingTurnRow-BZA_vQt3.js +0 -1
  52. package/dist/assets/web/dist/assets/ResidentsIndexPage-CaDvXhzJ.js +0 -1
  53. package/dist/assets/web/dist/assets/RunFoldRow-mgyLW0oV.js +0 -1
  54. package/dist/assets/web/dist/assets/RunRoutePage-DypJYMQa.js +0 -6
  55. package/dist/assets/web/dist/assets/RunsIndexPage-DYraPoWD.js +0 -1
  56. package/dist/assets/web/dist/assets/ScheduledPage-DXD2gLJk.js +0 -1
  57. package/dist/assets/web/dist/assets/UnitRoutePage-jhCrWW3i.js +0 -1
  58. package/dist/assets/web/dist/assets/indexRow-BD1VT8o8.js +0 -1
  59. package/dist/assets/web/dist/assets/localIso-L06jV29p.js +0 -1
@@ -1,6 +1,7 @@
1
1
  // Agent definitions. An agent is a system prompt + toolset + machine class + wall-clock budget.
2
2
  import type { Effort } from "../effort.js";
3
3
  import { BASH_TIMEOUT_MAX_MS } from "../execution/bashTimeout.js";
4
+ import { ASKS, RUNAWAY_TURNS_PER_MINUTE, runawayTurnCap, type LoopPreset } from "../core/budgets.js";
4
5
  import { CONTRACT_HEADING, CONTRACT_SECTION_HEADINGS, PR_TITLE_GUARD } from "../core/ship/contract.js";
5
6
  // Which model runs it is resolved separately by the config layers, so any
6
7
  // agent can run on any configured provider/model.
@@ -41,24 +42,16 @@ export function machineNeedsRepo(machine: MachineClass): boolean {
41
42
  export const IDENTITIES = ["none", "read", "write"] as const;
42
43
  export type Identity = (typeof IDENTITIES)[number];
43
44
 
44
- /** The pace that marks a run as looping rather than working: a model turn
45
- * every ten seconds, sustained for the whole wall clock. A busy run takes
46
- * 20–40 s a turn (a model think plus a tool call), so a run that averages six
47
- * a minute from start to end is re-issuing calls, not making progress — and
48
- * its turn cap ends it before the wall clock would, with a write-up that
49
- * says so (docs/reference/specs/harness-pi.md item 15). */
50
- export const RUNAWAY_TURNS_PER_MINUTE = 6;
51
-
52
- /** The turn cap a wall clock implies: `maxMinutes × RUNAWAY_TURNS_PER_MINUTE`.
53
- * Every preset that runs the loop derives its `maxTurns` from this, so the
54
- * cap is never a number a good run reaches — the minutes are the budget. */
55
- export function runawayTurnCap(maxMinutes: number): number {
56
- return maxMinutes * RUNAWAY_TURNS_PER_MINUTE;
57
- }
45
+ /** The wall clocks live in `src/core/budgets.ts` (docs/decisions/0046): a
46
+ * preset's ask, the turn cap derived from it and every allowance are rows
47
+ * there, and this registry reads them. Re-exported for the readers that
48
+ * learned them here. */
49
+ export { RUNAWAY_TURNS_PER_MINUTE, runawayTurnCap };
58
50
 
59
- /** A loop-running preset's budget as one fact: the wall clock, and the runaway
60
- * guard derived from it. */
61
- function loopBudget(maxMinutes: number): Pick<AgentDef, "maxMinutes" | "maxTurns"> {
51
+ /** A loop-running preset's budget as one fact read from the module: the wall
52
+ * clock it asks for, and the runaway guard derived from it. */
53
+ function loopBudget(preset: LoopPreset): Pick<AgentDef, "maxMinutes" | "maxTurns"> {
54
+ const maxMinutes = ASKS[preset];
62
55
  return { maxMinutes, maxTurns: runawayTurnCap(maxMinutes) };
63
56
  }
64
57
 
@@ -150,15 +143,18 @@ export function statusCardRule(examples = '"Implement the fix", "Run the test su
150
143
  );
151
144
  }
152
145
 
153
- const PR_DESCRIPTION_TEMPLATE = `PR description — submit it with the submit_pr_description tool for EVERY PR (this is the default, not something to wait to be asked for). Switchboard renders the GitHub body from the object you submit, so never author PR-body markdown yourself. Before submitting, judge your title with the ${PR_TITLE_GUARD} gate — \`npm run check:pr-title -- "<title>"\` — and submit only a title it accepts; the same gate refuses the PR in CI. Content contract per field (each renders as its own section): prose is unwrapped — no hard line breaks inside a paragraph. Always hyperlink the triggering issue/request. Never fabricate validation — state exactly what you ran and the real result. Keep each field concise, not padded.
154
- EVERY PR includes one that already exists when you push — opened by a person, by dependabot, or by an earlier run. After EVERY push to such a PR: read its current title and body (\`github_issue_get\` with the PR number works for pull requests; \`gh pr view\` where gh exists), judge them against the change as it now stands at the pushed head, and submit the object that describes the PR as it is NOW — carry forward what the existing body says that is still true (a dependency bump's release notes belong in whatWhy), add what you changed, and anchor the Tour at the new head. Switchboard replaces the PR's title and body with your rendering. A description that describes an earlier state of its branch is a bug; "it is someone else's PR" is never a reason to leave it.
146
+ const PR_DESCRIPTION_TEMPLATE = `PR description — submit it with the submit_pr_description tool for EVERY PR (this is the default, not something to wait to be asked for). Switchboard renders the GitHub body from the object you submit, so never author PR-body markdown yourself. Before submitting, judge your title with the ${PR_TITLE_GUARD} gate — \`npm run check:pr-title -- "<title>"\` — and submit only a title it accepts; the same gate refuses the PR in CI. The body is a fixed-size MAP for the reader with everything for agents collapsed under it; every field is capped in visible characters (a link's URL is not counted) and the tool refuses an object over a cap naming the field and the count — cut and resubmit. Prose is unwrapped — no hard line breaks inside a paragraph. Always hyperlink the triggering issue/request. Never fabricate validation — state exactly what you ran and the real result. BEFORE authoring the pointers, load the \`pr-description\` skill with use_skill — it defines how to choose at most seven pointers, the mechanical anchor rules and what goes below the fold; follow it for every PR.
147
+ EVERY PR includes one that already exists when you push — opened by a person, by dependabot, or by an earlier run. After EVERY push to such a PR: read its current title and body (\`github_issue_get\` with the PR number works for pull requests; \`gh pr view\` where gh exists), judge them against the change as it now stands at the pushed head, and submit the object that describes the PR as it is NOW — carry forward what the existing body says that is still true (a dependency bump's release notes belong in why), add what you changed, and anchor the pointers at the new head. Switchboard replaces the PR's title and body with your rendering. A description that describes an earlier state of its branch is a bug; "it is someone else's PR" is never a reason to leave it.
155
148
  - **title**: the PR title — one line naming the change, specific enough to pick out of a PR list.
156
- - **TL;DR** (\`tldr\`, rendered first): two sentences for a naive reader with zero context — what this PR does and why it matters.
157
- - **What & why** (\`whatWhy\`): the change and its motivation, linked to the triggering issue/request.
158
- - **Tour** (\`tour\` + \`remaining\`): the guided walkthrough of the change, replacing any prose list of changes — ordered steps of { title, description, optional lookFor, anchor }, each anchor a { path, from, to } line range at your pushed head; every touched file no step covers goes in \`remaining\` as { path, note }. BEFORE authoring the Tour steps, load the \`pr-tour\` skill with use_skill — it defines the reader-first step shape, the anchor rules, and the Remaining-changes catch-all. Follow it for every PR; if a later push changes what the steps point at, resubmit the description with corrected anchors.
159
- - **Decisions** (\`decisions\`): non-obvious choices as { title, rationale } — alternatives considered and rejected, trade-offs.
160
- - **Risks & implications** (\`risks\`): what could break, the blast radius, and any migration/rollout/compatibility concerns (or "none" — and why).
161
- - **Validation** (\`validation\`): what you tested and the actual results as { criterion, proof } rows (commands run, pass/fail), plus how the reviewer can verify it themselves; the optional summary line carries the overall result.`;
149
+ - **TL;DR** (\`tldr\`, rendered first, ≤300): two sentences for a reader with zero context — what this PR does and why it matters.
150
+ - **Why** (\`why\`, ≤400): the problem and the motivation, with the triggering issue/request, the record and the stack position hyperlinked. Why, never what: the diff shows what.
151
+ - **Where to look** (\`pointers\`, 1 to 7): the files a reviewer would open first, in reading order, each { label ≤60, text ≤160, optional risk ≤100, anchor } with the anchor a { path, from, to } line range at your pushed head, rendered as a link (never embedded code). One pointer per idea, never per file; when the change has more ideas than seven, keep the seven whose mistake would cost most.
152
+ - **Feedback wanted** (\`feedbackWanted\`, ≤200): the one or two things you want the reviewer's judgement on.
153
+ - **Risk** (\`risk\`, ≤300): what breaks if this is wrong, the blast radius, the rollback; over 400 changed lines, say so and name the split you considered.
154
+ - **Verified** (\`verified\`, ≤200): one line for a person — which suites ran and passed, what is still human-gated.
155
+ - **Decisions** (\`decisions\`, 0 to 10, collapsed): non-obvious choices as { title, rationale ≤400 } — the alternative rejected and the fact that decided it.
156
+ - **Validation** (\`validation\`, 1 to 30 criteria, collapsed): what you tested and the actual results as { criterion ≤200, proof ≤300 } rows (test ids, commands run, pass/fail), plus how the reviewer can verify it.
157
+ - **For agents** (\`agentNotes\`, optional, ≤2000, collapsed): what a reviewing agent needs that a person does not — the rebase you did, generated files to skip, the command that reproduces the bug.`;
162
158
 
163
159
  // Both coding prompts carry this verbatim: coding runs hold a write-scoped
164
160
  // token where a merge is one command away, so the boundary is spelled out the
@@ -521,7 +517,7 @@ Your tools work without a workspace: the GitHub tools — \`github_repos\` (the
521
517
 
522
518
  ${statusCardRule('"Read the issue and its thread", "Post the comment"')} A one-step answer needs no checklist; post one when the request has steps the person would wait on.
523
519
 
524
- You cannot run commands, clone repositories, edit code, or review pull requests, and you cannot search the web. Other Switchboard agents can: for code changes or PRs tell the user to re-send with \`agent:coding\`; for a PR review, \`agent:review\`; for a web-research question, \`agent:research\` (e.g. "\`agent:coding fix the failing login test in acme/api\`", "\`agent:research compare X and Y\`"). Delete an issue only when the user explicitly asked to delete it (closing is an update).`;
520
+ You cannot run commands, clone repositories, edit code, or review pull requests, and you cannot search the web. Other Switchboard agents can: for a code change or a pull request tell the user to re-send with \`agent:ship\` (it makes the change, opens the PR and loops review); for a PR review, \`agent:review\`; for a web-research question, \`agent:research\` (e.g. "\`agent:ship in acme/api: fix the failing login test\`", "\`agent:research compare X and Y\`"). Delete an issue only when the user explicitly asked to delete it (closing is an update).`;
525
521
 
526
522
  // The explore agent (docs/reference/specs/agent-explore.md): a long, read-only
527
523
  // investigation — "run our CI locally and validate the claims", "how long does
@@ -544,7 +540,7 @@ THE DELIVERABLE IS A CLAIM TABLE. Turn the request into the claims it makes or a
544
540
 
545
541
  TIME. Your budget is up to two hours — less when a boundary or the request's \`budget:\` directive clipped it, which the runtime-config block above says — and the wrap-up warning tells you when to stop starting new checks. A single command is capped at ${BASH_TIMEOUT_MAX_MS / 60_000} minutes (pass the bash tool's \`timeoutMs\`, up to ${BASH_TIMEOUT_MAX_MS} ms, for a long one). A job that needs longer — a full suite, a build, a pipeline run — is started detached and polled across tool calls: \`setsid -f sh -c '<command> > /tmp/job.log 2>&1; echo $? > /tmp/job.exit'\`, then \`tail -n 40 /tmp/job.log\` and \`cat /tmp/job.exit\` on later calls (a plain background job dies with the command that started it; a \`setsid -f\` job outlives it). Batch commands into few tool calls; never explore file by file.
546
542
 
547
- READ-ONLY: NEVER open a pull request, and never commit or push — no branch, no \`gh pr create\`, no PR or issue write of any kind. You hold a read credential and your job is to find out, not to change. If the investigation shows a change is needed, say exactly what and where in your write-up and point the user at \`agent:coding\`.
543
+ READ-ONLY: NEVER open a pull request, and never commit or push — no branch, no \`gh pr create\`, no PR or issue write of any kind. You hold a read credential and your job is to find out, not to change. If the investigation shows a change is needed, say exactly what and where in your write-up and point the user at \`agent:ship\` (it makes the change, opens the PR and loops review).
548
544
 
549
545
  You cannot attach or post files: your whole answer is text. Never say a file is attached or below — name its path in the workspace and describe it (what it shows, its size) instead; a person who needs the file itself asks \`agent:coding\`, which can attach.
550
546
 
@@ -638,7 +634,7 @@ const WORK_PRESETS = {
638
634
  machine: "none",
639
635
  identity: "none",
640
636
  maxTokens: 16000,
641
- ...loopBudget(5),
637
+ ...loopBudget("general"),
642
638
  },
643
639
  coding: {
644
640
  name: "coding",
@@ -648,7 +644,7 @@ const WORK_PRESETS = {
648
644
  seededSystem: CODING_SYSTEM_SEEDED,
649
645
  toolset: "full",
650
646
  maxTokens: 64000,
651
- ...loopBudget(45),
647
+ ...loopBudget("coding"),
652
648
  // No built-in effort: the deployment decides (`defaults.efforts.coding`,
653
649
  // `config set channel efforts.coding=…`, or `effort:` per request).
654
650
  machine: "repo-resident",
@@ -669,7 +665,7 @@ const WORK_PRESETS = {
669
665
  machine: "repo-resident",
670
666
  identity: "read", // a read-scoped token and a read-only worktree: it cannot post or push from inside
671
667
  maxTokens: 64000,
672
- ...loopBudget(25), // a safety net — typical reviews land in ~5 minutes
668
+ ...loopBudget("review"), // a safety net — typical reviews land in ~5 minutes
673
669
  effort: "medium", // fast turns; one big-context pass does the deep work
674
670
  },
675
671
  ship: {
@@ -698,7 +694,7 @@ const WORK_PRESETS = {
698
694
  // reviewed pull request, never code landing on main.
699
695
  maxTurns: 1,
700
696
  maxTokens: 16000,
701
- maxMinutes: 120,
697
+ maxMinutes: ASKS.ship,
702
698
  },
703
699
  research: {
704
700
  name: "research",
@@ -709,7 +705,7 @@ const WORK_PRESETS = {
709
705
  machine: "none", // web I/O only; no workspace is provisioned
710
706
  identity: "none",
711
707
  maxTokens: 24000,
712
- ...loopBudget(8),
708
+ ...loopBudget("research"),
713
709
  effort: "medium",
714
710
  },
715
711
  explore: {
@@ -723,7 +719,7 @@ const WORK_PRESETS = {
723
719
  machine: "repo-cold",
724
720
  identity: "read", // a read-scoped token: it can clone and read, never push — whatever the caller holds
725
721
  maxTokens: 64000,
726
- ...loopBudget(120),
722
+ ...loopBudget("explore"),
727
723
  // No built-in effort: the deployment decides, as for coding.
728
724
  },
729
725
  } satisfies Record<string, AgentDef>;
@@ -748,7 +744,7 @@ export const AGENTS: Record<string, AgentDef> = {
748
744
  // spawn exactly those — so no child runs that the record did not name.
749
745
  routable: false,
750
746
  maxTokens: 32000,
751
- ...loopBudget(120), // long enough to outlast a coding child; every child is capped by what remains of it
747
+ ...loopBudget("conductor"), // long enough to outlast a coding child; every child is capped by what remains of it
752
748
  // No built-in effort: the deployment decides, as for coding.
753
749
  },
754
750
  };
@@ -73,10 +73,40 @@ export function budgetedAgent(agent: AgentDef, profile: RunProfile): AgentDef {
73
73
 
74
74
  // ---- boundaries: a scope caps, never grants -----------------------------------
75
75
 
76
+ /** The blast-radius classes a scope may name as its `confirm`
77
+ * (docs/decisions/0044-a-routed-write-is-confirmed-in-proportion-to-its-blast-radius.md):
78
+ * the first class on the ladder the door hands back instead of running.
79
+ * `exec` is on the ladder for the comparison but not settable — a test or
80
+ * build never asks — and `never` is refused until the door's write misbind
81
+ * rate has been measured; the validator names both reasons. */
82
+ export type ConfirmClass = "write" | "destructive";
83
+ export const CONFIRM_CLASSES: readonly ConfirmClass[] = ["write", "destructive"];
84
+
85
+ /** The classes on the door's ladder — the command registry's `BlastRadius`,
86
+ * spelled here rather than imported: a type import still drags the registry's
87
+ * whole graph into every program that compiles this near-leaf module, the
88
+ * Workers included. The door indexes the ladder with the registry's type
89
+ * (`routedRunsAtOnce`), so a class added there without a rung here fails to
90
+ * compile at the one place the two vocabularies meet. */
91
+ export type ConfirmLadderClass = "read" | "exec" | "write" | "destructive";
92
+
93
+ /** The ladder the door compares on, `read < exec < write < destructive`: among
94
+ * the last three, from asking most to asking least — a `confirm` of `write`
95
+ * hands back every write and every destructive write, `destructive` only the
96
+ * destructive ones. A read is on the order so the comparison is total, but the
97
+ * door never asks for one. */
98
+ export const CONFIRM_ORDER: Record<ConfirmLadderClass, number> = { read: 0, exec: 1, write: 2, destructive: 3 };
99
+
100
+ /** The door's confirm class when no scope sets one: every routed write is
101
+ * handed back, a read or a test run at once — the door as it was before the
102
+ * axis existed. Attributed to `built-in`, a word that appears in no config. */
103
+ export const BUILT_IN_CONFIRM: ConfirmClass = "write";
104
+
76
105
  /** A cap on the three axes that any scope may set (`defaults`, `channels.<id>`,
77
- * `users.<id>`; docs/reference/specs/routing-and-config.md item 2). An absent
78
- * axis caps nothing. A boundary never grants: it is not a fourth grants axis,
79
- * and the policy table's one question (who may run a preset) is unchanged. */
106
+ * `users.<id>`; docs/reference/specs/routing-and-config.md item 2), and the
107
+ * door's `confirm` beside them. An absent axis caps nothing. A boundary never
108
+ * grants: it is not a fourth grants axis, and the policy table's one question
109
+ * (who may run a preset) is unchanged. */
80
110
  export interface Boundary {
81
111
  /** The most a run may have, in minutes; at least 2 (the bash tool keeps a 60 s reserve). */
82
112
  maxMinutes?: number;
@@ -84,6 +114,11 @@ export interface Boundary {
84
114
  maxIdentity?: Identity;
85
115
  /** The machine classes a run may execute on; a preset's class must be in every layer's set. */
86
116
  machines?: MachineClass[];
117
+ /** The first blast-radius class a command the router bound is handed back
118
+ * at instead of run (record 0044). Not a run cap: it rides the boundary for
119
+ * its scopes and its intersection-toward-caution, is read by the door alone
120
+ * (`effectiveConfirm`), and never enters `intersectBoundaries` or a profile. */
121
+ confirm?: ConfirmClass;
87
122
  }
88
123
 
89
124
  /** One layer's boundary with the scope it came from, in resolution order:
@@ -143,6 +178,36 @@ export function intersectBoundaries(layers: readonly ScopedBoundary[]): Effectiv
143
178
  return out.maxMinutes || out.maxIdentity || out.machines ? out : undefined;
144
179
  }
145
180
 
181
+ /** Where the door's confirm class came from: a scope that set it, or the
182
+ * built-in default when none did. Local to the confirm axis — `BoundaryScope`
183
+ * itself is unchanged, since no run cap is ever attributed to `built-in`. */
184
+ export type ConfirmScope = BoundaryScope | "built-in";
185
+
186
+ /** The door's decision on the confirm axis: the class and the scope it names. */
187
+ export interface EffectiveConfirm {
188
+ value: ConfirmClass;
189
+ scope: ConfirmScope;
190
+ }
191
+
192
+ /**
193
+ * The confirm axis intersected over a request's path, apart from the run caps:
194
+ * the earliest class on `CONFIRM_ORDER` any layer named — the most cautious,
195
+ * since a scope's value is the most permissive answer it allows and the org's
196
+ * is therefore a floor no layer below it can loosen — attributed to the layer
197
+ * that set it (on a tie the first layer named keeps it, the least specific
198
+ * scope, as `intersectBoundaries` does). No layer set one: the built-in
199
+ * `write`, attributed to `built-in`. Pure over the same layers `intersectBoundaries` takes.
200
+ */
201
+ export function effectiveConfirm(layers: readonly ScopedBoundary[]): EffectiveConfirm {
202
+ let out: EffectiveConfirm | undefined;
203
+ for (const { scope, boundary } of layers) {
204
+ if (boundary.confirm !== undefined && (!out || CONFIRM_ORDER[boundary.confirm] < CONFIRM_ORDER[out.value])) {
205
+ out = { value: boundary.confirm, scope };
206
+ }
207
+ }
208
+ return out ?? { value: BUILT_IN_CONFIRM, scope: "built-in" };
209
+ }
210
+
146
211
  /**
147
212
  * The boundaries on a child's path with its parent's remaining wall clock as
148
213
  * one more layer (docs/reference/specs/routing-and-config.md item 20): the
@@ -82,6 +82,10 @@ export const POLICY: readonly Rule[] = [
82
82
  { action: "friction:write", resource: "command", when: [grant("friction:write")] },
83
83
 
84
84
  // ── costs ────────────────────────────────────────────────────────────────
85
+ // `costs by` reads the snapshot's arithmetic: what every browser session
86
+ // holds (every group's read) and what a Slack user or a token is granted —
87
+ // never a chat baseline, since a by-user table names who spent what.
88
+ { action: "costs:read", resource: "command", when: [grant("costs:read")] },
85
89
  // `costs snapshot` reads both billing providers and replaces what every
86
90
  // viewer of the costs page sees: the grant, never a baseline (the admins'
87
91
  // `all` and a named `grants` entry hold it).
@@ -0,0 +1,313 @@
1
+ // Every wall clock in Switchboard as one table (docs/decisions/0046; docs/reference/specs/harness-pi.md
2
+ // item 15; docs/reference/specs/agent-ship.md item 8). A budget is a LEASE a
3
+ // parent carves from its own remainder, never a constant a file holds on its
4
+ // own: a preset ASKS for a lease and does useful work above a FLOOR; a parent
5
+ // that runs a loop holds back a RESERVE, derived from the floors and the
6
+ // provisioning of every round that must still follow, plus the merge wait's
7
+ // floor; and the lease covers everything the run does, the loop, the write-up
8
+ // and the post-step, each an ALLOWANCE named here. The registry reads its
9
+ // `maxMinutes` and `maxTurns` from this module and `budgets.check.test.ts`
10
+ // asserts the fits, so a number that breaks another's assumption is a red
11
+ // build. `carve` is called by the ship coordinator for every round (agent-ship
12
+ // item 8), `fit` by the config validator at load and by the fork over a clipped
13
+ // request, and `carveChildOfParent` by the conductor's spawn.
14
+ //
15
+ // Deliberately free of node: imports and of the agent registry, so the
16
+ // Workflow-driven coordinator and the deploy Workers can bundle it — the
17
+ // dependency runs registry → budgets, never the reverse.
18
+
19
+ export const MINUTE_MS = 60_000;
20
+
21
+ /** The presets that run the tool loop, and the one pipeline preset. */
22
+ export const LOOP_PRESETS = ["general", "coding", "review", "research", "explore", "conductor"] as const;
23
+ export type LoopPreset = (typeof LOOP_PRESETS)[number];
24
+ export type Preset = LoopPreset | "ship";
25
+
26
+ /** What each preset asks for when nothing above it is tighter, in minutes. The
27
+ * pipeline's (`ship`) is the wall clock one segment of its loop runs under;
28
+ * a deployment's `ship.maxMinutes` replaces it, held to `fit` below. */
29
+ export const ASKS: Readonly<Record<Preset, number>> = {
30
+ general: 5,
31
+ coding: 45,
32
+ review: 25,
33
+ ship: 120,
34
+ research: 8,
35
+ explore: 120,
36
+ conductor: 120,
37
+ };
38
+
39
+ /** The rounds a ship loop is made of. `fix` is a coding child handed the
40
+ * review's findings; `merge` is the runner's wait on the guards. */
41
+ export type RoundKind = "coding" | "review" | "fix" | "merge";
42
+
43
+ /** The least lease in which a round does useful work, in minutes. A carve that
44
+ * falls under the floor is refused rather than dispatched: a two-minute
45
+ * review or fix costs an attach and a model turn and finishes nothing.
46
+ * Review's is the ledger's 90th percentile of completed reviews (5.1 min over
47
+ * 181); coding's and the merge wait's are guesses until the ledger says. */
48
+ export const PRESET_FLOORS: Readonly<Record<LoopPreset, number>> = {
49
+ general: 2,
50
+ coding: 10,
51
+ review: 5,
52
+ research: 3,
53
+ explore: 15,
54
+ conductor: 15,
55
+ };
56
+ export const FLOORS: Readonly<Record<RoundKind, number>> = {
57
+ coding: PRESET_FLOORS.coding,
58
+ fix: PRESET_FLOORS.coding,
59
+ review: PRESET_FLOORS.review,
60
+ merge: 10,
61
+ };
62
+
63
+ /** The merge wait's own ask: how long the runner waits on the guards at most
64
+ * when the remainder allows it. */
65
+ export const MERGE_WAIT_ASK_MINUTES = 60;
66
+
67
+ /** The ship runner's waits, in minutes: the margin a child's wait allows past
68
+ * its budget, the slice a wait is asked in, the merge door's re-ask cadence,
69
+ * and the pause before a busy spawn is asked again. */
70
+ export const SHIP_WAIT = { marginMinutes: 5, chunkMinutes: 5, mergeChunkMinutes: 5, busyRetryMinutes: 2 } as const;
71
+
72
+ /** The named amounts a lease holds back, in minutes. Each stands for a step
73
+ * every run or round pays: `provision` is attach and restore before the
74
+ * harness's clock starts; `writeUp` is the final answer after the loop ends;
75
+ * `commandWriteUp` is what the last command leaves for that answer; `execCall`
76
+ * is the exec client's wait past a command's own budget; `bearerGrace` is how
77
+ * far past the lease the model bearer stays valid for the last call's tail. */
78
+ export const ALLOWANCES = {
79
+ provision: 3,
80
+ writeUp: 3,
81
+ commandWriteUp: 1,
82
+ execCall: 0.5,
83
+ bearerGrace: 1,
84
+ } as const;
85
+
86
+ /** The post-step turn a preset runs after its loop, in minutes: the coding
87
+ * run's description turn, the review's verdict turn, none for the rest. */
88
+ export const POST_STEP_MINUTES: Readonly<Record<Preset, number>> = {
89
+ coding: 5,
90
+ review: 3,
91
+ general: 0,
92
+ research: 0,
93
+ explore: 0,
94
+ conductor: 0,
95
+ ship: 0,
96
+ };
97
+
98
+ /** The post-step allowance of a preset named by its def's `name`: a name
99
+ * outside the table (a test's, a command run's) runs no post-step. */
100
+ export function postStepMinutes(preset: string): number {
101
+ return (POST_STEP_MINUTES as Readonly<Record<string, number>>)[preset] ?? 0;
102
+ }
103
+
104
+ /** The post-step turn's lease, in minutes (fractional when the remainder is:
105
+ * the harness turns it back into ms): the preset's allowance, or the lease's
106
+ * remainder when that is less — never under a minute, so a turn the loop's
107
+ * write-up crowded still gets its one chance. The bearer's grace covers that
108
+ * floor for a turn that starts at or before the lease's end; a turn that
109
+ * starts later runs on a bearer that may expire under it, and its refused
110
+ * call fails soft — the run's answer already stands. Without a remainder
111
+ * (no harness session says) the allowance stands. */
112
+ export function postStepLease(preset: string, remainingMs: number | undefined): number {
113
+ const allowance = postStepMinutes(preset);
114
+ if (remainingMs === undefined) return allowance;
115
+ return Math.min(allowance, Math.max(1, remainingMs / MINUTE_MS));
116
+ }
117
+
118
+ /** A follow-up turn's lease, in ms: the lesser of what it asks and what the
119
+ * run's lease still holds, never under a minute — the grace covers that floor
120
+ * when the turn starts by the lease's end; later, the bearer may expire under
121
+ * the turn and its call is refused, failing soft. */
122
+ export function turnLeaseMs(askMinutes: number, remainingMs: number): number {
123
+ return Math.min(askMinutes * MINUTE_MS, Math.max(MINUTE_MS, remainingMs));
124
+ }
125
+
126
+ /** The wrap-up warning's place before the loop's end: three minutes, or a
127
+ * quarter of the loop when the loop is shorter than twelve. */
128
+ export const WRAP_UP_WARNING = { minutes: 3, fraction: 0.25 } as const;
129
+
130
+ /** The clocks a harness keeps for one lease (docs/reference/specs/harness-pi.md
131
+ * items 6 and 15). The lease ENDS at `deadline`; the LOOP ENDS at `loopEnd`,
132
+ * the write-up allowance and the preset's post-step earlier, so both run
133
+ * inside the lease; the WARNING is steered at `warnAt`; the write-up is
134
+ * BOUNDED by `finaleMs`. A follow-up turn on the session (`kind: "turn"`)
135
+ * holds nothing back — its deliverable is a tool call, not a write-up — and
136
+ * its loop ends at its deadline. A lease shorter than its hold-back has no
137
+ * loop time: the loop ends at its start, never before it. */
138
+ export interface LoopClock {
139
+ startedAt: number;
140
+ deadline: number;
141
+ loopEnd: number;
142
+ warnAt: number;
143
+ finaleMs: number;
144
+ }
145
+
146
+ export function loopClock(
147
+ startedAt: number,
148
+ remainingMs: number,
149
+ preset: string,
150
+ kind: "loop" | "turn" = "loop",
151
+ ): LoopClock {
152
+ const deadline = startedAt + remainingMs;
153
+ const holdBackMs = kind === "loop" ? (ALLOWANCES.writeUp + postStepMinutes(preset)) * MINUTE_MS : 0;
154
+ const loopEnd = Math.max(startedAt, deadline - holdBackMs);
155
+ const warnAt =
156
+ loopEnd - Math.min(WRAP_UP_WARNING.minutes * MINUTE_MS, (loopEnd - startedAt) * WRAP_UP_WARNING.fraction);
157
+ return { startedAt, deadline, loopEnd, warnAt, finaleMs: ALLOWANCES.writeUp * MINUTE_MS };
158
+ }
159
+
160
+ /** When a run's model-proxy bearer expires: the lease's end plus the grace
161
+ * (docs/reference/specs/model-proxy.md item 2). */
162
+ export function bearerExpiresAt(leaseEndsAt: number): number {
163
+ return leaseEndsAt + ALLOWANCES.bearerGrace * MINUTE_MS;
164
+ }
165
+
166
+ /** The bearer's expiry as the mint at provisioning sets it, before the lease
167
+ * has started: the provisioning allowance, the lease and the grace — replaced
168
+ * by `bearerExpiresAt` the moment the harness starts the lease. */
169
+ export function provisionalBearerExpiresAt(now: number, leaseMinutes: number): number {
170
+ return now + (ALLOWANCES.provision + leaseMinutes + ALLOWANCES.bearerGrace) * MINUTE_MS;
171
+ }
172
+
173
+ /** The per-command bash budget's rows: the default when a call names none, the
174
+ * ceiling a call may raise it to, and the floor under which a number is a
175
+ * typo (docs/reference/specs/execution.md item 11). */
176
+ export const BASH_COMMAND = { defaultMinutes: 5, maxMinutes: 20, minMs: 1_000 } as const;
177
+
178
+ /** The pace that marks a run as looping rather than working: a model turn
179
+ * every ten seconds, sustained for the whole wall clock. A busy run takes
180
+ * 20–40 s a turn (a model think plus a tool call), so a run that averages six
181
+ * a minute from start to end is re-issuing calls, not making progress — and
182
+ * its turn cap ends it before the wall clock would, with a write-up that
183
+ * says so (docs/reference/specs/harness-pi.md item 15). */
184
+ export const RUNAWAY_TURNS_PER_MINUTE = 6;
185
+
186
+ /** The turn cap a wall clock implies: `maxMinutes × RUNAWAY_TURNS_PER_MINUTE`.
187
+ * Every preset that runs the loop derives its `maxTurns` from this, so the
188
+ * cap is never a number a good run reaches — the minutes are the budget. */
189
+ export function runawayTurnCap(maxMinutes: number): number {
190
+ return maxMinutes * RUNAWAY_TURNS_PER_MINUTE;
191
+ }
192
+
193
+ /** A pipeline's loop as the config allows it: `maxRounds` counts review rounds. */
194
+ export interface Loop {
195
+ maxRounds: number;
196
+ }
197
+
198
+ /** A round's position in its loop: its kind and its index, so the reserve is
199
+ * the rounds after that index, never a table keyed by kind alone. */
200
+ export interface RoundPosition {
201
+ kind: RoundKind;
202
+ index: number;
203
+ }
204
+
205
+ /** The rounds of a loop in order: the coding round, then a review and, after
206
+ * every review but the last, a fix, then the merge wait. */
207
+ export function loopRounds(loop: Loop): RoundKind[] {
208
+ const rounds: RoundKind[] = ["coding"];
209
+ for (let i = 0; i < loop.maxRounds; i++) {
210
+ rounds.push("review");
211
+ if (i < loop.maxRounds - 1) rounds.push("fix");
212
+ }
213
+ rounds.push("merge");
214
+ return rounds;
215
+ }
216
+
217
+ /** Where a round sits in `loopRounds`: the coding round first; review round
218
+ * `n` (counted from 1) and the fix that follows it at `2n − 1` and `2n`; the
219
+ * merge wait last. The ship coordinator numbers its rounds this way, so the
220
+ * reserve it carves with is the one the fit assumed. */
221
+ export function loopPosition(loop: Loop, kind: RoundKind, n = 0): number {
222
+ switch (kind) {
223
+ case "coding":
224
+ return 0;
225
+ case "review":
226
+ return 2 * n - 1;
227
+ case "fix":
228
+ return 2 * n;
229
+ case "merge":
230
+ return 2 * loop.maxRounds;
231
+ }
232
+ }
233
+
234
+ /** A child spawned outside any loop (a conductor's): the parent's whole
235
+ * remainder in whole minutes, refused under the child preset's floor — a
236
+ * child under its floor costs a thread and a model turn and finishes nothing. */
237
+ export function carveChildOfParent(
238
+ remainingMs: number,
239
+ preset: LoopPreset,
240
+ ): { kind: "carved"; minutes: number } | { kind: "refused"; reason: "under floor"; minutes: number; floor: number } {
241
+ const minutes = Math.max(0, Math.floor(remainingMs / MINUTE_MS));
242
+ const floor = PRESET_FLOORS[preset];
243
+ if (minutes < floor) return { kind: "refused", reason: "under floor", minutes, floor };
244
+ return { kind: "carved", minutes };
245
+ }
246
+
247
+ /** The minutes a round holds back for what must follow it in the loop: the
248
+ * floor plus provisioning of every later review and fix, and the merge
249
+ * wait's floor. A merge holds nothing back; a round with no loop (a
250
+ * conductor's child) holds nothing back either. */
251
+ export function reserveMinutes(
252
+ round: RoundPosition,
253
+ loop: Loop | undefined,
254
+ floors: Readonly<Record<RoundKind, number>> = FLOORS,
255
+ ): number {
256
+ if (!loop) return 0;
257
+ let reserve = 0;
258
+ for (const kind of loopRounds(loop).slice(round.index + 1)) {
259
+ reserve += kind === "merge" ? floors.merge : floors[kind] + ALLOWANCES.provision;
260
+ }
261
+ return reserve;
262
+ }
263
+
264
+ /** What a round asks for: the preset's ask for a coding or fix round, the
265
+ * review's for a review, the merge wait's own for the merge. */
266
+ export function roundAskMinutes(kind: RoundKind): number {
267
+ switch (kind) {
268
+ case "coding":
269
+ case "fix":
270
+ return ASKS.coding;
271
+ case "review":
272
+ return ASKS.review;
273
+ case "merge":
274
+ return MERGE_WAIT_ASK_MINUTES;
275
+ }
276
+ }
277
+
278
+ /** A carve's result: the minutes with what bounded them (`ask` when the round
279
+ * got its whole ask, `parent` when the remainder minus the reserve was
280
+ * tighter) and what the parent holds back, or a refusal when the minutes the
281
+ * remainder leaves fall under the round's floor. */
282
+ export type Carve =
283
+ | { kind: "carved"; minutes: number; boundedBy: "ask" | "parent"; holds: number }
284
+ | { kind: "refused"; reason: "under floor"; minutes: number; floor: number; holds: number };
285
+
286
+ /** The only place a round's minutes are computed: the parent's remainder in
287
+ * whole minutes minus the round's reserve, capped at the round's ask, refused
288
+ * under the round's floor. `loop` is undefined for a child outside any loop. */
289
+ export function carve(remainingMs: number, round: RoundPosition, loop: Loop | undefined): Carve {
290
+ const holds = reserveMinutes(round, loop);
291
+ const ask = roundAskMinutes(round.kind);
292
+ const headroom = Math.floor(remainingMs / MINUTE_MS) - holds;
293
+ const minutes = Math.min(ask, headroom);
294
+ const floor = FLOORS[round.kind];
295
+ // A refusal reports what the remainder left, never a negative number.
296
+ if (minutes < floor) return { kind: "refused", reason: "under floor", minutes: Math.max(0, minutes), floor, holds };
297
+ return { kind: "carved", minutes, boundedBy: minutes === ask ? "ask" : "parent", holds };
298
+ }
299
+
300
+ /** A pipeline's wall clock and its loop: what `fit` judges. */
301
+ export interface Pipeline extends Loop {
302
+ maxMinutes: number;
303
+ }
304
+
305
+ /** The fit: a pipeline holds its first child at its ask and every later round
306
+ * at its floor — `provision + ask(coding) + reserve(coding, loop) ≤
307
+ * maxMinutes`. Asserted at verify over the registry, at config load over the
308
+ * deployment's `ship` block (`validateShip`), and at the fork over a request a
309
+ * boundary or a `budget:` directive clipped. `need` is the sum a refusal names. */
310
+ export function fit(pipeline: Pipeline): { ok: boolean; need: number; have: number } {
311
+ const need = ALLOWANCES.provision + ASKS.coding + reserveMinutes({ kind: "coding", index: 0 }, pipeline);
312
+ return { ok: pipeline.maxMinutes >= need, need, have: pipeline.maxMinutes };
313
+ }
@@ -54,7 +54,6 @@ import {
54
54
  type ShipCaps,
55
55
  type StepReturn,
56
56
  type UnitEnding,
57
- type UnitPipelineInput,
58
57
  type UnitPipelineState,
59
58
  } from "../ship/coordinator.js";
60
59
  import { checksSettledEventType, isCoordinatorUnit, runFinishedEventType, type CoordinatorUnit } from "./contract.js";
@@ -170,7 +169,6 @@ interface PlanFacts {
170
169
  repo: string;
171
170
  base: string;
172
171
  caps: ShipCaps;
173
- childMinutes: UnitPipelineInput["childMinutes"];
174
172
  units: CoordinatorUnit[];
175
173
  }
176
174
 
@@ -184,12 +182,6 @@ function readPlan(a: BotAnswer): PlanFacts {
184
182
  if (typeof b.repo !== "string" || typeof b.base !== "string") throw new UnreadableAnswer("plan", a, "repo and base");
185
183
  if (!isMinutes(b.caps) || typeof b.caps.maxRounds !== "number" || typeof b.caps.maxMinutes !== "number")
186
184
  throw new UnreadableAnswer("plan", a, "caps");
187
- if (
188
- !isMinutes(b.childMinutes) ||
189
- typeof b.childMinutes.coding !== "number" ||
190
- typeof b.childMinutes.review !== "number"
191
- )
192
- throw new UnreadableAnswer("plan", a, "childMinutes");
193
185
  const units: unknown = b.units;
194
186
  if (!Array.isArray(units) || !units.every(isCoordinatorUnit)) throw new UnreadableAnswer("plan", a, "units");
195
187
  return {
@@ -204,7 +196,6 @@ function readPlan(a: BotAnswer): PlanFacts {
204
196
  repo: b.repo,
205
197
  base: b.base,
206
198
  caps: { maxRounds: b.caps.maxRounds, maxMinutes: b.caps.maxMinutes },
207
- childMinutes: { coding: b.childMinutes.coding, review: b.childMinutes.review },
208
199
  units,
209
200
  };
210
201
  }
@@ -486,7 +477,6 @@ async function runUnit(
486
477
  repo: plan.repo,
487
478
  base: plan.base,
488
479
  caps: plan.caps,
489
- childMinutes: plan.childMinutes,
490
480
  // The instance's field decides who merges (record 0031's merge grant),
491
481
  // carried here by the plan route: the hand-off wrote `runner` on a
492
482
  // seeded plan and `person` on a task, and the door re-checks it — the