@intentic/sandbox-contract 1.246.1 → 1.247.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (75) hide show
  1. package/dist/batch-runs.d.ts +2 -0
  2. package/dist/batch-runs.d.ts.map +1 -1
  3. package/dist/batch-runs.js +1 -0
  4. package/dist/batch-runs.js.map +1 -1
  5. package/dist/contracts/agent.contract.d.ts +19 -0
  6. package/dist/contracts/agent.contract.d.ts.map +1 -1
  7. package/dist/contracts/settings.contract.d.ts +97 -10
  8. package/dist/contracts/settings.contract.d.ts.map +1 -1
  9. package/dist/contracts/usage.contract.d.ts +22 -0
  10. package/dist/contracts/usage.contract.d.ts.map +1 -1
  11. package/dist/contracts/usage.contract.js +19 -0
  12. package/dist/contracts/usage.contract.js.map +1 -1
  13. package/dist/definition.d.ts +16 -20
  14. package/dist/definition.d.ts.map +1 -1
  15. package/dist/fast-tier.js +1 -1
  16. package/dist/fast-tier.js.map +1 -1
  17. package/dist/index.d.ts +140 -12
  18. package/dist/index.d.ts.map +1 -1
  19. package/dist/index.js +2 -2
  20. package/dist/index.js.map +1 -1
  21. package/dist/model-pins.d.ts +17 -0
  22. package/dist/model-pins.d.ts.map +1 -0
  23. package/dist/{quick-model.js → model-pins.js} +16 -10
  24. package/dist/model-pins.js.map +1 -0
  25. package/dist/model-roles.d.ts +144 -0
  26. package/dist/model-roles.d.ts.map +1 -0
  27. package/dist/model-roles.js +129 -0
  28. package/dist/model-roles.js.map +1 -0
  29. package/dist/schemas/agent.d.ts +21 -2
  30. package/dist/schemas/agent.d.ts.map +1 -1
  31. package/dist/schemas/agent.js +4 -2
  32. package/dist/schemas/agent.js.map +1 -1
  33. package/dist/schemas/plan-limits.d.ts +20 -0
  34. package/dist/schemas/plan-limits.d.ts.map +1 -1
  35. package/dist/schemas/plan-limits.js +21 -0
  36. package/dist/schemas/plan-limits.js.map +1 -1
  37. package/dist/schemas/settings.d.ts +82 -5
  38. package/dist/schemas/settings.d.ts.map +1 -1
  39. package/dist/schemas/settings.js +15 -18
  40. package/dist/schemas/settings.js.map +1 -1
  41. package/dist/schemas/usage.d.ts +5 -0
  42. package/dist/schemas/usage.d.ts.map +1 -1
  43. package/dist/schemas/usage.js +5 -0
  44. package/dist/schemas/usage.js.map +1 -1
  45. package/package.json +4 -4
  46. package/src/agent-catalog.ts +1 -1
  47. package/src/batch-runs.test.ts +10 -5
  48. package/src/batch-runs.ts +10 -3
  49. package/src/contracts/usage.contract.ts +31 -0
  50. package/src/events.ts +1 -1
  51. package/src/fast-tier.test.ts +1 -1
  52. package/src/fast-tier.ts +5 -5
  53. package/src/index.ts +2 -2
  54. package/src/{quick-model.test.ts → model-pins.test.ts} +73 -29
  55. package/src/model-pins.ts +183 -0
  56. package/src/model-roles.ts +224 -0
  57. package/src/plan-pools.ts +1 -1
  58. package/src/prompt-complexity.test.ts +1 -1
  59. package/src/prompt-complexity.ts +2 -2
  60. package/src/provider-specs.test.ts +1 -1
  61. package/src/schemas/agent.ts +42 -17
  62. package/src/schemas/agents.ts +2 -2
  63. package/src/schemas/plan-limits.ts +50 -0
  64. package/src/schemas/settings.ts +77 -88
  65. package/src/schemas/usage.ts +59 -0
  66. package/dist/agent-run-model.d.ts +0 -4
  67. package/dist/agent-run-model.d.ts.map +0 -1
  68. package/dist/agent-run-model.js +0 -13
  69. package/dist/agent-run-model.js.map +0 -1
  70. package/dist/quick-model.d.ts +0 -15
  71. package/dist/quick-model.d.ts.map +0 -1
  72. package/dist/quick-model.js.map +0 -1
  73. package/src/agent-run-model.test.ts +0 -76
  74. package/src/agent-run-model.ts +0 -65
  75. package/src/quick-model.ts +0 -162
@@ -0,0 +1,183 @@
1
+ import { accessFor, modelsFor, PROVIDERS } from "./agent-catalog.js";
2
+ import { ACCESS_COST } from "./provider-specs.js";
3
+ import { compareCheapestFirst, familyOf, tierRankOf } from "./model-order.js";
4
+ import { type ModelRole, modelRole } from "./model-roles.js";
5
+ import type { AgentProvider, ModelPin } from "./schemas/agent.js";
6
+
7
+ /* WHICH MODELS A ROLE MAY RUN, IN THE ORDER TO TRY THEM. One resolver over every list in
8
+ * settings.modelRoles, and the browser and the daemon both read it.
9
+ *
10
+ * IT IS AN ORDER, NOT A MODEL, and that is the shape of every list this file answers for. A single pick is a
11
+ * single point of failure: the account it names spends its allowance on the chat all morning, and the role
12
+ * fails on a limit for the rest of the day while three other connected providers sit idle. So a setting is a
13
+ * LIST read top to bottom, this hands back the whole ladder, and the caller walks it until one answers. Nothing
14
+ * here decides WHICH failures are worth stepping over — only the runner has made the call and seen it fail —
15
+ * this side says what the running order is.
16
+ *
17
+ * THE RULE LIVES IN THE CONTRACT because both sides need the same answer for different jobs: the daemon runs
18
+ * the model, and the browser has to NAME it, in the settings row's "Auto: …" line, before anything has run. Two
19
+ * implementations would drift precisely where it matters most, since a row promising Haiku while the daemon
20
+ * bills Opus is worse than no row at all.
21
+ *
22
+ * WHAT AN EMPTY LIST MEANS IS THE ROLE'S OWN ANSWER (model-roles.ts): a `helper` role derives the Auto ladder
23
+ * from whatever is connected, a `run` role resolves to nothing and lets the caller's floor answer. That fork
24
+ * used to be two near-identical files; it is one line here because it was always one difference. */
25
+
26
+ /* One provider's standing in the decision: whether a turn on it can be sent at all, and what its catalog holds.
27
+ *
28
+ * ACP agents are deliberately not expressible here — an ACP row's model id is empty because the agent owns its
29
+ * own model, so there is no rung to point it at. `endpoint/<id>` providers ARE, and have to be: their models
30
+ * appear in the same picker the settings rows build their options from, so a pin naming one has to hold rather
31
+ * than fall silently back to Auto and spend an account the user was deliberately steering away from. */
32
+ export interface ModelSource {
33
+ // AgentProvider, not NativeProvider: an endpoint's id is user-created and cannot be in a fixed union. Auto's
34
+ // ranking degrades gracefully for one, costOf falls to the metered rung and an id with no tier word is
35
+ // UNRANKED, which is genuine last place, so an endpoint effectively only wins Auto when nothing else is
36
+ // connected, while a PIN on one holds. Both are the right answers: what a turn on someone's own model server
37
+ // costs is not a fact this repo can know, so it is not one Auto should be asserting.
38
+ readonly provider: AgentProvider;
39
+ // The same connection predicate every other surface gates on (access.ts web-side, the daemon's own account
40
+ // stores daemon-side). A catalog is never empty by construction, so "has rows" says nothing about "can send".
41
+ readonly ready: boolean;
42
+ readonly models: readonly string[];
43
+ }
44
+
45
+ export interface ModelChoice {
46
+ readonly provider: AgentProvider;
47
+ readonly model: string;
48
+ }
49
+
50
+ // A (provider, model) pair on the wire: `${provider}:${modelId}`, the same key shape the model picker mints for
51
+ // its entries (PickerEntry.key). Every role list stores PINS rather than these keys — an entry says how it runs
52
+ // as well as which model it is — but the key is still how two entries are compared, how a role list dedupes,
53
+ // and how `autoFastModels` (which pins no knobs, see its own note in settings.ts) is stored.
54
+ export const modelPinKey = (choice: ModelChoice): string => `${choice.provider}:${choice.model}`;
55
+
56
+ /* Split on the FIRST colon only: a provider id never contains one and a model id might. Exported because
57
+ * `autoFastModels` stores these keys, and because a session composed from a pin travels as one
58
+ * (composeSession). */
59
+ export const parsePinned = (pinned: string): ModelChoice | undefined => {
60
+ const separator = pinned.indexOf(`:`);
61
+ if (separator <= 0 || separator === pinned.length - 1) {
62
+ return undefined;
63
+ }
64
+ return { provider: pinned.slice(0, separator), model: pinned.slice(separator + 1) };
65
+ };
66
+
67
+ /* A pin as a person reads it: the catalog's own label for the id, or the id itself for one the static catalog
68
+ * has not caught up with (the picker offers a custom-id escape hatch, so this is a real case rather than a
69
+ * defensive branch). Beside parsePinned because the two are always wanted together, by any surface that has to
70
+ * name what a click is about to spend BEFORE it spends it, and the two loudest of those are extensions that
71
+ * share no other code with each other. */
72
+ export const pinnedModelLabel = (choice: ModelChoice): string =>
73
+ modelsFor(choice.provider).find((option) => option.value === choice.model)?.label ?? choice.model;
74
+
75
+ // The cheapest row a provider publishes, its whole catalog read from the cheap end. Undefined for a catalog
76
+ // that hasn't loaded yet, which is a real state: every provider serves a floor, but only once something has
77
+ // asked it.
78
+ const cheapestOf = (source: ModelSource): string | undefined => source.models.toSorted(compareCheapestFirst)[0];
79
+
80
+ // Where a provider's cheapest row sits on the shared tier scale, and therefore how well it answers the question
81
+ // Auto asks. UNRANKED (-1) is a genuine last place: it means the id carries no tier word we know, so the row is
82
+ // the provider's base line rather than its budget one.
83
+ const tierOf = (model: string): number => tierRankOf(familyOf(model));
84
+
85
+ // PROVIDERS order, as the final tiebreak. Arbitrary, but the SAME arbitrary answer on every read, the property
86
+ // compareUnrankedModelIds exists to guarantee, and the one a default actually needs. An endpoint is in no fixed
87
+ // list, so it reads -1 and leads the tiebreak; unreachable in practice, since it can never tie on cost.
88
+ const providerOrder = (provider: AgentProvider): number => PROVIDERS.findIndex((entry) => entry.value === provider);
89
+
90
+ /* WHAT AN ENDPOINT COSTS, one rung past every provider's, and the reason it is a number here rather than a
91
+ * member of AccessKind. That axis describes the providers this repo ships, and every one of them is unlocked by
92
+ * signing in to something the user already holds, so none of them is metered per call. An endpoint is the
93
+ * opposite: whatever gateway somebody pointed us at, whose bill this repo cannot see. Reading it as dearer than
94
+ * anything on the table is the conservative answer, and it is what keeps Auto from reaching for a paid gateway
95
+ * on its own initiative. */
96
+ const METERED_COST = Math.max(...Object.values(ACCESS_COST)) + 1;
97
+
98
+ // How much a call on this provider costs at the margin. Every native provider declares an access kind; an
99
+ // endpoint declares none, and takes the metered rung above.
100
+ const costOf = (provider: AgentProvider): number => {
101
+ const access = accessFor(provider);
102
+ return access === undefined ? METERED_COST : ACCESS_COST[access.kind];
103
+ };
104
+
105
+ /* AUTO, every connected provider's cheapest row, best-first, as a ladder rather than a winner. The floor under
106
+ * every `helper` role, and under nothing else.
107
+ *
108
+ * Ranked on TIER FIRST, then cost. That order is the point: a helper's Auto exists to not be the frontier
109
+ * model, so a free flagship is still the wrong tool, while a free Haiku-class row and a subscription
110
+ * Haiku-class row differ only in whose quota they spend. Cost then breaks that tie towards the channel the user
111
+ * is not paying per token for, and against the one they are.
112
+ *
113
+ * NOTE WHAT AUTO IS AND IS NOT AN ARGUMENT FOR. It is the answer for an owner who has said nothing, not a claim
114
+ * that cheap is right: an owner who pins Opus to commit messages is not being talked out of it, which is the
115
+ * whole reason these lists are per role. Auto is what a row says while it is empty.
116
+ *
117
+ * The whole ladder, not just its head, because the same ranking that picks the best answer also states the best
118
+ * SECOND answer, and a sandbox with three accounts connected should not lose its commit messages for six hours
119
+ * because one of them is spent. */
120
+ export const autoLadder = (sources: readonly ModelSource[]): readonly ModelPin[] =>
121
+ sources
122
+ .filter((source) => source.ready)
123
+ .flatMap((source) => {
124
+ const model = cheapestOf(source);
125
+ return model === undefined ? [] : [{ provider: source.provider, model }];
126
+ })
127
+ .toSorted(
128
+ (left, right) =>
129
+ tierOf(right.model) - tierOf(left.model) ||
130
+ costOf(left.provider) - costOf(right.provider) ||
131
+ providerOrder(left.provider) - providerOrder(right.provider),
132
+ );
133
+
134
+ /* WHICH MODELS THIS ROLE MAY RUN, IN THE ORDER TO TRY THEM, given what this sandbox has connected.
135
+ * `pinned` is the stored setting: settings.modelRoles[role], an ordered list of pins, empty for the role's own
136
+ * floor.
137
+ *
138
+ * A pin only holds while its provider is READY: an account the user disconnected would otherwise sit at the
139
+ * head of the chain failing on a credential error, when the sandbox can plainly still answer. Dropping it is
140
+ * the same move the composer already makes when a live catalog stops offering the selected model. It stays on
141
+ * SCREEN, greyed — the settings row renders the stored list, not this one — because a setting that vanished
142
+ * from view would look like the app had eaten it.
143
+ *
144
+ * THE PINNED LIST IS THE WHOLE ANSWER whenever any of it survives that filter. The floor is NOT appended
145
+ * underneath, and that is deliberate: a user who writes down three models has said which accounts this job may
146
+ * spend, and quietly reaching for a fourth when all three are out is exactly the "spend an account they were
147
+ * steering away from" failure a pin exists to prevent. When NONE of the pins is connected any more the list has
148
+ * stopped saying anything about this sandbox, so the floor takes over rather than leaving a dead button.
149
+ *
150
+ * THE WHOLE PIN SURVIVES, not the pair inside it: an entry's effort, thinking, speed and harness are what the
151
+ * work is composed from, so a resolver handing back a bare (provider, model) would silently run the head of the
152
+ * list at the provider's defaults. Nothing here reads or judges those fields, which is the point of carrying
153
+ * them whole.
154
+ *
155
+ * Empty out means the role has nothing it can reach. For a `helper` that is a sandbox with nothing connected at
156
+ * all, and the caller renders a control that says so rather than a live button that fails on click; for a `run`
157
+ * it is the ordinary state of an unpinned role, and the caller's own floor answers. */
158
+ export const resolveRoleModels = (sources: readonly ModelSource[], pinned: readonly ModelPin[], role: ModelRole): readonly ModelPin[] => {
159
+ const ready = new Set(sources.filter((source) => source.ready).map((source) => source.provider));
160
+ // Taken verbatim, unvalidated against the catalog on purpose: the picker offers a custom-id escape hatch for
161
+ // a model a catalog hasn't caught up with, and second-guessing the user's own id here would silently run a
162
+ // different model than the settings row names.
163
+ const requested = pinned.filter((pin) => ready.has(pin.provider));
164
+ /* The same model twice would spend two attempts proving one account is out — a real state, since the list is
165
+ * hand-edited and Auto's ladder can rank a provider the user has also pinned.
166
+ *
167
+ * THE FIRST OF A PAIR WINS, WHOLE. Two entries can name one model and differ in their knobs (the same Sonnet
168
+ * at Max and again at Low, written while reordering the list), and the one the user reads first is the one
169
+ * they meant; keeping the earlier position with the later entry's effort would run a tier that appears
170
+ * nowhere the pin does. */
171
+ const chain: ModelPin[] = [];
172
+ for (const pin of requested) {
173
+ if (!chain.some((held) => modelPinKey(held) === modelPinKey(pin))) {
174
+ chain.push(pin);
175
+ }
176
+ }
177
+ if (chain.length > 0) {
178
+ return chain;
179
+ }
180
+ // The floor, which the role declares. An id outside the table has no floor to fall to, and answering with
181
+ // the cheapest connected model for it would be this file inventing a job.
182
+ return modelRole(role)?.kind === `helper` ? autoLadder(sources) : [];
183
+ };
@@ -0,0 +1,224 @@
1
+ import { z } from "zod";
2
+
3
+ /* EVERY JOB IN THIS SANDBOX THAT PICKS A MODEL, LISTED BY THE JOB IT IS, one row per answer the owner is
4
+ * entitled to give differently.
5
+ *
6
+ * WHAT THIS REPLACED, and why it had to go. Model settings used to be grouped by how HARD the work was assumed
7
+ * to be: a "quick model" for the small automatic jobs, an "agent runs" tier for the big ones. Both names
8
+ * described an intensity rather than a job, and an intensity is a guess somebody else made about work the owner
9
+ * knows better. A commit message written by a frontier model is not a mistake, it is a preference, and the
10
+ * grouping made it unsayable: pinning Opus to get better commit subjects also pinned it to session titles, to
11
+ * every loop verdict, and to the safety judge. One bundled row could not say the thing anyone actually wanted
12
+ * to say.
13
+ *
14
+ * SO THE UNIT IS THE ROLE. Each entry here is one place a model gets chosen, and each gets its own ordered list
15
+ * in settings.modelRoles. Seventeen rows is more than four, and that is the point rather than a cost being
16
+ * absorbed: the configuration and the UI both already existed per use-case, and collapsing them was the only
17
+ * thing standing between the owner and a choice the machinery could already honour. A simpler face over this
18
+ * (presets: "cheap everywhere", "frontier everywhere") is a thing that can be built ON TOP of a true model, and
19
+ * cannot be unpicked from a lossy one.
20
+ *
21
+ * EVERY LIST IS AN ORDERED LADDER, whatever the role, and for one reason: the interesting failure of a pinned
22
+ * model is not that it is wrong, it is that it is CONNECTED AND WILL NOT ANSWER TODAY. The account's allowance
23
+ * went on the chat this morning, and one spent provider takes the role down for hours while three others sit
24
+ * idle. Written in order, the next entry catches it.
25
+ *
26
+ * TWO KINDS, and the only thing that separates them is what an EMPTY list means. That is a real fork rather
27
+ * than a leftover of the old grouping, and it is declared per role because it is a property of the job:
28
+ *
29
+ * helper — a one-shot. One prompt, no tools, one string back, and it is over. An empty list resolves to the
30
+ * AUTO LADDER: every connected provider's cheapest row, best-first (model-pins.ts). Deriving is the
31
+ * right default here because the job is small and repeatable, so the cost of a wrong guess is one
32
+ * cheap call, and because a derived answer improves by itself when an account is connected tomorrow.
33
+ *
34
+ * run — a whole session with tools and a worktree, started by a surface rather than by a person at a
35
+ * composer. An empty list resolves to NOTHING, and the caller's own floor answers: the model the
36
+ * owner picked for their chat. Nothing here can judge whether a job is worth the frontier tier, and
37
+ * a wrong guess is billed in whole sessions rather than in tokens, so the honest fallback is a
38
+ * choice they made rather than one this table invented.
39
+ *
40
+ * EVERY ENTRY IS A FULL PIN (ModelPinSchema): which model, and how it runs — effort, thinking, speed, harness.
41
+ * The helper roles carry them too, which they did not use to: a one-shot ran with reasoning forcibly off, so
42
+ * pinning a reasoning model to commit messages bought the price of one and the behaviour of neither. The knobs
43
+ * now ride through the one-shot path (the daemon's role-model.ts), so an entry means what it says wherever it
44
+ * is written.
45
+ *
46
+ * ADDING A ROLE IS ONE ROW HERE. The settings key, the resolver's floor, the daemon's lookup and the settings
47
+ * page's row all read this table, so a job that starts picking a model tomorrow becomes configurable by saying
48
+ * what it is. That is the property the old grouping cost: a new surface inherited "agent runs" by being
49
+ * unattended, which is how a documentation sweep and a production incident came to share one tier. */
50
+
51
+ export const ModelRoleKindSchema = z.enum(["helper", "run"]);
52
+ export type ModelRoleKind = z.infer<typeof ModelRoleKindSchema>;
53
+
54
+ /* The shape of a row. `id` is a bare string HERE and narrowed on the exported type below, because the id union
55
+ * is derived from this very table: a self-referential `satisfies` would be a type that has to know its own
56
+ * answer before it can check it. */
57
+ interface ModelRoleRow {
58
+ readonly id: string;
59
+ // What the settings row is called. A JOB, in the owner's words, never a tier.
60
+ readonly label: string;
61
+ // The row's one line: what this model is asked to do. Read beside the label, so it says what the label
62
+ // cannot rather than restating it.
63
+ readonly blurb: string;
64
+ readonly kind: ModelRoleKind;
65
+ // The row's glyph, from the shared icon set. Here rather than in a web-side map because the whole value of
66
+ // this table is that a role is declared ONCE; a second table keyed by the same ids is the drift this
67
+ // replaced, moved one layer up.
68
+ readonly icon: string;
69
+ }
70
+
71
+ /* THE TABLE. Ordered as the settings page draws it, and the order is an argument about reach: the one-shots
72
+ * first, because nobody chose a model for them and they run constantly; then the runs somebody's click starts;
73
+ * then the runs that start themselves, which are the ones an owner is least likely to be watching and most
74
+ * likely to want held to a budget.
75
+ *
76
+ * The ids are the wire vocabulary: a turn carries one (AgentTurn.runRole), so they are kebab-case and stable,
77
+ * and renaming one is a breaking change to the setting rather than a cosmetic edit. */
78
+ export const MODEL_ROLES = [
79
+ {
80
+ id: "commit-message",
81
+ label: "Commit messages",
82
+ blurb: "The subject written when an agent's work lands, and the release note under it.",
83
+ kind: "helper",
84
+ icon: "file-edit",
85
+ },
86
+ {
87
+ id: "session-title",
88
+ label: "Session titles",
89
+ blurb: "The name a conversation wears on the board, written a second into its first turn.",
90
+ kind: "helper",
91
+ icon: "pencil",
92
+ },
93
+ {
94
+ /* THE ONE HELPER WHOSE INPUT IS ADVERSARIAL, and the reason the old bundling was worst here. Its prompt
95
+ * contains a command the agent is about to run, which may have arrived from a stranger's web page, and a
96
+ * small model can be talked round by it. Being wrong is expensive in both directions: a needless card
97
+ * teaches the owner to click through the next one. */
98
+ id: "safety-judge",
99
+ label: "Safety judge",
100
+ blurb: "Which model reads your safety policy before a flagged command runs.",
101
+ kind: "helper",
102
+ icon: "shield",
103
+ },
104
+ {
105
+ id: "loop-verdict",
106
+ label: "Loop verdicts",
107
+ blurb: "Whether a loop's iteration met the goal, or the loop goes round again.",
108
+ kind: "helper",
109
+ icon: "check-square",
110
+ },
111
+ {
112
+ id: "pipeline-fix",
113
+ label: "Pipeline fixes",
114
+ blurb: "The agent started by Fix on a red pipeline.",
115
+ kind: "run",
116
+ icon: "wave-pulse",
117
+ },
118
+ {
119
+ id: "deployment-fix",
120
+ label: "Deployment fixes",
121
+ blurb: "The agent started by Fix on a deployment that is down.",
122
+ kind: "run",
123
+ icon: "server",
124
+ },
125
+ {
126
+ id: "maintenance-chore",
127
+ label: "Maintenance chores",
128
+ blurb: "A chore run started from the Maintenance board.",
129
+ kind: "run",
130
+ icon: "wrench",
131
+ },
132
+ {
133
+ id: "documentation-run",
134
+ label: "Documentation runs",
135
+ blurb: "A pass over a repo's own documentation.",
136
+ kind: "run",
137
+ icon: "book",
138
+ },
139
+ {
140
+ id: "acceptance-run",
141
+ label: "Acceptance runs",
142
+ blurb: "One session per story in an acceptance fan-out.",
143
+ kind: "run",
144
+ icon: "list-check",
145
+ },
146
+ {
147
+ id: "pre-push-fix",
148
+ label: "Pre-push fixes",
149
+ blurb: "The fix proposed when a check fails on the way to a push.",
150
+ kind: "run",
151
+ icon: "cloud-upload",
152
+ },
153
+ {
154
+ id: "automation-wake",
155
+ label: "Automation wakes",
156
+ blurb: "A turn an automation fires: a schedule, a webhook, a message from outside.",
157
+ kind: "run",
158
+ icon: "clock",
159
+ },
160
+ {
161
+ id: "approval-queue",
162
+ label: "Approvals queue",
163
+ blurb: "The turn that publishes or acts on what you approved.",
164
+ kind: "run",
165
+ icon: "check-circle",
166
+ },
167
+ {
168
+ id: "extension-review",
169
+ label: "Extension update reviews",
170
+ blurb: "The agent that reads an extension update before it is applied.",
171
+ kind: "run",
172
+ icon: "box",
173
+ },
174
+ {
175
+ id: "loop-iteration",
176
+ label: "Loop iterations",
177
+ blurb: "Each round of a loop working towards its goal.",
178
+ kind: "run",
179
+ icon: "repeat",
180
+ },
181
+ {
182
+ id: "watch-wake",
183
+ label: "Watch wakes",
184
+ blurb: "The turn a watch starts when the thing it was watching happens.",
185
+ kind: "run",
186
+ icon: "eye",
187
+ },
188
+ {
189
+ id: "verify-nudge",
190
+ label: "Verify nudges",
191
+ blurb: "The follow-up turn sent when work was left unverified.",
192
+ kind: "run",
193
+ icon: "search",
194
+ },
195
+ {
196
+ /* THE ONE ROLE THAT ANSWERS A CHOICE RATHER THAN FILLING A SILENCE. A spawning agent may name its
197
+ * child's provider, and when it does that wins, exactly as a caret pick wins on every other run role.
198
+ * This is what answers when it names none, which used to be a hardcoded "claude". */
199
+ id: "child-agent",
200
+ label: "Child agents",
201
+ blurb: "What an agent's own subagents run on when it names no model for them.",
202
+ kind: "run",
203
+ icon: "users",
204
+ },
205
+ ] as const satisfies readonly ModelRoleRow[];
206
+
207
+ export type ModelRole = (typeof MODEL_ROLES)[number]["id"];
208
+ export type ModelRoleSpec = ModelRoleRow & { readonly id: ModelRole };
209
+
210
+ export const MODEL_ROLE_IDS = MODEL_ROLES.map((role) => role.id) as readonly ModelRole[];
211
+
212
+ /* The wire form. An enum rather than a string, unlike most ids in this contract, because there is no case for
213
+ * an unknown one: a role is a place in THIS codebase where a model gets chosen, so a value outside the table
214
+ * names nothing, and a settings file or a turn carrying one is a typo worth a clean error rather than a list
215
+ * silently ignored. */
216
+ export const ModelRoleSchema = z.enum(MODEL_ROLE_IDS as [ModelRole, ...ModelRole[]]);
217
+
218
+ const BY_ID = new Map<string, ModelRoleSpec>(MODEL_ROLES.map((role) => [role.id, role]));
219
+
220
+ /** What this role is, or undefined for an id no build of this table declares. */
221
+ export const modelRole = (id: string): ModelRoleSpec | undefined => BY_ID.get(id);
222
+
223
+ /** The roles of one kind, in table order: the two blocks the settings page draws. */
224
+ export const modelRolesOfKind = (kind: ModelRoleKind): readonly ModelRoleSpec[] => MODEL_ROLES.filter((role) => role.kind === kind);
package/src/plan-pools.ts CHANGED
@@ -6,7 +6,7 @@ import type { AccountUsage, UsageWindow, WindowGates } from "./schemas/plan-limi
6
6
  * call, is this Google fleet spent for Claude Opus, when does the pool that refused this turn reopen, what
7
7
  * does the ring beside the composer measure. The pools a reading carries answer that only through their
8
8
  * `gates` (UsageWindowSchema says why the reader decides them), and this file is the one place the gate is
9
- * read, so the daemon's account picker, its quick-model walk, its refusal dressing and the browser's rings,
9
+ * read, so the daemon's account picker, its one-shot helper walk, its refusal dressing and the browser's rings,
10
10
  * rail and picker rows all agree about which pool is binding for a given model.
11
11
  *
12
12
  * WITHOUT A MODEL the answer is the account's own tightest pool, which is what a roster or a rail that has not
@@ -88,7 +88,7 @@ test("plan mode is a request to think, so it is never answered by the model that
88
88
  });
89
89
 
90
90
  test("a surface-started run is never downgraded, because nobody is watching it fail", () => {
91
- // Same call agentRunModels already makes in the other direction: a run billed whole, with a worktree in
91
+ // Same call a `run` role already makes in the other direction: a run billed whole, with a worktree in
92
92
  // it, is not the place to spend a guess.
93
93
  expect(tierOf(`what is a closure?`, { unattended: true })).toBe(`standard`);
94
94
  });
@@ -3,7 +3,7 @@
3
3
  * The job is narrow on purpose: decide whether this turn could have run on the cheap rung of the provider the
4
4
  * user is already on. Nothing here picks a model, nothing here reads a catalog, and nothing here calls
5
5
  * anything. It is a pure function over the turn's own words and shape, so the daemon and the composer can both
6
- * ask it and get the same answer, which is the same reason quick-model.ts lives in the contract rather than in
6
+ * ask it and get the same answer, which is the same reason model-pins.ts lives in the contract rather than in
7
7
  * either of them.
8
8
  *
9
9
  * IT CAN ONLY EVER ROUTE DOWN. The standard tier is not a setting: it is whatever the user already picked. So
@@ -213,7 +213,7 @@ const forcing = (input: ComplexityInput, text: string): ComplexityRule[] => {
213
213
  rules.push("plan-mode");
214
214
  }
215
215
  /* An unattended run is billed whole and nobody is watching it fail. The settings this repo already ships
216
- * make the same call in the other direction: agentRunModels resolves to NOTHING when empty precisely
216
+ * make the same call in the other direction: a `run` role resolves to NOTHING when empty precisely
217
217
  * because "nothing here can judge whether a job is worth the frontier tier". This file does judge, but not
218
218
  * for the runs where a wrong guess costs a whole session with a worktree in it. */
219
219
  if (input.unattended) {
@@ -115,7 +115,7 @@ test("provider ids are unique", () => {
115
115
 
116
116
  /* An id may contain neither a slash nor a colon, and both exclusions are load-bearing rather than tidy.
117
117
  * `endpoint/<id>` uses the slash to namespace a capability-minted provider, and the picker's pinned selections
118
- * are `${provider}:${model}` split on the FIRST colon (quick-model.ts), so an id carrying either would parse as
118
+ * are `${provider}:${model}` split on the FIRST colon (model-pins.ts), so an id carrying either would parse as
119
119
  * something else entirely, silently. */
120
120
  test("no provider id can be mistaken for an endpoint or a pinned selection", () => {
121
121
  for (const spec of PROVIDER_SPECS) {
@@ -1,4 +1,5 @@
1
1
  import { z } from "zod";
2
+ import { ModelRoleSchema } from "../model-roles.js";
2
3
  import { NATIVE_PROVIDERS } from "../provider-specs.js";
3
4
  import { AgentPlacementSchema } from "../runner-protocol.js";
4
5
  import { entryId } from "./internal.js";
@@ -267,6 +268,23 @@ export const AgentTurnSchema = z
267
268
  .describe(
268
269
  "Whether this turn's work merges into the workspace when it finishes. Overrides the conversation's own setting for this turn only.",
269
270
  ),
271
+ /* WHAT STARTED THIS TURN, when it was not a person at a composer, and therefore which of the owner's
272
+ * model lists answers for it (settings.modelRoles, model-roles.ts declares the roles).
273
+ *
274
+ * IT NAMES THE JOB, NOT THE TIER. Before this field there was one list for every unattended turn, so a
275
+ * production incident and a documentation sweep were configured together, and a surface added tomorrow
276
+ * inherited that tier by saying nothing. Now a starter says what it IS, and the owner gets to answer
277
+ * per job: cheap for the sweep, frontier for the incident.
278
+ *
279
+ * ONLY FILLS A SILENCE. The daemon applies the role's list to a turn that named no model AND no
280
+ * provider (turn-resume.ts): a caret pick, an acceptance run's own choice, a workflow step pinned to a
281
+ * provider — all of those already answered the question, and this must not overrule them.
282
+ *
283
+ * A `run` role, always. The `helper` roles never reach here: a one-shot is not a turn, it has no
284
+ * conversation and no worktree, and it asks its own role directly at the seam that spends it. */
285
+ runRole: ModelRoleSchema.optional().describe(
286
+ "What started this turn, when it was not a person typing: which of the sandbox's per-job model lists answers for it. Only used when the turn names no model of its own.",
287
+ ),
270
288
  // Set ONLY by the daemon's own automation dispatchers: this turn opens a conversation on behalf of an
271
289
  // outside message rather than a user. Recorded on the registry entry so the fleet can say where the
272
290
  // agent came from. Requires conversationId, there is nothing to record it on otherwise.
@@ -311,7 +329,7 @@ export const AgentTurnSchema = z
311
329
  * opposite defaults, the chat wants the provider's own catalog default, an unattended run wants the
312
330
  * tier its owner chose for work that spends money while they are not watching.
313
331
  *
314
- * The daemon fills `agent`/`model` and the pinned entry's own knobs from agentRunModels for any turn that says
332
+ * The daemon fills `agent`/`model` and the pinned entry's own knobs from the turn's own role list for any turn that says
315
333
  * this and names none of them (startConversationTurn), walking that list until one can actually be
316
334
  * started. Naming one still wins: every surface-started run now carries a caret that overrides the list
317
335
  * for that run alone, and Acceptance picks per run because it fans a session out per story. Either way
@@ -416,14 +434,14 @@ export type AgentTurn = z.infer<typeof AgentTurnSchema>;
416
434
  * Shared rather than re-declared per route because every surface that starts an agent for the user now carries
417
435
  * that caret, and they must all mean the same thing by it: the pair rides onto the turn as `agent`/`model`, and
418
436
  * the daemon's own fill step then leaves it alone (turn-resume.ts fills only what is absent). ABSENT is the
419
- * ordinary case and the one to keep cheap, nobody touched the caret, so `agentRunModels` answers.
437
+ * ordinary case and the one to keep cheap, nobody touched the caret, so the turn's `runRole` list answers.
420
438
  *
421
439
  * Both halves or neither, because a model id is only meaningful to the provider that vends it: half a pick
422
440
  * would send a Codex model id to Claude. Routes that accept this pass it through verbatim; a model this build
423
441
  * has never heard of is a supported pick, since the picker offers a custom-id escape hatch.
424
442
  *
425
443
  * AND THE TIER IT RUNS AT, because naming a model is only half of what the standing setting says. A pinned
426
- * entry carries its own effort (AgentRunPinSchema), and the daemon applies the pin's knobs ONLY to a turn that
444
+ * entry carries its own effort (ModelPinSchema), and the daemon applies the pin's knobs ONLY to a turn that
427
445
  * named no model (turn-resume.ts): so a caret that could re-point the model but not the tier moved every
428
446
  * override onto the provider's own default effort, and the one moment somebody reaches for the caret is the
429
447
  * failure that just beat the standing order. Optional, and absent means absent, the turn goes out without an
@@ -442,29 +460,36 @@ export const AgentRunPickSchema = z
442
460
  })
443
461
  .optional();
444
462
  export type AgentRunPick = z.infer<typeof AgentRunPickSchema>;
445
- /* A MODEL PINNED FOR EVERY SURFACE-STARTED RUN, one entry of settings.agentRunModels: the standing version of
446
- * the pick above, and not merely which model but HOW it is to be run.
463
+ /* ONE ENTRY OF ONE ROLE'S MODEL LIST (settings.modelRoles): the standing version of the pick above, and not
464
+ * merely which model but HOW it is to be run.
447
465
  *
448
- * THE KNOBS RIDE THE ENTRY RATHER THAN THE LIST, which is the whole reason this is an object where the setting
449
- * used to hold a `${provider}:${model}` string. The reasoning effort was a single field beside the list, so one
450
- * tier answered for every model in it — and the entries of that list are deliberately NOT interchangeable: it
451
- * is a frontier pin with the cheap account underneath that catches it when the first is spent. A tier scale is
452
- * a property of the MODEL as well ('max' is off Kimi's scale entirely, and off Claude's own the moment thinking
466
+ * THE KNOBS RIDE THE ENTRY RATHER THAN THE LIST, which is the whole reason this is an object rather than a
467
+ * `${provider}:${model}` string. The reasoning effort was once a single field beside a list, so one tier
468
+ * answered for every model in it — and the entries of such a list are deliberately NOT interchangeable: it is a
469
+ * frontier pin with the cheap account underneath that catches it when the first is spent. A tier scale is a
470
+ * property of the MODEL as well ('max' is off Kimi's scale entirely, and off Claude's own the moment thinking
453
471
  * is switched off), so a shared effort was either off-scale for half the list or the lowest common rung for all
454
- * of it. Each entry now carries what the composer's picker configures for the turn in front of you.
472
+ * of it. Each entry carries what the composer's picker configures for the turn in front of you.
473
+ *
474
+ * THE SAME SHAPE FOR EVERY ROLE, one-shot helpers included, and that is a deliberate widening. A commit
475
+ * message or a session title used to be pinnable by model alone, on the argument that the daemon runs those
476
+ * with reasoning off and no effort, so a control for either would be a switch with nothing behind it. True of
477
+ * the machinery, and it made the machinery the argument: an owner who pins a reasoning model to their commit
478
+ * subjects was paying that model's price to have its distinguishing feature suppressed. The knobs now travel
479
+ * through the one-shot path too, so an entry means the same thing wherever it is written.
455
480
  *
456
- * EVERY FIELD BUT THE PAIR IS OPTIONAL, AND ABSENT MEANS ABSENT: the turn goes out without the field and the
481
+ * EVERY FIELD BUT THE PAIR IS OPTIONAL, AND ABSENT MEANS ABSENT: the work goes out without the field and the
457
482
  * provider's own default answers, exactly as an unconfigured pin always did. Nothing here invents a "low".
458
483
  *
459
484
  * NO TIER HOLD, and its absence is the rule rather than an omission: automatic tier selection gates on
460
- * `unattended` (prompt-complexity.ts), so a surface-started run is never downgraded in the first place and a
461
- * veto over it would be a control whose state can make no difference to anything.
485
+ * `unattended` (prompt-complexity.ts), so a role-started run is never downgraded in the first place and a veto
486
+ * over it would be a control whose state can make no difference to anything.
462
487
  *
463
488
  * The pair is BOTH HALVES for the reason the pick above is: a model id is only meaningful to the provider that
464
489
  * vends it, so half a pin would send a Codex id to Claude. Taken verbatim, never validated against a catalog:
465
490
  * the picker offers a custom-id escape hatch, so a model this build has never heard of is a supported pin. */
466
- export const AgentRunPinSchema = z.object({
467
- provider: AgentProviderSchema.describe("Which provider serves the run."),
491
+ export const ModelPinSchema = z.object({
492
+ provider: AgentProviderSchema.describe("Which provider serves this work."),
468
493
  model: z.string().min(1).describe("Which of its models. Both halves, because a model name only means anything to the provider that serves it."),
469
494
  effort: z
470
495
  .string()
@@ -474,7 +499,7 @@ export const AgentRunPinSchema = z.object({
474
499
  fast: z.boolean().optional().describe("Ask for this model's work at a higher rate for a higher price. A request rather than a promise."),
475
500
  harness: AgentHarnessSchema.optional().describe("Which agentic loop runs it. Leave it out to use the provider's own."),
476
501
  });
477
- export type AgentRunPin = z.infer<typeof AgentRunPinSchema>;
502
+ export type ModelPin = z.infer<typeof ModelPinSchema>;
478
503
  // POST /agent's ack: the daemon-minted id of the detached turn run it started. The turn executes daemon-side
479
504
  // regardless of any client connection; every window, the initiator included, renders it via /agent/attach.
480
505
  export const StartedTurnSchema = z.object({
@@ -182,8 +182,8 @@ export const LandedMessageSchema = z.object({
182
182
  .describe("What this change takes away, for anything already relying on it. Nearly always absent: it is for removals, not for additions."),
183
183
  });
184
184
  export type LandedMessage = z.infer<typeof LandedMessageSchema>;
185
- /* ONE MODEL'S TURN IN THE DRAFTING WALK, asked, and what became of the ask. The quick-model chain tries the
186
- * connected models in order (agent/quick-model.ts), and each rung ends one of four ways:
185
+ /* ONE MODEL'S TURN IN THE DRAFTING WALK, asked, and what became of the ask. The one-shot helper chain tries the
186
+ * connected models in order (agent/role-model.ts), and each rung ends one of four ways:
187
187
  * asking , in flight right now; `ms` absent because it is still being spent.
188
188
  * answered, it wrote the sentence, in `ms`.
189
189
  * refused , it failed or declined, in `ms`, with its own words in `reason`.
@@ -58,6 +58,56 @@ export const AccountUsageSchema = z.object({
58
58
  measuredAt: z.number(),
59
59
  });
60
60
  export type AccountUsage = z.infer<typeof AccountUsageSchema>;
61
+ /* THE ONE WAY PAST A SPENT SESSION WINDOW THAT IS NOT WAITING, which Anthropic grants once a week per account.
62
+ *
63
+ * Every other affordance around a refused turn is about WHEN: arm the appointment, count down to the reset,
64
+ * press when it opens. This one moves the clock. The provider reopens the five-hour window immediately and
65
+ * charges the account one of its weekly resets; the WEEKLY allowance is untouched and still binds, so this
66
+ * buys back the session pool and nothing else. Upstream's own CLI spells it `/limit-reset`.
67
+ *
68
+ * THE ANSWER IS THE PROVIDER'S, NEVER OURS. There is no rule here to re-derive: eligibility turns on the plan
69
+ * tier, how long the account has existed, whether it is actually at the wall, whether another experiment holds
70
+ * it, and whether the week's reset is already spent — all of it decided server-side and none of it visible from
71
+ * a usage reading. So this shape is a transcription of what the endpoint said, and the button exists only while
72
+ * it says `available`. A client must never infer availability from a 100% window.
73
+ *
74
+ * `reason` is the provider's own word for the refusal ("tier", "tenure", "not_at_wall", "weekly_limit",
75
+ * "already_used", …) and is carried rather than translated, because the set is the provider's to extend and a
76
+ * word we don't recognise is still worth showing to somebody asking why the button is not there. */
77
+ export const LimitResetStatusSchema = z.object({
78
+ available: z
79
+ .boolean()
80
+ .describe("Whether the provider will reopen this account's session window right now. The only thing a button may be drawn from."),
81
+ reason: z
82
+ .string()
83
+ .optional()
84
+ .describe("Why not, in the provider's own word, when it gave one. Absent when it is available, or when the provider said nothing."),
85
+ // Both epoch SECONDS, matching every other reset instant on the wire (UsageWindow.resetsAt, limitResetsAt).
86
+ nextAvailableAt: z
87
+ .number()
88
+ .optional()
89
+ .describe("When the next reset may be claimed, in epoch seconds, where the provider publishes it. Absent means unknown, never 'now'."),
90
+ weeklyResetsAt: z.number().optional().describe("When the weekly allowance itself reopens, in epoch seconds, where the provider publishes it."),
91
+ });
92
+ export type LimitResetStatus = z.infer<typeof LimitResetStatusSchema>;
93
+ /* WHAT CLAIMING IT DID, in the provider's own vocabulary plus the two failures that are ours.
94
+ *
95
+ * `reset` is the only outcome that changed anything, and the caller's cue to send the held turn again. The rest
96
+ * are all "nothing happened", and they are kept APART rather than folded into one failure because they are read
97
+ * by somebody who just pressed a button and is owed the difference: `already_used` means come back next week,
98
+ * `not_limited` means the window reopened while they were reading, `ineligible` means this account never had
99
+ * it, and `unavailable`/`error` mean try again. Collapsing them would make every one of those read as a fault.
100
+ *
101
+ * Never throws over the wire: a claim that fails leaves the account exactly as it was, and the honest answer to
102
+ * a press is a word, not a stack trace. */
103
+ export const LimitResetClaimSchema = z.object({
104
+ result: z
105
+ .enum(["reset", "already_used", "not_limited", "ineligible", "unavailable", "error"])
106
+ .describe("What the provider did. Only `reset` reopened the window; every other value means nothing changed."),
107
+ nextAvailableAt: z.number().optional().describe("When another reset may be claimed, in epoch seconds, where the provider published it."),
108
+ detail: z.string().optional().describe("What went wrong, in words, for the two outcomes that are this sandbox's fault rather than the plan's."),
109
+ });
110
+ export type LimitResetClaim = z.infer<typeof LimitResetClaimSchema>;
61
111
  /* THE LAST TIME A PROVIDER ACTUALLY REFUSED A TURN, the other half of "can I run on this", and the half no
62
112
  * meter can supply.
63
113
  *