@intentic/sandbox-contract 1.246.1 → 1.248.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (106) hide show
  1. package/dist/batch-runs.d.ts +2 -0
  2. package/dist/batch-runs.d.ts.map +1 -1
  3. package/dist/batch-runs.js +1 -0
  4. package/dist/batch-runs.js.map +1 -1
  5. package/dist/contracts/agent.contract.d.ts +20 -0
  6. package/dist/contracts/agent.contract.d.ts.map +1 -1
  7. package/dist/contracts/personas.contract.d.ts +38 -4
  8. package/dist/contracts/personas.contract.d.ts.map +1 -1
  9. package/dist/contracts/personas.contract.js +10 -1
  10. package/dist/contracts/personas.contract.js.map +1 -1
  11. package/dist/contracts/runner.contract.d.ts +93 -93
  12. package/dist/contracts/settings.contract.d.ts +109 -12
  13. package/dist/contracts/settings.contract.d.ts.map +1 -1
  14. package/dist/contracts/usage.contract.d.ts +22 -0
  15. package/dist/contracts/usage.contract.d.ts.map +1 -1
  16. package/dist/contracts/usage.contract.js +19 -0
  17. package/dist/contracts/usage.contract.js.map +1 -1
  18. package/dist/definition.d.ts +20 -24
  19. package/dist/definition.d.ts.map +1 -1
  20. package/dist/fast-tier.js +1 -1
  21. package/dist/fast-tier.js.map +1 -1
  22. package/dist/index.d.ts +191 -19
  23. package/dist/index.d.ts.map +1 -1
  24. package/dist/index.js +2 -3
  25. package/dist/index.js.map +1 -1
  26. package/dist/model-pins.d.ts +15 -0
  27. package/dist/model-pins.d.ts.map +1 -0
  28. package/dist/model-pins.js +22 -0
  29. package/dist/model-pins.js.map +1 -0
  30. package/dist/model-roles.d.ts +172 -0
  31. package/dist/model-roles.d.ts.map +1 -0
  32. package/dist/model-roles.js +167 -0
  33. package/dist/model-roles.js.map +1 -0
  34. package/dist/schemas/agent.d.ts +22 -2
  35. package/dist/schemas/agent.d.ts.map +1 -1
  36. package/dist/schemas/agent.js +4 -2
  37. package/dist/schemas/agent.js.map +1 -1
  38. package/dist/schemas/automations.d.ts.map +1 -1
  39. package/dist/schemas/automations.js.map +1 -1
  40. package/dist/schemas/personas.d.ts +48 -4
  41. package/dist/schemas/personas.d.ts.map +1 -1
  42. package/dist/schemas/personas.js +28 -5
  43. package/dist/schemas/personas.js.map +1 -1
  44. package/dist/schemas/plan-limits.d.ts +20 -0
  45. package/dist/schemas/plan-limits.d.ts.map +1 -1
  46. package/dist/schemas/plan-limits.js +21 -0
  47. package/dist/schemas/plan-limits.js.map +1 -1
  48. package/dist/schemas/settings.d.ts +88 -6
  49. package/dist/schemas/settings.d.ts.map +1 -1
  50. package/dist/schemas/settings.js +19 -23
  51. package/dist/schemas/settings.js.map +1 -1
  52. package/dist/schemas/usage.d.ts +5 -0
  53. package/dist/schemas/usage.d.ts.map +1 -1
  54. package/dist/schemas/usage.js +5 -0
  55. package/dist/schemas/usage.js.map +1 -1
  56. package/dist/workspace-state.d.ts +0 -5
  57. package/dist/workspace-state.d.ts.map +1 -1
  58. package/dist/workspace-state.js +0 -1
  59. package/dist/workspace-state.js.map +1 -1
  60. package/package.json +4 -4
  61. package/src/agent-catalog.ts +1 -1
  62. package/src/batch-runs.test.ts +10 -5
  63. package/src/batch-runs.ts +10 -3
  64. package/src/chores/chores.ts +1 -1
  65. package/src/chores/verdict.test.ts +2 -2
  66. package/src/chores/verdict.ts +3 -3
  67. package/src/contracts/personas.contract.ts +17 -0
  68. package/src/contracts/usage.contract.ts +31 -0
  69. package/src/events.ts +1 -1
  70. package/src/fast-tier.test.ts +1 -1
  71. package/src/fast-tier.ts +5 -5
  72. package/src/index.ts +2 -3
  73. package/src/model-pins.test.ts +121 -0
  74. package/src/model-pins.ts +132 -0
  75. package/src/model-roles.test.ts +52 -0
  76. package/src/model-roles.ts +320 -0
  77. package/src/plan-pools.ts +1 -1
  78. package/src/prompt-complexity.test.ts +1 -1
  79. package/src/prompt-complexity.ts +2 -2
  80. package/src/provider-specs.test.ts +1 -1
  81. package/src/schemas/agent.ts +42 -17
  82. package/src/schemas/agents.ts +2 -2
  83. package/src/schemas/automations.ts +6 -2
  84. package/src/schemas/personas.ts +85 -18
  85. package/src/schemas/plan-limits.ts +50 -0
  86. package/src/schemas/settings.ts +92 -98
  87. package/src/schemas/usage.ts +59 -0
  88. package/src/workspace-state.test.ts +0 -1
  89. package/src/workspace-state.ts +0 -6
  90. package/dist/agent-run-model.d.ts +0 -4
  91. package/dist/agent-run-model.d.ts.map +0 -1
  92. package/dist/agent-run-model.js +0 -13
  93. package/dist/agent-run-model.js.map +0 -1
  94. package/dist/quick-model.d.ts +0 -15
  95. package/dist/quick-model.d.ts.map +0 -1
  96. package/dist/quick-model.js +0 -39
  97. package/dist/quick-model.js.map +0 -1
  98. package/dist/schemas/context.d.ts +0 -30
  99. package/dist/schemas/context.d.ts.map +0 -1
  100. package/dist/schemas/context.js +0 -34
  101. package/dist/schemas/context.js.map +0 -1
  102. package/src/agent-run-model.test.ts +0 -76
  103. package/src/agent-run-model.ts +0 -65
  104. package/src/quick-model.test.ts +0 -158
  105. package/src/quick-model.ts +0 -162
  106. package/src/schemas/context.ts +0 -87
@@ -0,0 +1,132 @@
1
+ import { modelsFor } from "./agent-catalog.js";
2
+ import type { AgentProvider, ModelPin } from "./schemas/agent.js";
3
+
4
+ /* WHICH MODELS A ROLE MAY RUN, IN THE ORDER TO TRY THEM. One resolver over every list in
5
+ * settings.modelRoles, and the browser and the daemon both read it.
6
+ *
7
+ * IT IS AN ORDER, NOT A MODEL, and that is the shape of every list this file answers for. A single pick is a
8
+ * single point of failure: the account it names spends its allowance on the chat all morning, and the role
9
+ * fails on a limit for the rest of the day while three other connected providers sit idle. So a setting is a
10
+ * LIST read top to bottom, this hands back the whole ladder, and the caller walks it until one answers. Nothing
11
+ * here decides WHICH failures are worth stepping over — only the runner has made the call and seen it fail —
12
+ * this side says what the running order is.
13
+ *
14
+ * THE RULE LIVES IN THE CONTRACT because both sides need the same answer for different jobs: the daemon runs
15
+ * the model, and the browser has to NAME it, in that job's settings row, before anything has run. Two
16
+ * implementations would drift precisely where it matters most, since a row promising Haiku while the daemon
17
+ * bills Opus is worse than no row at all.
18
+ *
19
+ * AN EMPTY LIST RESOLVES TO NOTHING, FOR EVERY ROLE, and this file no longer derives a floor for any of them.
20
+ * It used to: a `helper` role with no list got an "Auto ladder" worked out from whatever was connected —
21
+ * every provider's cheapest row, best-first — so an owner who had never opened the settings page still got
22
+ * commit messages, session titles and safety verdicts from a model this file picked. That was the wrong
23
+ * default, and the settings row saying "Auto: Gemini 3 Flash Lite, then Claude Haiku 4.5, then …" was the
24
+ * tell: a recommendation nobody asked for, over accounts they had connected for something else, changing
25
+ * under them whenever an account was added. NOT SET NOW MEANS NOT SET. Nothing is auto-selected and nothing
26
+ * is recommended: the owner names the models for a job or the job does not run, which is a state they can
27
+ * read off the row and a bill they cannot be surprised by.
28
+ *
29
+ * The two kinds of role (model-roles.ts) still differ in what the CALLER does with an empty answer — a
30
+ * `helper` is simply off, a `run` falls to the model the owner picked for their own chat — but that is the
31
+ * caller's business, and nothing here has to know which kind it is holding. */
32
+
33
+ /* One provider's standing in the decision: whether a turn on it can be sent at all, and what its catalog holds.
34
+ *
35
+ * ACP agents are deliberately not expressible here — an ACP row's model id is empty because the agent owns its
36
+ * own model, so there is no rung to point it at. `endpoint/<id>` providers ARE, and have to be: their models
37
+ * appear in the same picker the settings rows build their options from, so a pin naming one has to hold rather
38
+ * than drop out and leave the job running on an account the user was deliberately steering away from. */
39
+ export interface ModelSource {
40
+ // AgentProvider, not NativeProvider: an endpoint's id is user-created and cannot be in a fixed union, and
41
+ // what a turn on somebody's own model server costs is not a fact this repo can know — which is fine here,
42
+ // because nothing on this side ranks anything. A pin either names a provider that can run it or it does not.
43
+ readonly provider: AgentProvider;
44
+ // The same connection predicate every other surface gates on (access.ts web-side, the daemon's own account
45
+ // stores daemon-side). A catalog is never empty by construction, so "has rows" says nothing about "can send".
46
+ readonly ready: boolean;
47
+ // What the provider publishes. NOTHING IN THIS FILE READS IT any more: it was the input to the derived Auto
48
+ // ladder, and a pin is taken verbatim. Kept because it is what a source IS, and the daemon's helper walk
49
+ // still gathers it (role-model.ts); a caller with no catalog to hand passes an empty list and loses nothing.
50
+ readonly models: readonly string[];
51
+ }
52
+
53
+ export interface ModelChoice {
54
+ readonly provider: AgentProvider;
55
+ readonly model: string;
56
+ }
57
+
58
+ // A (provider, model) pair on the wire: `${provider}:${modelId}`, the same key shape the model picker mints for
59
+ // its entries (PickerEntry.key). Every role list stores PINS rather than these keys — an entry says how it runs
60
+ // as well as which model it is — but the key is still how two entries are compared, how a role list dedupes,
61
+ // and how `autoFastModels` (which pins no knobs, see its own note in settings.ts) is stored.
62
+ export const modelPinKey = (choice: ModelChoice): string => `${choice.provider}:${choice.model}`;
63
+
64
+ /* Split on the FIRST colon only: a provider id never contains one and a model id might. Exported because
65
+ * `autoFastModels` stores these keys, and because a session composed from a pin travels as one
66
+ * (composeSession). */
67
+ export const parsePinned = (pinned: string): ModelChoice | undefined => {
68
+ const separator = pinned.indexOf(`:`);
69
+ if (separator <= 0 || separator === pinned.length - 1) {
70
+ return undefined;
71
+ }
72
+ return { provider: pinned.slice(0, separator), model: pinned.slice(separator + 1) };
73
+ };
74
+
75
+ /* A pin as a person reads it: the catalog's own label for the id, or the id itself for one the static catalog
76
+ * has not caught up with (the picker offers a custom-id escape hatch, so this is a real case rather than a
77
+ * defensive branch). Beside parsePinned because the two are always wanted together, by any surface that has to
78
+ * name what a click is about to spend BEFORE it spends it, and the two loudest of those are extensions that
79
+ * share no other code with each other. */
80
+ export const pinnedModelLabel = (choice: ModelChoice): string =>
81
+ modelsFor(choice.provider).find((option) => option.value === choice.model)?.label ?? choice.model;
82
+
83
+ /* WHICH MODELS THIS ROLE MAY RUN, IN THE ORDER TO TRY THEM, given what this sandbox has connected.
84
+ * `pinned` is the stored setting: settings.modelRoles[role], an ordered list of pins.
85
+ *
86
+ * A pin only holds while its provider is READY: an account the user disconnected would otherwise sit at the
87
+ * head of the chain failing on a credential error while the sandbox can plainly still answer from the rung
88
+ * below. It stays on SCREEN, greyed — the settings row renders the stored list, not this one — because a
89
+ * setting that vanished from view would look like the app had eaten it.
90
+ *
91
+ * THE PINNED LIST IS THE WHOLE ANSWER, and there is nothing underneath it. A user who writes down three models
92
+ * has said which accounts this job may spend, and reaching for a fourth when all three are out is exactly the
93
+ * "spend an account they were steering away from" failure a pin exists to prevent. A user who writes down none
94
+ * has said the job picks no model at all.
95
+ *
96
+ * THE WHOLE PIN SURVIVES, not the pair inside it: an entry's effort, thinking, speed and harness are what the
97
+ * work is composed from, so a resolver handing back a bare (provider, model) would silently run the head of the
98
+ * list at the provider's defaults. Nothing here reads or judges those fields, which is the point of carrying
99
+ * them whole.
100
+ *
101
+ * EMPTY OUT MEANS THE LIST HAS NOTHING IT MAY REACH, from two different causes the caller can tell apart by
102
+ * looking at `pinned`: an empty list is an owner who set no model, and a full list that survives none of the
103
+ * readiness filter is an owner whose accounts have gone. The first is the job being switched off, the second
104
+ * is worth a sentence about the accounts.
105
+ *
106
+ * ONE FUNCTION FOR BOTH KINDS OF LADDER, and it is `resolveRoleModels` that stopped existing rather than this
107
+ * one arriving to replace it. A role's list used to add the role's own floor beneath the ready chain, which is
108
+ * the only thing it did that a persona card's list (schemas/personas.ts `personaModels`) did not — so the walk
109
+ * was split out to be shared. With the floor gone there is no difference left to share around: a role's list
110
+ * and a card's list are the same question over the same sources, and two names for it would be two places to
111
+ * read before believing they agree. */
112
+ export const readyChain = (sources: readonly ModelSource[], pinned: readonly ModelPin[]): readonly ModelPin[] => {
113
+ const ready = new Set(sources.filter((source) => source.ready).map((source) => source.provider));
114
+ // Taken verbatim, unvalidated against the catalog on purpose: the picker offers a custom-id escape hatch for
115
+ // a model a catalog hasn't caught up with, and second-guessing the user's own id here would silently run a
116
+ // different model than the settings row names.
117
+ const requested = pinned.filter((pin) => ready.has(pin.provider));
118
+ /* The same model twice would spend two attempts proving one account is out — a real state, since the list is
119
+ * hand-edited and the bulk editor writes one pin across many jobs.
120
+ *
121
+ * THE FIRST OF A PAIR WINS, WHOLE. Two entries can name one model and differ in their knobs (the same Sonnet
122
+ * at Max and again at Low, written while reordering the list), and the one the user reads first is the one
123
+ * they meant; keeping the earlier position with the later entry's effort would run a tier that appears
124
+ * nowhere the pin does. */
125
+ const chain: ModelPin[] = [];
126
+ for (const pin of requested) {
127
+ if (!chain.some((held) => modelPinKey(held) === modelPinKey(pin))) {
128
+ chain.push(pin);
129
+ }
130
+ }
131
+ return chain;
132
+ };
@@ -0,0 +1,52 @@
1
+ import { expect, test } from "vitest";
2
+ import { MODEL_ROLE_BLOCKS, MODEL_ROLES, type ModelRoleSpec } from "./model-roles.js";
3
+
4
+ /* THE CATALOG DRAWS THE SETTINGS PAGE, so the properties the page relies on have to be true of the TABLE rather
5
+ * than remembered by whoever last added a row. Eighteen jobs in one unbroken list is what the blocks exist to
6
+ * break up, and the failure they replace is a silent one: a role that belongs to no block is simply missing
7
+ * from Sandbox ▸ Agent ▸ Models, with no error anywhere and a page that looks completely normal. */
8
+
9
+ // The table at its declared width rather than as the literal tuple: `trigger` is absent from a helper's literal
10
+ // type, and what is under test is exactly whether it is absent where it should be.
11
+ const roles: readonly ModelRoleSpec[] = MODEL_ROLES;
12
+
13
+ /* WHAT STARTS A RUN IS THE BLOCK IT IS READ IN, and asserting the two against each other is what stops them
14
+ * drifting: a run whose trigger says one thing while the page draws it under the other heading is a row telling
15
+ * an owner that a session nobody is watching is one they start. A helper answers with its own kind, because it
16
+ * has no trigger and its block is the kind itself. */
17
+ test("every whole session says what starts it, and it is the block it is drawn in", () => {
18
+ const blockOf = new Map(MODEL_ROLE_BLOCKS.flatMap((block) => block.roles.map((role) => [role.id, block.id] as const)));
19
+
20
+ for (const role of roles) {
21
+ expect(role.kind === `run` ? role.trigger : role.kind, role.id).toBe(blockOf.get(role.id));
22
+ }
23
+ });
24
+
25
+ test("a one-shot declares no trigger at all: nothing presses a commit subject into being", () => {
26
+ for (const helper of roles.filter((role) => role.kind === `helper`)) {
27
+ expect(Object.keys(helper), helper.id).not.toContain(`trigger`);
28
+ }
29
+ });
30
+
31
+ /* THE BLOCKS PARTITION THE TABLE: every role in exactly one, nothing invented. Asserted as a partition rather
32
+ * than block by block, because the two ways to get this wrong are opposites and only one of them is visible —
33
+ * a role in no block vanishes from the page, a role in two is drawn twice and would be caught by eye. */
34
+ test("the blocks hold every role once, in the table's own order", () => {
35
+ const blocked = MODEL_ROLE_BLOCKS.flatMap((block) => block.roles.map((role) => role.id));
36
+
37
+ expect(blocked.toSorted()).toEqual(MODEL_ROLES.map((role) => role.id).toSorted());
38
+ /* AND THE ORDER SURVIVES THE SPLIT. Reading the blocks in order is reading the table in order: the table
39
+ * argues its sequence is REACH (jobs nobody picked a model for, then sessions a click starts, then sessions
40
+ * that start without one), and a block re-sorted here would leave that argument describing a page it no
41
+ * longer matches. */
42
+ expect(blocked).toEqual(MODEL_ROLES.map((role) => role.id));
43
+ });
44
+
45
+ test("each block says what it is, so the page keeps no headings of its own", () => {
46
+ for (const block of MODEL_ROLE_BLOCKS) {
47
+ expect(block.label, block.id).toMatch(/\S/);
48
+ expect(block.caption, block.id).toMatch(/\S/);
49
+ // A block with nothing in it would draw a heading over an empty surface.
50
+ expect(block.roles.length, block.id).toBeGreaterThan(0);
51
+ }
52
+ });
@@ -0,0 +1,320 @@
1
+ import { z } from "zod";
2
+
3
+ /* EVERY JOB IN THIS SANDBOX THAT PICKS A MODEL, LISTED BY THE JOB IT IS, one row per answer the owner is
4
+ * entitled to give differently.
5
+ *
6
+ * WHAT THIS REPLACED, and why it had to go. Model settings used to be grouped by how HARD the work was assumed
7
+ * to be: a "quick model" for the small automatic jobs, an "agent runs" tier for the big ones. Both names
8
+ * described an intensity rather than a job, and an intensity is a guess somebody else made about work the owner
9
+ * knows better. A commit message written by a frontier model is not a mistake, it is a preference, and the
10
+ * grouping made it unsayable: pinning Opus to get better commit subjects also pinned it to session titles, to
11
+ * every loop verdict, and to the safety judge. One bundled row could not say the thing anyone actually wanted
12
+ * to say.
13
+ *
14
+ * SO THE UNIT IS THE ROLE. Each entry here is one place a model gets chosen, and each gets its own ordered list
15
+ * in settings.modelRoles. Seventeen rows is more than four, and that is the point rather than a cost being
16
+ * absorbed: the configuration and the UI both already existed per use-case, and collapsing them was the only
17
+ * thing standing between the owner and a choice the machinery could already honour. A simpler face over this
18
+ * (presets: "cheap everywhere", "frontier everywhere") is a thing that can be built ON TOP of a true model, and
19
+ * cannot be unpicked from a lossy one.
20
+ *
21
+ * EVERY LIST IS AN ORDERED LADDER, whatever the role, and for one reason: the interesting failure of a pinned
22
+ * model is not that it is wrong, it is that it is CONNECTED AND WILL NOT ANSWER TODAY. The account's allowance
23
+ * went on the chat this morning, and one spent provider takes the role down for hours while three others sit
24
+ * idle. Written in order, the next entry catches it.
25
+ *
26
+ * AN EMPTY LIST IS THE JOB SWITCHED OFF, and NOTHING IS DERIVED FOR IT. A helper role used to fall to an
27
+ * "Auto ladder" worked out from whatever was connected — every provider's cheapest row, best-first — so a
28
+ * sandbox that had never been configured still spent somebody's account on commit messages and safety
29
+ * verdicts, on a recommendation this table invented, which re-ranked itself the day another account was
30
+ * connected. Not set now means not set: no auto-selection, no recommendation, and the owner names the models
31
+ * for a job or the job does not happen.
32
+ *
33
+ * TWO KINDS, and what separates them is what the CALLER does with that empty answer. It is declared per role
34
+ * because it is a property of the job:
35
+ *
36
+ * helper — a one-shot. One prompt, no tools, one string back, and it is over. With no list the job does not
37
+ * run at all: no commit subject is drafted, no session is renamed, no command is judged. Every one
38
+ * of them already had a road for "the model could not answer" (the derived title stands, the
39
+ * commit box stays empty, the gate falls to its standing rule), so an owner who wants none of them
40
+ * leaves the row empty and pays nothing.
41
+ *
42
+ * run — a whole session with tools and a worktree, started by a surface rather than by a person at a
43
+ * composer. With no list the caller's own floor answers: the model the owner picked for their chat.
44
+ * That is not this table recommending anything — it is a choice they already made, in front of
45
+ * them, on the composer they work in.
46
+ *
47
+ * EVERY ENTRY IS A FULL PIN (ModelPinSchema): which model, and how it runs — effort, thinking, speed, harness.
48
+ * The helper roles carry them too, which they did not use to: a one-shot ran with reasoning forcibly off, so
49
+ * pinning a reasoning model to commit messages bought the price of one and the behaviour of neither. The knobs
50
+ * now ride through the one-shot path (the daemon's role-model.ts), so an entry means what it says wherever it
51
+ * is written.
52
+ *
53
+ * ADDING A ROLE IS ONE ROW HERE. The settings key, the resolver's floor, the daemon's lookup and the settings
54
+ * page's row all read this table, so a job that starts picking a model tomorrow becomes configurable by saying
55
+ * what it is. That is the property the old grouping cost: a new surface inherited "agent runs" by being
56
+ * unattended, which is how a documentation sweep and a production incident came to share one tier. */
57
+
58
+ export const ModelRoleKindSchema = z.enum(["helper", "run"]);
59
+ export type ModelRoleKind = z.infer<typeof ModelRoleKindSchema>;
60
+
61
+ /* WHAT STARTS A RUN, declared for the run roles alone.
62
+ *
63
+ * It is not a wire value and never has been: it decides which BLOCK a job is read in, and the argument for
64
+ * reading them apart is the same one that separates a helper from a run — an owner holds a session nobody is
65
+ * watching to a different budget from one they are sitting in front of. The settings page used to make that
66
+ * argument in a comment over a thirteen-row list ("the ones somebody presses ahead of the ones that fire on
67
+ * their own"), which is a claim no reader can check and no row has to honour. Declared, it draws the page.
68
+ *
69
+ * A HELPER DECLARES NONE, and that is the honest answer rather than a gap: nobody presses "write me a commit
70
+ * subject". It happens because something else did. */
71
+ export type ModelRoleTrigger = "pressed" | "unprompted";
72
+
73
+ /* The shape of a row. `id` is a bare string HERE and narrowed on the exported type below, because the id union
74
+ * is derived from this very table: a self-referential `satisfies` would be a type that has to know its own
75
+ * answer before it can check it. */
76
+ interface ModelRoleRow {
77
+ readonly id: string;
78
+ // What the settings row is called. A JOB, in the owner's words, never a tier.
79
+ readonly label: string;
80
+ // The row's one line: what this model is asked to do. Read beside the label, so it says what the label
81
+ // cannot rather than restating it.
82
+ readonly blurb: string;
83
+ readonly kind: ModelRoleKind;
84
+ /** Runs only, and required for every one of them: see `ModelRoleTrigger`. */
85
+ readonly trigger?: ModelRoleTrigger;
86
+ // The row's glyph, from the shared icon set. Here rather than in a web-side map because the whole value of
87
+ // this table is that a role is declared ONCE; a second table keyed by the same ids is the drift this
88
+ // replaced, moved one layer up.
89
+ readonly icon: string;
90
+ }
91
+
92
+ /* THE TABLE. Ordered as the settings page draws it, and the order is an argument about reach: the one-shots
93
+ * first, because nobody chose a model for them and they run constantly; then the runs somebody's click starts;
94
+ * then the runs that start themselves, which are the ones an owner is least likely to be watching and most
95
+ * likely to want held to a budget. Those three are BLOCKS now (see MODEL_ROLE_BLOCKS) rather than an ordering
96
+ * this comment asserts and nothing holds to.
97
+ *
98
+ * The ids are the wire vocabulary: a turn carries one (AgentTurn.runRole), so they are kebab-case and stable,
99
+ * and renaming one is a breaking change to the setting rather than a cosmetic edit. */
100
+ export const MODEL_ROLES = [
101
+ {
102
+ id: "commit-message",
103
+ label: "Commit messages",
104
+ blurb: "The subject written when an agent's work lands, and the release note under it.",
105
+ kind: "helper",
106
+ icon: "file-edit",
107
+ },
108
+ {
109
+ id: "session-title",
110
+ label: "Session titles",
111
+ blurb: "The name a conversation wears on the board, written a second into its first turn.",
112
+ kind: "helper",
113
+ icon: "pencil",
114
+ },
115
+ {
116
+ /* THE ONE HELPER WHOSE INPUT IS ADVERSARIAL, and the reason the old bundling was worst here. Its prompt
117
+ * contains a command the agent is about to run, which may have arrived from a stranger's web page, and a
118
+ * small model can be talked round by it. Being wrong is expensive in both directions: a needless card
119
+ * teaches the owner to click through the next one. */
120
+ id: "safety-judge",
121
+ label: "Safety judge",
122
+ blurb: "Which model reads your safety policy before a flagged command runs.",
123
+ kind: "helper",
124
+ icon: "shield",
125
+ },
126
+ {
127
+ id: "loop-verdict",
128
+ label: "Loop verdicts",
129
+ blurb: "Whether a loop's iteration met the goal, or the loop goes round again.",
130
+ kind: "helper",
131
+ icon: "check-square",
132
+ },
133
+ {
134
+ /* THE ONE HELPER THAT ANSWERS A CLASSIFICATION rather than writing prose: one card id, or none, from the
135
+ * owner's own short list (schemas/personas.ts `brief`). Cheap by construction, once per chat, and the
136
+ * job a small model does well, which is the whole argument for routing chats onto static cards rather
137
+ * than asking a model to compose a context per session. */
138
+ id: "persona-router",
139
+ label: "Persona routing",
140
+ blurb: "Which model reads a new chat's first message and picks the persona for it.",
141
+ kind: "helper",
142
+ icon: "users",
143
+ },
144
+ {
145
+ id: "pipeline-fix",
146
+ label: "Pipeline fixes",
147
+ blurb: "The agent started by Fix on a red pipeline.",
148
+ kind: "run",
149
+ trigger: "pressed",
150
+ icon: "wave-pulse",
151
+ },
152
+ {
153
+ id: "deployment-fix",
154
+ label: "Deployment fixes",
155
+ blurb: "The agent started by Fix on a deployment that is down.",
156
+ kind: "run",
157
+ trigger: "pressed",
158
+ icon: "server",
159
+ },
160
+ {
161
+ id: "maintenance-chore",
162
+ label: "Maintenance chores",
163
+ blurb: "A chore run started from the Maintenance board.",
164
+ kind: "run",
165
+ trigger: "pressed",
166
+ icon: "wrench",
167
+ },
168
+ {
169
+ id: "documentation-run",
170
+ label: "Documentation runs",
171
+ blurb: "A pass over a repo's own documentation.",
172
+ kind: "run",
173
+ trigger: "pressed",
174
+ icon: "book",
175
+ },
176
+ {
177
+ id: "acceptance-run",
178
+ label: "Acceptance runs",
179
+ blurb: "One session per story in an acceptance fan-out.",
180
+ kind: "run",
181
+ trigger: "pressed",
182
+ icon: "list-check",
183
+ },
184
+ {
185
+ id: "pre-push-fix",
186
+ label: "Pre-push fixes",
187
+ blurb: "The fix proposed when a check fails on the way to a push.",
188
+ kind: "run",
189
+ trigger: "pressed",
190
+ icon: "cloud-upload",
191
+ },
192
+ {
193
+ /* PRESSED, ON THE STRENGTH OF THE APPROVAL. Nothing here runs until somebody reads the item and says
194
+ * yes, and that press is the start of this turn as much as Fix is the start of a pipeline run — the
195
+ * queue between the two is machinery, not a second decision. */
196
+ id: "approval-queue",
197
+ label: "Approvals queue",
198
+ blurb: "The turn that publishes or acts on what you approved.",
199
+ kind: "run",
200
+ trigger: "pressed",
201
+ icon: "check-circle",
202
+ },
203
+ {
204
+ id: "automation-wake",
205
+ label: "Automation wakes",
206
+ blurb: "A turn an automation fires: a schedule, a webhook, a message from outside.",
207
+ kind: "run",
208
+ trigger: "unprompted",
209
+ icon: "clock",
210
+ },
211
+ {
212
+ id: "extension-review",
213
+ label: "Extension update reviews",
214
+ blurb: "The agent that reads an extension update before it is applied.",
215
+ kind: "run",
216
+ trigger: "unprompted",
217
+ icon: "box",
218
+ },
219
+ {
220
+ id: "loop-iteration",
221
+ label: "Loop iterations",
222
+ blurb: "Each round of a loop working towards its goal.",
223
+ kind: "run",
224
+ trigger: "unprompted",
225
+ icon: "repeat",
226
+ },
227
+ {
228
+ id: "watch-wake",
229
+ label: "Watch wakes",
230
+ blurb: "The turn a watch starts when the thing it was watching happens.",
231
+ kind: "run",
232
+ trigger: "unprompted",
233
+ icon: "eye",
234
+ },
235
+ {
236
+ id: "verify-nudge",
237
+ label: "Verify nudges",
238
+ blurb: "The follow-up turn sent when work was left unverified.",
239
+ kind: "run",
240
+ trigger: "unprompted",
241
+ icon: "search",
242
+ },
243
+ {
244
+ /* THE ONE ROLE THAT ANSWERS A CHOICE RATHER THAN FILLING A SILENCE. A spawning agent may name its
245
+ * child's provider, and when it does that wins, exactly as a caret pick wins on every other run role.
246
+ * This is what answers when it names none, which used to be a hardcoded "claude". */
247
+ id: "child-agent",
248
+ label: "Child agents",
249
+ blurb: "What an agent's own subagents run on when it names no model for them.",
250
+ kind: "run",
251
+ trigger: "unprompted",
252
+ icon: "users",
253
+ },
254
+ ] as const satisfies readonly ModelRoleRow[];
255
+
256
+ export type ModelRole = (typeof MODEL_ROLES)[number]["id"];
257
+ export type ModelRoleSpec = ModelRoleRow & { readonly id: ModelRole };
258
+
259
+ export const MODEL_ROLE_IDS = MODEL_ROLES.map((role) => role.id) as readonly ModelRole[];
260
+
261
+ /* The wire form. An enum rather than a string, unlike most ids in this contract, because there is no case for
262
+ * an unknown one: a role is a place in THIS codebase where a model gets chosen, so a value outside the table
263
+ * names nothing, and a settings file or a turn carrying one is a typo worth a clean error rather than a list
264
+ * silently ignored. */
265
+ export const ModelRoleSchema = z.enum(MODEL_ROLE_IDS as [ModelRole, ...ModelRole[]]);
266
+
267
+ /* ═══ THE BLOCKS THE SETTINGS PAGE READS IN ═══
268
+ *
269
+ * EIGHTEEN JOBS IN ONE LIST IS A TABLE, NOT A PAGE. The unit being the role is right and is not in question —
270
+ * it is what lets an owner pin Opus to commit subjects without pinning it to every session title — but the cost
271
+ * lands on whoever opens the page: one unbroken run of eighteen rows, each with a name, a sentence, a control
272
+ * and a list under it, with no landmark to say where you are in it or which rows are like the one you came for.
273
+ *
274
+ * SO THE BLOCKS ARE DECLARED, AND THEY ARE THE DISTINCTIONS THE TABLE ALREADY MAKES. Nothing here is a fresh
275
+ * taxonomy invented for the layout: `kind` was always the difference between a one-shot and a whole session,
276
+ * and `trigger` is the sentence the old table wrote in a comment about its own ordering. A block is what those
277
+ * two answers already separate, given a name and a heading.
278
+ *
279
+ * THE LABELS LIVE HERE, beside the role labels, for the same reason those do: a heading kept in a web-side map
280
+ * keyed by the same ids is the drift a single table exists to stop. What is NOT here is anything about how the
281
+ * page draws them — that is the page's, and it changes on its own schedule.
282
+ *
283
+ * ORDER IS REACH, and it is the order of the table itself: jobs nobody picked a model for, then sessions a
284
+ * click of yours starts, then sessions that start without one. Every role belongs to exactly one block and
285
+ * every block keeps the table's order, which is what makes a role added upstairs appear on the page by
286
+ * existing. */
287
+ export type ModelRoleBlockId = "helper" | "pressed" | "unprompted";
288
+
289
+ export interface ModelRoleBlock {
290
+ readonly id: ModelRoleBlockId;
291
+ /** The group's heading on the settings page. */
292
+ readonly label: string;
293
+ /** One line beside it: what the jobs in this block have in common that the others do not. */
294
+ readonly caption: string;
295
+ /** Its roles, in table order. */
296
+ readonly roles: readonly ModelRoleSpec[];
297
+ }
298
+
299
+ const rolesWhere = (match: (role: ModelRoleSpec) => boolean): readonly ModelRoleSpec[] => MODEL_ROLES.filter((role) => match(role));
300
+
301
+ export const MODEL_ROLE_BLOCKS: readonly ModelRoleBlock[] = [
302
+ {
303
+ id: "helper",
304
+ label: "Automatic helpers",
305
+ caption: "One prompt, no tools, one answer back.",
306
+ roles: rolesWhere((role) => role.kind === "helper"),
307
+ },
308
+ {
309
+ id: "pressed",
310
+ label: "Runs you start",
311
+ caption: "Whole sessions, begun by a click of yours.",
312
+ roles: rolesWhere((role) => role.kind === "run" && role.trigger === "pressed"),
313
+ },
314
+ {
315
+ id: "unprompted",
316
+ label: "Runs that start themselves",
317
+ caption: "Whole sessions nobody pressed for.",
318
+ roles: rolesWhere((role) => role.kind === "run" && role.trigger === "unprompted"),
319
+ },
320
+ ];
package/src/plan-pools.ts CHANGED
@@ -6,7 +6,7 @@ import type { AccountUsage, UsageWindow, WindowGates } from "./schemas/plan-limi
6
6
  * call, is this Google fleet spent for Claude Opus, when does the pool that refused this turn reopen, what
7
7
  * does the ring beside the composer measure. The pools a reading carries answer that only through their
8
8
  * `gates` (UsageWindowSchema says why the reader decides them), and this file is the one place the gate is
9
- * read, so the daemon's account picker, its quick-model walk, its refusal dressing and the browser's rings,
9
+ * read, so the daemon's account picker, its one-shot helper walk, its refusal dressing and the browser's rings,
10
10
  * rail and picker rows all agree about which pool is binding for a given model.
11
11
  *
12
12
  * WITHOUT A MODEL the answer is the account's own tightest pool, which is what a roster or a rail that has not
@@ -88,7 +88,7 @@ test("plan mode is a request to think, so it is never answered by the model that
88
88
  });
89
89
 
90
90
  test("a surface-started run is never downgraded, because nobody is watching it fail", () => {
91
- // Same call agentRunModels already makes in the other direction: a run billed whole, with a worktree in
91
+ // Same call a `run` role already makes in the other direction: a run billed whole, with a worktree in
92
92
  // it, is not the place to spend a guess.
93
93
  expect(tierOf(`what is a closure?`, { unattended: true })).toBe(`standard`);
94
94
  });
@@ -3,7 +3,7 @@
3
3
  * The job is narrow on purpose: decide whether this turn could have run on the cheap rung of the provider the
4
4
  * user is already on. Nothing here picks a model, nothing here reads a catalog, and nothing here calls
5
5
  * anything. It is a pure function over the turn's own words and shape, so the daemon and the composer can both
6
- * ask it and get the same answer, which is the same reason quick-model.ts lives in the contract rather than in
6
+ * ask it and get the same answer, which is the same reason model-pins.ts lives in the contract rather than in
7
7
  * either of them.
8
8
  *
9
9
  * IT CAN ONLY EVER ROUTE DOWN. The standard tier is not a setting: it is whatever the user already picked. So
@@ -213,7 +213,7 @@ const forcing = (input: ComplexityInput, text: string): ComplexityRule[] => {
213
213
  rules.push("plan-mode");
214
214
  }
215
215
  /* An unattended run is billed whole and nobody is watching it fail. The settings this repo already ships
216
- * make the same call in the other direction: agentRunModels resolves to NOTHING when empty precisely
216
+ * make the same call in the other direction: a `run` role resolves to NOTHING when empty precisely
217
217
  * because "nothing here can judge whether a job is worth the frontier tier". This file does judge, but not
218
218
  * for the runs where a wrong guess costs a whole session with a worktree in it. */
219
219
  if (input.unattended) {
@@ -115,7 +115,7 @@ test("provider ids are unique", () => {
115
115
 
116
116
  /* An id may contain neither a slash nor a colon, and both exclusions are load-bearing rather than tidy.
117
117
  * `endpoint/<id>` uses the slash to namespace a capability-minted provider, and the picker's pinned selections
118
- * are `${provider}:${model}` split on the FIRST colon (quick-model.ts), so an id carrying either would parse as
118
+ * are `${provider}:${model}` split on the FIRST colon (model-pins.ts), so an id carrying either would parse as
119
119
  * something else entirely, silently. */
120
120
  test("no provider id can be mistaken for an endpoint or a pinned selection", () => {
121
121
  for (const spec of PROVIDER_SPECS) {