@intentic/sandbox-contract 1.246.1 → 1.248.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/batch-runs.d.ts +2 -0
- package/dist/batch-runs.d.ts.map +1 -1
- package/dist/batch-runs.js +1 -0
- package/dist/batch-runs.js.map +1 -1
- package/dist/contracts/agent.contract.d.ts +20 -0
- package/dist/contracts/agent.contract.d.ts.map +1 -1
- package/dist/contracts/personas.contract.d.ts +38 -4
- package/dist/contracts/personas.contract.d.ts.map +1 -1
- package/dist/contracts/personas.contract.js +10 -1
- package/dist/contracts/personas.contract.js.map +1 -1
- package/dist/contracts/runner.contract.d.ts +93 -93
- package/dist/contracts/settings.contract.d.ts +109 -12
- package/dist/contracts/settings.contract.d.ts.map +1 -1
- package/dist/contracts/usage.contract.d.ts +22 -0
- package/dist/contracts/usage.contract.d.ts.map +1 -1
- package/dist/contracts/usage.contract.js +19 -0
- package/dist/contracts/usage.contract.js.map +1 -1
- package/dist/definition.d.ts +20 -24
- package/dist/definition.d.ts.map +1 -1
- package/dist/fast-tier.js +1 -1
- package/dist/fast-tier.js.map +1 -1
- package/dist/index.d.ts +191 -19
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +2 -3
- package/dist/index.js.map +1 -1
- package/dist/model-pins.d.ts +15 -0
- package/dist/model-pins.d.ts.map +1 -0
- package/dist/model-pins.js +22 -0
- package/dist/model-pins.js.map +1 -0
- package/dist/model-roles.d.ts +172 -0
- package/dist/model-roles.d.ts.map +1 -0
- package/dist/model-roles.js +167 -0
- package/dist/model-roles.js.map +1 -0
- package/dist/schemas/agent.d.ts +22 -2
- package/dist/schemas/agent.d.ts.map +1 -1
- package/dist/schemas/agent.js +4 -2
- package/dist/schemas/agent.js.map +1 -1
- package/dist/schemas/automations.d.ts.map +1 -1
- package/dist/schemas/automations.js.map +1 -1
- package/dist/schemas/personas.d.ts +48 -4
- package/dist/schemas/personas.d.ts.map +1 -1
- package/dist/schemas/personas.js +28 -5
- package/dist/schemas/personas.js.map +1 -1
- package/dist/schemas/plan-limits.d.ts +20 -0
- package/dist/schemas/plan-limits.d.ts.map +1 -1
- package/dist/schemas/plan-limits.js +21 -0
- package/dist/schemas/plan-limits.js.map +1 -1
- package/dist/schemas/settings.d.ts +88 -6
- package/dist/schemas/settings.d.ts.map +1 -1
- package/dist/schemas/settings.js +19 -23
- package/dist/schemas/settings.js.map +1 -1
- package/dist/schemas/usage.d.ts +5 -0
- package/dist/schemas/usage.d.ts.map +1 -1
- package/dist/schemas/usage.js +5 -0
- package/dist/schemas/usage.js.map +1 -1
- package/dist/workspace-state.d.ts +0 -5
- package/dist/workspace-state.d.ts.map +1 -1
- package/dist/workspace-state.js +0 -1
- package/dist/workspace-state.js.map +1 -1
- package/package.json +4 -4
- package/src/agent-catalog.ts +1 -1
- package/src/batch-runs.test.ts +10 -5
- package/src/batch-runs.ts +10 -3
- package/src/chores/chores.ts +1 -1
- package/src/chores/verdict.test.ts +2 -2
- package/src/chores/verdict.ts +3 -3
- package/src/contracts/personas.contract.ts +17 -0
- package/src/contracts/usage.contract.ts +31 -0
- package/src/events.ts +1 -1
- package/src/fast-tier.test.ts +1 -1
- package/src/fast-tier.ts +5 -5
- package/src/index.ts +2 -3
- package/src/model-pins.test.ts +121 -0
- package/src/model-pins.ts +132 -0
- package/src/model-roles.test.ts +52 -0
- package/src/model-roles.ts +320 -0
- package/src/plan-pools.ts +1 -1
- package/src/prompt-complexity.test.ts +1 -1
- package/src/prompt-complexity.ts +2 -2
- package/src/provider-specs.test.ts +1 -1
- package/src/schemas/agent.ts +42 -17
- package/src/schemas/agents.ts +2 -2
- package/src/schemas/automations.ts +6 -2
- package/src/schemas/personas.ts +85 -18
- package/src/schemas/plan-limits.ts +50 -0
- package/src/schemas/settings.ts +92 -98
- package/src/schemas/usage.ts +59 -0
- package/src/workspace-state.test.ts +0 -1
- package/src/workspace-state.ts +0 -6
- package/dist/agent-run-model.d.ts +0 -4
- package/dist/agent-run-model.d.ts.map +0 -1
- package/dist/agent-run-model.js +0 -13
- package/dist/agent-run-model.js.map +0 -1
- package/dist/quick-model.d.ts +0 -15
- package/dist/quick-model.d.ts.map +0 -1
- package/dist/quick-model.js +0 -39
- package/dist/quick-model.js.map +0 -1
- package/dist/schemas/context.d.ts +0 -30
- package/dist/schemas/context.d.ts.map +0 -1
- package/dist/schemas/context.js +0 -34
- package/dist/schemas/context.js.map +0 -1
- package/src/agent-run-model.test.ts +0 -76
- package/src/agent-run-model.ts +0 -65
- package/src/quick-model.test.ts +0 -158
- package/src/quick-model.ts +0 -162
- package/src/schemas/context.ts +0 -87
|
@@ -0,0 +1,132 @@
|
|
|
1
|
+
import { modelsFor } from "./agent-catalog.js";
|
|
2
|
+
import type { AgentProvider, ModelPin } from "./schemas/agent.js";
|
|
3
|
+
|
|
4
|
+
/* WHICH MODELS A ROLE MAY RUN, IN THE ORDER TO TRY THEM. One resolver over every list in
|
|
5
|
+
* settings.modelRoles, and the browser and the daemon both read it.
|
|
6
|
+
*
|
|
7
|
+
* IT IS AN ORDER, NOT A MODEL, and that is the shape of every list this file answers for. A single pick is a
|
|
8
|
+
* single point of failure: the account it names spends its allowance on the chat all morning, and the role
|
|
9
|
+
* fails on a limit for the rest of the day while three other connected providers sit idle. So a setting is a
|
|
10
|
+
* LIST read top to bottom, this hands back the whole ladder, and the caller walks it until one answers. Nothing
|
|
11
|
+
* here decides WHICH failures are worth stepping over — only the runner has made the call and seen it fail —
|
|
12
|
+
* this side says what the running order is.
|
|
13
|
+
*
|
|
14
|
+
* THE RULE LIVES IN THE CONTRACT because both sides need the same answer for different jobs: the daemon runs
|
|
15
|
+
* the model, and the browser has to NAME it, in that job's settings row, before anything has run. Two
|
|
16
|
+
* implementations would drift precisely where it matters most, since a row promising Haiku while the daemon
|
|
17
|
+
* bills Opus is worse than no row at all.
|
|
18
|
+
*
|
|
19
|
+
* AN EMPTY LIST RESOLVES TO NOTHING, FOR EVERY ROLE, and this file no longer derives a floor for any of them.
|
|
20
|
+
* It used to: a `helper` role with no list got an "Auto ladder" worked out from whatever was connected —
|
|
21
|
+
* every provider's cheapest row, best-first — so an owner who had never opened the settings page still got
|
|
22
|
+
* commit messages, session titles and safety verdicts from a model this file picked. That was the wrong
|
|
23
|
+
* default, and the settings row saying "Auto: Gemini 3 Flash Lite, then Claude Haiku 4.5, then …" was the
|
|
24
|
+
* tell: a recommendation nobody asked for, over accounts they had connected for something else, changing
|
|
25
|
+
* under them whenever an account was added. NOT SET NOW MEANS NOT SET. Nothing is auto-selected and nothing
|
|
26
|
+
* is recommended: the owner names the models for a job or the job does not run, which is a state they can
|
|
27
|
+
* read off the row and a bill they cannot be surprised by.
|
|
28
|
+
*
|
|
29
|
+
* The two kinds of role (model-roles.ts) still differ in what the CALLER does with an empty answer — a
|
|
30
|
+
* `helper` is simply off, a `run` falls to the model the owner picked for their own chat — but that is the
|
|
31
|
+
* caller's business, and nothing here has to know which kind it is holding. */
|
|
32
|
+
|
|
33
|
+
/* One provider's standing in the decision: whether a turn on it can be sent at all, and what its catalog holds.
|
|
34
|
+
*
|
|
35
|
+
* ACP agents are deliberately not expressible here — an ACP row's model id is empty because the agent owns its
|
|
36
|
+
* own model, so there is no rung to point it at. `endpoint/<id>` providers ARE, and have to be: their models
|
|
37
|
+
* appear in the same picker the settings rows build their options from, so a pin naming one has to hold rather
|
|
38
|
+
* than drop out and leave the job running on an account the user was deliberately steering away from. */
|
|
39
|
+
export interface ModelSource {
|
|
40
|
+
// AgentProvider, not NativeProvider: an endpoint's id is user-created and cannot be in a fixed union, and
|
|
41
|
+
// what a turn on somebody's own model server costs is not a fact this repo can know — which is fine here,
|
|
42
|
+
// because nothing on this side ranks anything. A pin either names a provider that can run it or it does not.
|
|
43
|
+
readonly provider: AgentProvider;
|
|
44
|
+
// The same connection predicate every other surface gates on (access.ts web-side, the daemon's own account
|
|
45
|
+
// stores daemon-side). A catalog is never empty by construction, so "has rows" says nothing about "can send".
|
|
46
|
+
readonly ready: boolean;
|
|
47
|
+
// What the provider publishes. NOTHING IN THIS FILE READS IT any more: it was the input to the derived Auto
|
|
48
|
+
// ladder, and a pin is taken verbatim. Kept because it is what a source IS, and the daemon's helper walk
|
|
49
|
+
// still gathers it (role-model.ts); a caller with no catalog to hand passes an empty list and loses nothing.
|
|
50
|
+
readonly models: readonly string[];
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
export interface ModelChoice {
|
|
54
|
+
readonly provider: AgentProvider;
|
|
55
|
+
readonly model: string;
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
// A (provider, model) pair on the wire: `${provider}:${modelId}`, the same key shape the model picker mints for
|
|
59
|
+
// its entries (PickerEntry.key). Every role list stores PINS rather than these keys — an entry says how it runs
|
|
60
|
+
// as well as which model it is — but the key is still how two entries are compared, how a role list dedupes,
|
|
61
|
+
// and how `autoFastModels` (which pins no knobs, see its own note in settings.ts) is stored.
|
|
62
|
+
export const modelPinKey = (choice: ModelChoice): string => `${choice.provider}:${choice.model}`;
|
|
63
|
+
|
|
64
|
+
/* Split on the FIRST colon only: a provider id never contains one and a model id might. Exported because
|
|
65
|
+
* `autoFastModels` stores these keys, and because a session composed from a pin travels as one
|
|
66
|
+
* (composeSession). */
|
|
67
|
+
export const parsePinned = (pinned: string): ModelChoice | undefined => {
|
|
68
|
+
const separator = pinned.indexOf(`:`);
|
|
69
|
+
if (separator <= 0 || separator === pinned.length - 1) {
|
|
70
|
+
return undefined;
|
|
71
|
+
}
|
|
72
|
+
return { provider: pinned.slice(0, separator), model: pinned.slice(separator + 1) };
|
|
73
|
+
};
|
|
74
|
+
|
|
75
|
+
/* A pin as a person reads it: the catalog's own label for the id, or the id itself for one the static catalog
|
|
76
|
+
* has not caught up with (the picker offers a custom-id escape hatch, so this is a real case rather than a
|
|
77
|
+
* defensive branch). Beside parsePinned because the two are always wanted together, by any surface that has to
|
|
78
|
+
* name what a click is about to spend BEFORE it spends it, and the two loudest of those are extensions that
|
|
79
|
+
* share no other code with each other. */
|
|
80
|
+
export const pinnedModelLabel = (choice: ModelChoice): string =>
|
|
81
|
+
modelsFor(choice.provider).find((option) => option.value === choice.model)?.label ?? choice.model;
|
|
82
|
+
|
|
83
|
+
/* WHICH MODELS THIS ROLE MAY RUN, IN THE ORDER TO TRY THEM, given what this sandbox has connected.
|
|
84
|
+
* `pinned` is the stored setting: settings.modelRoles[role], an ordered list of pins.
|
|
85
|
+
*
|
|
86
|
+
* A pin only holds while its provider is READY: an account the user disconnected would otherwise sit at the
|
|
87
|
+
* head of the chain failing on a credential error while the sandbox can plainly still answer from the rung
|
|
88
|
+
* below. It stays on SCREEN, greyed — the settings row renders the stored list, not this one — because a
|
|
89
|
+
* setting that vanished from view would look like the app had eaten it.
|
|
90
|
+
*
|
|
91
|
+
* THE PINNED LIST IS THE WHOLE ANSWER, and there is nothing underneath it. A user who writes down three models
|
|
92
|
+
* has said which accounts this job may spend, and reaching for a fourth when all three are out is exactly the
|
|
93
|
+
* "spend an account they were steering away from" failure a pin exists to prevent. A user who writes down none
|
|
94
|
+
* has said the job picks no model at all.
|
|
95
|
+
*
|
|
96
|
+
* THE WHOLE PIN SURVIVES, not the pair inside it: an entry's effort, thinking, speed and harness are what the
|
|
97
|
+
* work is composed from, so a resolver handing back a bare (provider, model) would silently run the head of the
|
|
98
|
+
* list at the provider's defaults. Nothing here reads or judges those fields, which is the point of carrying
|
|
99
|
+
* them whole.
|
|
100
|
+
*
|
|
101
|
+
* EMPTY OUT MEANS THE LIST HAS NOTHING IT MAY REACH, from two different causes the caller can tell apart by
|
|
102
|
+
* looking at `pinned`: an empty list is an owner who set no model, and a full list that survives none of the
|
|
103
|
+
* readiness filter is an owner whose accounts have gone. The first is the job being switched off, the second
|
|
104
|
+
* is worth a sentence about the accounts.
|
|
105
|
+
*
|
|
106
|
+
* ONE FUNCTION FOR BOTH KINDS OF LADDER, and it is `resolveRoleModels` that stopped existing rather than this
|
|
107
|
+
* one arriving to replace it. A role's list used to add the role's own floor beneath the ready chain, which is
|
|
108
|
+
* the only thing it did that a persona card's list (schemas/personas.ts `personaModels`) did not — so the walk
|
|
109
|
+
* was split out to be shared. With the floor gone there is no difference left to share around: a role's list
|
|
110
|
+
* and a card's list are the same question over the same sources, and two names for it would be two places to
|
|
111
|
+
* read before believing they agree. */
|
|
112
|
+
export const readyChain = (sources: readonly ModelSource[], pinned: readonly ModelPin[]): readonly ModelPin[] => {
|
|
113
|
+
const ready = new Set(sources.filter((source) => source.ready).map((source) => source.provider));
|
|
114
|
+
// Taken verbatim, unvalidated against the catalog on purpose: the picker offers a custom-id escape hatch for
|
|
115
|
+
// a model a catalog hasn't caught up with, and second-guessing the user's own id here would silently run a
|
|
116
|
+
// different model than the settings row names.
|
|
117
|
+
const requested = pinned.filter((pin) => ready.has(pin.provider));
|
|
118
|
+
/* The same model twice would spend two attempts proving one account is out — a real state, since the list is
|
|
119
|
+
* hand-edited and the bulk editor writes one pin across many jobs.
|
|
120
|
+
*
|
|
121
|
+
* THE FIRST OF A PAIR WINS, WHOLE. Two entries can name one model and differ in their knobs (the same Sonnet
|
|
122
|
+
* at Max and again at Low, written while reordering the list), and the one the user reads first is the one
|
|
123
|
+
* they meant; keeping the earlier position with the later entry's effort would run a tier that appears
|
|
124
|
+
* nowhere the pin does. */
|
|
125
|
+
const chain: ModelPin[] = [];
|
|
126
|
+
for (const pin of requested) {
|
|
127
|
+
if (!chain.some((held) => modelPinKey(held) === modelPinKey(pin))) {
|
|
128
|
+
chain.push(pin);
|
|
129
|
+
}
|
|
130
|
+
}
|
|
131
|
+
return chain;
|
|
132
|
+
};
|
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
import { expect, test } from "vitest";
|
|
2
|
+
import { MODEL_ROLE_BLOCKS, MODEL_ROLES, type ModelRoleSpec } from "./model-roles.js";
|
|
3
|
+
|
|
4
|
+
/* THE CATALOG DRAWS THE SETTINGS PAGE, so the properties the page relies on have to be true of the TABLE rather
|
|
5
|
+
* than remembered by whoever last added a row. Eighteen jobs in one unbroken list is what the blocks exist to
|
|
6
|
+
* break up, and the failure they replace is a silent one: a role that belongs to no block is simply missing
|
|
7
|
+
* from Sandbox ▸ Agent ▸ Models, with no error anywhere and a page that looks completely normal. */
|
|
8
|
+
|
|
9
|
+
// The table at its declared width rather than as the literal tuple: `trigger` is absent from a helper's literal
|
|
10
|
+
// type, and what is under test is exactly whether it is absent where it should be.
|
|
11
|
+
const roles: readonly ModelRoleSpec[] = MODEL_ROLES;
|
|
12
|
+
|
|
13
|
+
/* WHAT STARTS A RUN IS THE BLOCK IT IS READ IN, and asserting the two against each other is what stops them
|
|
14
|
+
* drifting: a run whose trigger says one thing while the page draws it under the other heading is a row telling
|
|
15
|
+
* an owner that a session nobody is watching is one they start. A helper answers with its own kind, because it
|
|
16
|
+
* has no trigger and its block is the kind itself. */
|
|
17
|
+
test("every whole session says what starts it, and it is the block it is drawn in", () => {
|
|
18
|
+
const blockOf = new Map(MODEL_ROLE_BLOCKS.flatMap((block) => block.roles.map((role) => [role.id, block.id] as const)));
|
|
19
|
+
|
|
20
|
+
for (const role of roles) {
|
|
21
|
+
expect(role.kind === `run` ? role.trigger : role.kind, role.id).toBe(blockOf.get(role.id));
|
|
22
|
+
}
|
|
23
|
+
});
|
|
24
|
+
|
|
25
|
+
test("a one-shot declares no trigger at all: nothing presses a commit subject into being", () => {
|
|
26
|
+
for (const helper of roles.filter((role) => role.kind === `helper`)) {
|
|
27
|
+
expect(Object.keys(helper), helper.id).not.toContain(`trigger`);
|
|
28
|
+
}
|
|
29
|
+
});
|
|
30
|
+
|
|
31
|
+
/* THE BLOCKS PARTITION THE TABLE: every role in exactly one, nothing invented. Asserted as a partition rather
|
|
32
|
+
* than block by block, because the two ways to get this wrong are opposites and only one of them is visible —
|
|
33
|
+
* a role in no block vanishes from the page, a role in two is drawn twice and would be caught by eye. */
|
|
34
|
+
test("the blocks hold every role once, in the table's own order", () => {
|
|
35
|
+
const blocked = MODEL_ROLE_BLOCKS.flatMap((block) => block.roles.map((role) => role.id));
|
|
36
|
+
|
|
37
|
+
expect(blocked.toSorted()).toEqual(MODEL_ROLES.map((role) => role.id).toSorted());
|
|
38
|
+
/* AND THE ORDER SURVIVES THE SPLIT. Reading the blocks in order is reading the table in order: the table
|
|
39
|
+
* argues its sequence is REACH (jobs nobody picked a model for, then sessions a click starts, then sessions
|
|
40
|
+
* that start without one), and a block re-sorted here would leave that argument describing a page it no
|
|
41
|
+
* longer matches. */
|
|
42
|
+
expect(blocked).toEqual(MODEL_ROLES.map((role) => role.id));
|
|
43
|
+
});
|
|
44
|
+
|
|
45
|
+
test("each block says what it is, so the page keeps no headings of its own", () => {
|
|
46
|
+
for (const block of MODEL_ROLE_BLOCKS) {
|
|
47
|
+
expect(block.label, block.id).toMatch(/\S/);
|
|
48
|
+
expect(block.caption, block.id).toMatch(/\S/);
|
|
49
|
+
// A block with nothing in it would draw a heading over an empty surface.
|
|
50
|
+
expect(block.roles.length, block.id).toBeGreaterThan(0);
|
|
51
|
+
}
|
|
52
|
+
});
|
|
@@ -0,0 +1,320 @@
|
|
|
1
|
+
import { z } from "zod";
|
|
2
|
+
|
|
3
|
+
/* EVERY JOB IN THIS SANDBOX THAT PICKS A MODEL, LISTED BY THE JOB IT IS, one row per answer the owner is
|
|
4
|
+
* entitled to give differently.
|
|
5
|
+
*
|
|
6
|
+
* WHAT THIS REPLACED, and why it had to go. Model settings used to be grouped by how HARD the work was assumed
|
|
7
|
+
* to be: a "quick model" for the small automatic jobs, an "agent runs" tier for the big ones. Both names
|
|
8
|
+
* described an intensity rather than a job, and an intensity is a guess somebody else made about work the owner
|
|
9
|
+
* knows better. A commit message written by a frontier model is not a mistake, it is a preference, and the
|
|
10
|
+
* grouping made it unsayable: pinning Opus to get better commit subjects also pinned it to session titles, to
|
|
11
|
+
* every loop verdict, and to the safety judge. One bundled row could not say the thing anyone actually wanted
|
|
12
|
+
* to say.
|
|
13
|
+
*
|
|
14
|
+
* SO THE UNIT IS THE ROLE. Each entry here is one place a model gets chosen, and each gets its own ordered list
|
|
15
|
+
* in settings.modelRoles. Seventeen rows is more than four, and that is the point rather than a cost being
|
|
16
|
+
* absorbed: the configuration and the UI both already existed per use-case, and collapsing them was the only
|
|
17
|
+
* thing standing between the owner and a choice the machinery could already honour. A simpler face over this
|
|
18
|
+
* (presets: "cheap everywhere", "frontier everywhere") is a thing that can be built ON TOP of a true model, and
|
|
19
|
+
* cannot be unpicked from a lossy one.
|
|
20
|
+
*
|
|
21
|
+
* EVERY LIST IS AN ORDERED LADDER, whatever the role, and for one reason: the interesting failure of a pinned
|
|
22
|
+
* model is not that it is wrong, it is that it is CONNECTED AND WILL NOT ANSWER TODAY. The account's allowance
|
|
23
|
+
* went on the chat this morning, and one spent provider takes the role down for hours while three others sit
|
|
24
|
+
* idle. Written in order, the next entry catches it.
|
|
25
|
+
*
|
|
26
|
+
* AN EMPTY LIST IS THE JOB SWITCHED OFF, and NOTHING IS DERIVED FOR IT. A helper role used to fall to an
|
|
27
|
+
* "Auto ladder" worked out from whatever was connected — every provider's cheapest row, best-first — so a
|
|
28
|
+
* sandbox that had never been configured still spent somebody's account on commit messages and safety
|
|
29
|
+
* verdicts, on a recommendation this table invented, which re-ranked itself the day another account was
|
|
30
|
+
* connected. Not set now means not set: no auto-selection, no recommendation, and the owner names the models
|
|
31
|
+
* for a job or the job does not happen.
|
|
32
|
+
*
|
|
33
|
+
* TWO KINDS, and what separates them is what the CALLER does with that empty answer. It is declared per role
|
|
34
|
+
* because it is a property of the job:
|
|
35
|
+
*
|
|
36
|
+
* helper — a one-shot. One prompt, no tools, one string back, and it is over. With no list the job does not
|
|
37
|
+
* run at all: no commit subject is drafted, no session is renamed, no command is judged. Every one
|
|
38
|
+
* of them already had a road for "the model could not answer" (the derived title stands, the
|
|
39
|
+
* commit box stays empty, the gate falls to its standing rule), so an owner who wants none of them
|
|
40
|
+
* leaves the row empty and pays nothing.
|
|
41
|
+
*
|
|
42
|
+
* run — a whole session with tools and a worktree, started by a surface rather than by a person at a
|
|
43
|
+
* composer. With no list the caller's own floor answers: the model the owner picked for their chat.
|
|
44
|
+
* That is not this table recommending anything — it is a choice they already made, in front of
|
|
45
|
+
* them, on the composer they work in.
|
|
46
|
+
*
|
|
47
|
+
* EVERY ENTRY IS A FULL PIN (ModelPinSchema): which model, and how it runs — effort, thinking, speed, harness.
|
|
48
|
+
* The helper roles carry them too, which they did not use to: a one-shot ran with reasoning forcibly off, so
|
|
49
|
+
* pinning a reasoning model to commit messages bought the price of one and the behaviour of neither. The knobs
|
|
50
|
+
* now ride through the one-shot path (the daemon's role-model.ts), so an entry means what it says wherever it
|
|
51
|
+
* is written.
|
|
52
|
+
*
|
|
53
|
+
* ADDING A ROLE IS ONE ROW HERE. The settings key, the resolver's floor, the daemon's lookup and the settings
|
|
54
|
+
* page's row all read this table, so a job that starts picking a model tomorrow becomes configurable by saying
|
|
55
|
+
* what it is. That is the property the old grouping cost: a new surface inherited "agent runs" by being
|
|
56
|
+
* unattended, which is how a documentation sweep and a production incident came to share one tier. */
|
|
57
|
+
|
|
58
|
+
export const ModelRoleKindSchema = z.enum(["helper", "run"]);
|
|
59
|
+
export type ModelRoleKind = z.infer<typeof ModelRoleKindSchema>;
|
|
60
|
+
|
|
61
|
+
/* WHAT STARTS A RUN, declared for the run roles alone.
|
|
62
|
+
*
|
|
63
|
+
* It is not a wire value and never has been: it decides which BLOCK a job is read in, and the argument for
|
|
64
|
+
* reading them apart is the same one that separates a helper from a run — an owner holds a session nobody is
|
|
65
|
+
* watching to a different budget from one they are sitting in front of. The settings page used to make that
|
|
66
|
+
* argument in a comment over a thirteen-row list ("the ones somebody presses ahead of the ones that fire on
|
|
67
|
+
* their own"), which is a claim no reader can check and no row has to honour. Declared, it draws the page.
|
|
68
|
+
*
|
|
69
|
+
* A HELPER DECLARES NONE, and that is the honest answer rather than a gap: nobody presses "write me a commit
|
|
70
|
+
* subject". It happens because something else did. */
|
|
71
|
+
export type ModelRoleTrigger = "pressed" | "unprompted";
|
|
72
|
+
|
|
73
|
+
/* The shape of a row. `id` is a bare string HERE and narrowed on the exported type below, because the id union
|
|
74
|
+
* is derived from this very table: a self-referential `satisfies` would be a type that has to know its own
|
|
75
|
+
* answer before it can check it. */
|
|
76
|
+
interface ModelRoleRow {
|
|
77
|
+
readonly id: string;
|
|
78
|
+
// What the settings row is called. A JOB, in the owner's words, never a tier.
|
|
79
|
+
readonly label: string;
|
|
80
|
+
// The row's one line: what this model is asked to do. Read beside the label, so it says what the label
|
|
81
|
+
// cannot rather than restating it.
|
|
82
|
+
readonly blurb: string;
|
|
83
|
+
readonly kind: ModelRoleKind;
|
|
84
|
+
/** Runs only, and required for every one of them: see `ModelRoleTrigger`. */
|
|
85
|
+
readonly trigger?: ModelRoleTrigger;
|
|
86
|
+
// The row's glyph, from the shared icon set. Here rather than in a web-side map because the whole value of
|
|
87
|
+
// this table is that a role is declared ONCE; a second table keyed by the same ids is the drift this
|
|
88
|
+
// replaced, moved one layer up.
|
|
89
|
+
readonly icon: string;
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
/* THE TABLE. Ordered as the settings page draws it, and the order is an argument about reach: the one-shots
|
|
93
|
+
* first, because nobody chose a model for them and they run constantly; then the runs somebody's click starts;
|
|
94
|
+
* then the runs that start themselves, which are the ones an owner is least likely to be watching and most
|
|
95
|
+
* likely to want held to a budget. Those three are BLOCKS now (see MODEL_ROLE_BLOCKS) rather than an ordering
|
|
96
|
+
* this comment asserts and nothing holds to.
|
|
97
|
+
*
|
|
98
|
+
* The ids are the wire vocabulary: a turn carries one (AgentTurn.runRole), so they are kebab-case and stable,
|
|
99
|
+
* and renaming one is a breaking change to the setting rather than a cosmetic edit. */
|
|
100
|
+
export const MODEL_ROLES = [
|
|
101
|
+
{
|
|
102
|
+
id: "commit-message",
|
|
103
|
+
label: "Commit messages",
|
|
104
|
+
blurb: "The subject written when an agent's work lands, and the release note under it.",
|
|
105
|
+
kind: "helper",
|
|
106
|
+
icon: "file-edit",
|
|
107
|
+
},
|
|
108
|
+
{
|
|
109
|
+
id: "session-title",
|
|
110
|
+
label: "Session titles",
|
|
111
|
+
blurb: "The name a conversation wears on the board, written a second into its first turn.",
|
|
112
|
+
kind: "helper",
|
|
113
|
+
icon: "pencil",
|
|
114
|
+
},
|
|
115
|
+
{
|
|
116
|
+
/* THE ONE HELPER WHOSE INPUT IS ADVERSARIAL, and the reason the old bundling was worst here. Its prompt
|
|
117
|
+
* contains a command the agent is about to run, which may have arrived from a stranger's web page, and a
|
|
118
|
+
* small model can be talked round by it. Being wrong is expensive in both directions: a needless card
|
|
119
|
+
* teaches the owner to click through the next one. */
|
|
120
|
+
id: "safety-judge",
|
|
121
|
+
label: "Safety judge",
|
|
122
|
+
blurb: "Which model reads your safety policy before a flagged command runs.",
|
|
123
|
+
kind: "helper",
|
|
124
|
+
icon: "shield",
|
|
125
|
+
},
|
|
126
|
+
{
|
|
127
|
+
id: "loop-verdict",
|
|
128
|
+
label: "Loop verdicts",
|
|
129
|
+
blurb: "Whether a loop's iteration met the goal, or the loop goes round again.",
|
|
130
|
+
kind: "helper",
|
|
131
|
+
icon: "check-square",
|
|
132
|
+
},
|
|
133
|
+
{
|
|
134
|
+
/* THE ONE HELPER THAT ANSWERS A CLASSIFICATION rather than writing prose: one card id, or none, from the
|
|
135
|
+
* owner's own short list (schemas/personas.ts `brief`). Cheap by construction, once per chat, and the
|
|
136
|
+
* job a small model does well, which is the whole argument for routing chats onto static cards rather
|
|
137
|
+
* than asking a model to compose a context per session. */
|
|
138
|
+
id: "persona-router",
|
|
139
|
+
label: "Persona routing",
|
|
140
|
+
blurb: "Which model reads a new chat's first message and picks the persona for it.",
|
|
141
|
+
kind: "helper",
|
|
142
|
+
icon: "users",
|
|
143
|
+
},
|
|
144
|
+
{
|
|
145
|
+
id: "pipeline-fix",
|
|
146
|
+
label: "Pipeline fixes",
|
|
147
|
+
blurb: "The agent started by Fix on a red pipeline.",
|
|
148
|
+
kind: "run",
|
|
149
|
+
trigger: "pressed",
|
|
150
|
+
icon: "wave-pulse",
|
|
151
|
+
},
|
|
152
|
+
{
|
|
153
|
+
id: "deployment-fix",
|
|
154
|
+
label: "Deployment fixes",
|
|
155
|
+
blurb: "The agent started by Fix on a deployment that is down.",
|
|
156
|
+
kind: "run",
|
|
157
|
+
trigger: "pressed",
|
|
158
|
+
icon: "server",
|
|
159
|
+
},
|
|
160
|
+
{
|
|
161
|
+
id: "maintenance-chore",
|
|
162
|
+
label: "Maintenance chores",
|
|
163
|
+
blurb: "A chore run started from the Maintenance board.",
|
|
164
|
+
kind: "run",
|
|
165
|
+
trigger: "pressed",
|
|
166
|
+
icon: "wrench",
|
|
167
|
+
},
|
|
168
|
+
{
|
|
169
|
+
id: "documentation-run",
|
|
170
|
+
label: "Documentation runs",
|
|
171
|
+
blurb: "A pass over a repo's own documentation.",
|
|
172
|
+
kind: "run",
|
|
173
|
+
trigger: "pressed",
|
|
174
|
+
icon: "book",
|
|
175
|
+
},
|
|
176
|
+
{
|
|
177
|
+
id: "acceptance-run",
|
|
178
|
+
label: "Acceptance runs",
|
|
179
|
+
blurb: "One session per story in an acceptance fan-out.",
|
|
180
|
+
kind: "run",
|
|
181
|
+
trigger: "pressed",
|
|
182
|
+
icon: "list-check",
|
|
183
|
+
},
|
|
184
|
+
{
|
|
185
|
+
id: "pre-push-fix",
|
|
186
|
+
label: "Pre-push fixes",
|
|
187
|
+
blurb: "The fix proposed when a check fails on the way to a push.",
|
|
188
|
+
kind: "run",
|
|
189
|
+
trigger: "pressed",
|
|
190
|
+
icon: "cloud-upload",
|
|
191
|
+
},
|
|
192
|
+
{
|
|
193
|
+
/* PRESSED, ON THE STRENGTH OF THE APPROVAL. Nothing here runs until somebody reads the item and says
|
|
194
|
+
* yes, and that press is the start of this turn as much as Fix is the start of a pipeline run — the
|
|
195
|
+
* queue between the two is machinery, not a second decision. */
|
|
196
|
+
id: "approval-queue",
|
|
197
|
+
label: "Approvals queue",
|
|
198
|
+
blurb: "The turn that publishes or acts on what you approved.",
|
|
199
|
+
kind: "run",
|
|
200
|
+
trigger: "pressed",
|
|
201
|
+
icon: "check-circle",
|
|
202
|
+
},
|
|
203
|
+
{
|
|
204
|
+
id: "automation-wake",
|
|
205
|
+
label: "Automation wakes",
|
|
206
|
+
blurb: "A turn an automation fires: a schedule, a webhook, a message from outside.",
|
|
207
|
+
kind: "run",
|
|
208
|
+
trigger: "unprompted",
|
|
209
|
+
icon: "clock",
|
|
210
|
+
},
|
|
211
|
+
{
|
|
212
|
+
id: "extension-review",
|
|
213
|
+
label: "Extension update reviews",
|
|
214
|
+
blurb: "The agent that reads an extension update before it is applied.",
|
|
215
|
+
kind: "run",
|
|
216
|
+
trigger: "unprompted",
|
|
217
|
+
icon: "box",
|
|
218
|
+
},
|
|
219
|
+
{
|
|
220
|
+
id: "loop-iteration",
|
|
221
|
+
label: "Loop iterations",
|
|
222
|
+
blurb: "Each round of a loop working towards its goal.",
|
|
223
|
+
kind: "run",
|
|
224
|
+
trigger: "unprompted",
|
|
225
|
+
icon: "repeat",
|
|
226
|
+
},
|
|
227
|
+
{
|
|
228
|
+
id: "watch-wake",
|
|
229
|
+
label: "Watch wakes",
|
|
230
|
+
blurb: "The turn a watch starts when the thing it was watching happens.",
|
|
231
|
+
kind: "run",
|
|
232
|
+
trigger: "unprompted",
|
|
233
|
+
icon: "eye",
|
|
234
|
+
},
|
|
235
|
+
{
|
|
236
|
+
id: "verify-nudge",
|
|
237
|
+
label: "Verify nudges",
|
|
238
|
+
blurb: "The follow-up turn sent when work was left unverified.",
|
|
239
|
+
kind: "run",
|
|
240
|
+
trigger: "unprompted",
|
|
241
|
+
icon: "search",
|
|
242
|
+
},
|
|
243
|
+
{
|
|
244
|
+
/* THE ONE ROLE THAT ANSWERS A CHOICE RATHER THAN FILLING A SILENCE. A spawning agent may name its
|
|
245
|
+
* child's provider, and when it does that wins, exactly as a caret pick wins on every other run role.
|
|
246
|
+
* This is what answers when it names none, which used to be a hardcoded "claude". */
|
|
247
|
+
id: "child-agent",
|
|
248
|
+
label: "Child agents",
|
|
249
|
+
blurb: "What an agent's own subagents run on when it names no model for them.",
|
|
250
|
+
kind: "run",
|
|
251
|
+
trigger: "unprompted",
|
|
252
|
+
icon: "users",
|
|
253
|
+
},
|
|
254
|
+
] as const satisfies readonly ModelRoleRow[];
|
|
255
|
+
|
|
256
|
+
export type ModelRole = (typeof MODEL_ROLES)[number]["id"];
|
|
257
|
+
export type ModelRoleSpec = ModelRoleRow & { readonly id: ModelRole };
|
|
258
|
+
|
|
259
|
+
export const MODEL_ROLE_IDS = MODEL_ROLES.map((role) => role.id) as readonly ModelRole[];
|
|
260
|
+
|
|
261
|
+
/* The wire form. An enum rather than a string, unlike most ids in this contract, because there is no case for
|
|
262
|
+
* an unknown one: a role is a place in THIS codebase where a model gets chosen, so a value outside the table
|
|
263
|
+
* names nothing, and a settings file or a turn carrying one is a typo worth a clean error rather than a list
|
|
264
|
+
* silently ignored. */
|
|
265
|
+
export const ModelRoleSchema = z.enum(MODEL_ROLE_IDS as [ModelRole, ...ModelRole[]]);
|
|
266
|
+
|
|
267
|
+
/* ═══ THE BLOCKS THE SETTINGS PAGE READS IN ═══
|
|
268
|
+
*
|
|
269
|
+
* EIGHTEEN JOBS IN ONE LIST IS A TABLE, NOT A PAGE. The unit being the role is right and is not in question —
|
|
270
|
+
* it is what lets an owner pin Opus to commit subjects without pinning it to every session title — but the cost
|
|
271
|
+
* lands on whoever opens the page: one unbroken run of eighteen rows, each with a name, a sentence, a control
|
|
272
|
+
* and a list under it, with no landmark to say where you are in it or which rows are like the one you came for.
|
|
273
|
+
*
|
|
274
|
+
* SO THE BLOCKS ARE DECLARED, AND THEY ARE THE DISTINCTIONS THE TABLE ALREADY MAKES. Nothing here is a fresh
|
|
275
|
+
* taxonomy invented for the layout: `kind` was always the difference between a one-shot and a whole session,
|
|
276
|
+
* and `trigger` is the sentence the old table wrote in a comment about its own ordering. A block is what those
|
|
277
|
+
* two answers already separate, given a name and a heading.
|
|
278
|
+
*
|
|
279
|
+
* THE LABELS LIVE HERE, beside the role labels, for the same reason those do: a heading kept in a web-side map
|
|
280
|
+
* keyed by the same ids is the drift a single table exists to stop. What is NOT here is anything about how the
|
|
281
|
+
* page draws them — that is the page's, and it changes on its own schedule.
|
|
282
|
+
*
|
|
283
|
+
* ORDER IS REACH, and it is the order of the table itself: jobs nobody picked a model for, then sessions a
|
|
284
|
+
* click of yours starts, then sessions that start without one. Every role belongs to exactly one block and
|
|
285
|
+
* every block keeps the table's order, which is what makes a role added upstairs appear on the page by
|
|
286
|
+
* existing. */
|
|
287
|
+
export type ModelRoleBlockId = "helper" | "pressed" | "unprompted";
|
|
288
|
+
|
|
289
|
+
export interface ModelRoleBlock {
|
|
290
|
+
readonly id: ModelRoleBlockId;
|
|
291
|
+
/** The group's heading on the settings page. */
|
|
292
|
+
readonly label: string;
|
|
293
|
+
/** One line beside it: what the jobs in this block have in common that the others do not. */
|
|
294
|
+
readonly caption: string;
|
|
295
|
+
/** Its roles, in table order. */
|
|
296
|
+
readonly roles: readonly ModelRoleSpec[];
|
|
297
|
+
}
|
|
298
|
+
|
|
299
|
+
const rolesWhere = (match: (role: ModelRoleSpec) => boolean): readonly ModelRoleSpec[] => MODEL_ROLES.filter((role) => match(role));
|
|
300
|
+
|
|
301
|
+
export const MODEL_ROLE_BLOCKS: readonly ModelRoleBlock[] = [
|
|
302
|
+
{
|
|
303
|
+
id: "helper",
|
|
304
|
+
label: "Automatic helpers",
|
|
305
|
+
caption: "One prompt, no tools, one answer back.",
|
|
306
|
+
roles: rolesWhere((role) => role.kind === "helper"),
|
|
307
|
+
},
|
|
308
|
+
{
|
|
309
|
+
id: "pressed",
|
|
310
|
+
label: "Runs you start",
|
|
311
|
+
caption: "Whole sessions, begun by a click of yours.",
|
|
312
|
+
roles: rolesWhere((role) => role.kind === "run" && role.trigger === "pressed"),
|
|
313
|
+
},
|
|
314
|
+
{
|
|
315
|
+
id: "unprompted",
|
|
316
|
+
label: "Runs that start themselves",
|
|
317
|
+
caption: "Whole sessions nobody pressed for.",
|
|
318
|
+
roles: rolesWhere((role) => role.kind === "run" && role.trigger === "unprompted"),
|
|
319
|
+
},
|
|
320
|
+
];
|
package/src/plan-pools.ts
CHANGED
|
@@ -6,7 +6,7 @@ import type { AccountUsage, UsageWindow, WindowGates } from "./schemas/plan-limi
|
|
|
6
6
|
* call, is this Google fleet spent for Claude Opus, when does the pool that refused this turn reopen, what
|
|
7
7
|
* does the ring beside the composer measure. The pools a reading carries answer that only through their
|
|
8
8
|
* `gates` (UsageWindowSchema says why the reader decides them), and this file is the one place the gate is
|
|
9
|
-
* read, so the daemon's account picker, its
|
|
9
|
+
* read, so the daemon's account picker, its one-shot helper walk, its refusal dressing and the browser's rings,
|
|
10
10
|
* rail and picker rows all agree about which pool is binding for a given model.
|
|
11
11
|
*
|
|
12
12
|
* WITHOUT A MODEL the answer is the account's own tightest pool, which is what a roster or a rail that has not
|
|
@@ -88,7 +88,7 @@ test("plan mode is a request to think, so it is never answered by the model that
|
|
|
88
88
|
});
|
|
89
89
|
|
|
90
90
|
test("a surface-started run is never downgraded, because nobody is watching it fail", () => {
|
|
91
|
-
// Same call
|
|
91
|
+
// Same call a `run` role already makes in the other direction: a run billed whole, with a worktree in
|
|
92
92
|
// it, is not the place to spend a guess.
|
|
93
93
|
expect(tierOf(`what is a closure?`, { unattended: true })).toBe(`standard`);
|
|
94
94
|
});
|
package/src/prompt-complexity.ts
CHANGED
|
@@ -3,7 +3,7 @@
|
|
|
3
3
|
* The job is narrow on purpose: decide whether this turn could have run on the cheap rung of the provider the
|
|
4
4
|
* user is already on. Nothing here picks a model, nothing here reads a catalog, and nothing here calls
|
|
5
5
|
* anything. It is a pure function over the turn's own words and shape, so the daemon and the composer can both
|
|
6
|
-
* ask it and get the same answer, which is the same reason
|
|
6
|
+
* ask it and get the same answer, which is the same reason model-pins.ts lives in the contract rather than in
|
|
7
7
|
* either of them.
|
|
8
8
|
*
|
|
9
9
|
* IT CAN ONLY EVER ROUTE DOWN. The standard tier is not a setting: it is whatever the user already picked. So
|
|
@@ -213,7 +213,7 @@ const forcing = (input: ComplexityInput, text: string): ComplexityRule[] => {
|
|
|
213
213
|
rules.push("plan-mode");
|
|
214
214
|
}
|
|
215
215
|
/* An unattended run is billed whole and nobody is watching it fail. The settings this repo already ships
|
|
216
|
-
* make the same call in the other direction:
|
|
216
|
+
* make the same call in the other direction: a `run` role resolves to NOTHING when empty precisely
|
|
217
217
|
* because "nothing here can judge whether a job is worth the frontier tier". This file does judge, but not
|
|
218
218
|
* for the runs where a wrong guess costs a whole session with a worktree in it. */
|
|
219
219
|
if (input.unattended) {
|
|
@@ -115,7 +115,7 @@ test("provider ids are unique", () => {
|
|
|
115
115
|
|
|
116
116
|
/* An id may contain neither a slash nor a colon, and both exclusions are load-bearing rather than tidy.
|
|
117
117
|
* `endpoint/<id>` uses the slash to namespace a capability-minted provider, and the picker's pinned selections
|
|
118
|
-
* are `${provider}:${model}` split on the FIRST colon (
|
|
118
|
+
* are `${provider}:${model}` split on the FIRST colon (model-pins.ts), so an id carrying either would parse as
|
|
119
119
|
* something else entirely, silently. */
|
|
120
120
|
test("no provider id can be mistaken for an endpoint or a pinned selection", () => {
|
|
121
121
|
for (const spec of PROVIDER_SPECS) {
|