@intentic/sandbox-contract 1.247.0 → 1.248.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/contracts/agent.contract.d.ts +1 -0
- package/dist/contracts/agent.contract.d.ts.map +1 -1
- package/dist/contracts/personas.contract.d.ts +38 -4
- package/dist/contracts/personas.contract.d.ts.map +1 -1
- package/dist/contracts/personas.contract.js +10 -1
- package/dist/contracts/personas.contract.js.map +1 -1
- package/dist/contracts/runner.contract.d.ts +93 -93
- package/dist/contracts/settings.contract.d.ts +12 -2
- package/dist/contracts/settings.contract.d.ts.map +1 -1
- package/dist/definition.d.ts +8 -8
- package/dist/index.d.ts +51 -7
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +0 -1
- package/dist/index.js.map +1 -1
- package/dist/model-pins.d.ts +1 -3
- package/dist/model-pins.d.ts.map +1 -1
- package/dist/model-pins.js +3 -26
- package/dist/model-pins.js.map +1 -1
- package/dist/model-roles.d.ts +36 -8
- package/dist/model-roles.d.ts.map +1 -1
- package/dist/model-roles.js +48 -10
- package/dist/model-roles.js.map +1 -1
- package/dist/schemas/agent.d.ts +1 -0
- package/dist/schemas/agent.d.ts.map +1 -1
- package/dist/schemas/automations.d.ts.map +1 -1
- package/dist/schemas/automations.js.map +1 -1
- package/dist/schemas/personas.d.ts +48 -4
- package/dist/schemas/personas.d.ts.map +1 -1
- package/dist/schemas/personas.js +28 -5
- package/dist/schemas/personas.js.map +1 -1
- package/dist/schemas/settings.d.ts +6 -1
- package/dist/schemas/settings.d.ts.map +1 -1
- package/dist/schemas/settings.js +5 -6
- package/dist/schemas/settings.js.map +1 -1
- package/dist/workspace-state.d.ts +0 -5
- package/dist/workspace-state.d.ts.map +1 -1
- package/dist/workspace-state.js +0 -1
- package/dist/workspace-state.js.map +1 -1
- package/package.json +4 -4
- package/src/chores/chores.ts +1 -1
- package/src/chores/verdict.test.ts +2 -2
- package/src/chores/verdict.ts +3 -3
- package/src/contracts/personas.contract.ts +17 -0
- package/src/fast-tier.test.ts +1 -1
- package/src/fast-tier.ts +1 -1
- package/src/index.ts +0 -1
- package/src/model-pins.test.ts +32 -113
- package/src/model-pins.ts +44 -95
- package/src/model-roles.test.ts +52 -0
- package/src/model-roles.ts +119 -23
- package/src/schemas/automations.ts +6 -2
- package/src/schemas/personas.ts +85 -18
- package/src/schemas/settings.ts +22 -17
- package/src/workspace-state.test.ts +0 -1
- package/src/workspace-state.ts +0 -6
- package/dist/schemas/context.d.ts +0 -30
- package/dist/schemas/context.d.ts.map +0 -1
- package/dist/schemas/context.js +0 -34
- package/dist/schemas/context.js.map +0 -1
- package/src/schemas/context.ts +0 -87
package/src/model-pins.test.ts
CHANGED
|
@@ -1,19 +1,17 @@
|
|
|
1
1
|
import { expect, test } from "vitest";
|
|
2
|
-
import { type ModelSource, modelPinKey, parsePinned,
|
|
2
|
+
import { type ModelSource, modelPinKey, parsePinned, readyChain } from "./model-pins.js";
|
|
3
3
|
import type { ModelPin } from "./schemas/agent.js";
|
|
4
4
|
|
|
5
5
|
/* Which models a job spends, and in which order. The rule answers two surfaces at once: the daemon walks it,
|
|
6
|
-
* the browser names its head in that job's settings row, so what these tests pin is that
|
|
7
|
-
*
|
|
8
|
-
* first one whenever the sandbox has another account to reach for.
|
|
6
|
+
* the browser names its head in that job's settings row, so what these tests pin is that the OWNER'S OWN LIST
|
|
7
|
+
* decides it and nothing else, and that there is a rung underneath the first one whenever they wrote one.
|
|
9
8
|
*
|
|
10
|
-
*
|
|
11
|
-
*
|
|
12
|
-
*
|
|
13
|
-
*
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
const RUN = `pipeline-fix` as const;
|
|
9
|
+
* NOTHING IS DERIVED FOR AN EMPTY LIST, and that is the property most of this file used to be about. A `helper`
|
|
10
|
+
* role with no models used to fall to an "Auto ladder" — every connected provider's cheapest row, best-first —
|
|
11
|
+
* so a sandbox nobody had configured still spent an account on commit messages and safety verdicts, on a
|
|
12
|
+
* ranking this package invented and re-ranked whenever an account was connected. Not set now means not set,
|
|
13
|
+
* for every role, and what the caller does with an empty answer is the caller's business: a one-shot does not
|
|
14
|
+
* run, a whole session opens on the owner's own composer pick. */
|
|
17
15
|
|
|
18
16
|
// A pin as the settings rows store one. The tests are about ORDER, so most of them name only the pair; the
|
|
19
17
|
// knobs an entry can carry ride through untouched and are asserted where that matters.
|
|
@@ -22,7 +20,8 @@ const pin = (key: string): ModelPin => {
|
|
|
22
20
|
return { provider: key.slice(0, at), model: key.slice(at + 1) };
|
|
23
21
|
};
|
|
24
22
|
|
|
25
|
-
// Catalogs as their providers actually publish them
|
|
23
|
+
// Catalogs as their providers actually publish them. The resolver reads none of them — a pin is taken verbatim
|
|
24
|
+
// — but a ModelSource carries one, and a fixture that lied about it would hide a resolver that started looking.
|
|
26
25
|
const CLAUDE: ModelSource = { provider: `claude`, ready: true, models: [`claude-opus-5`, `claude-sonnet-5`, `claude-haiku-4-5-20251001`] };
|
|
27
26
|
const GOOGLE: ModelSource = { provider: `gemini`, ready: true, models: [`gemini-3-flash`, `gemini-3-flash-lite`, `gemini-3-pro`] };
|
|
28
27
|
const CODEX: ModelSource = { provider: `codex`, ready: true, models: [`gpt-5.4-mini`, `gpt-5.6`] };
|
|
@@ -30,42 +29,9 @@ const KIMI: ModelSource = { provider: `kimi`, ready: true, models: [`kimi-k2.6`,
|
|
|
30
29
|
|
|
31
30
|
const offline = (source: ModelSource): ModelSource => ({ ...source, ready: false });
|
|
32
31
|
|
|
33
|
-
// The model that answers when nothing goes wrong: the head of the chain, which is what
|
|
34
|
-
//
|
|
35
|
-
const head = (sources: readonly ModelSource[], pinned: readonly string[]): ModelPin | undefined =>
|
|
36
|
-
resolveRoleModels(sources, pinned.map(pin), HELPER)[0];
|
|
37
|
-
|
|
38
|
-
test("reaches for the efficient rung of the one connected provider, never its flagship", () => {
|
|
39
|
-
expect(head([CLAUDE], [])).toEqual({ provider: `claude`, model: `claude-haiku-4-5-20251001` });
|
|
40
|
-
});
|
|
41
|
-
|
|
42
|
-
test("spends the FREE channel over the subscription when both offer the same rung", () => {
|
|
43
|
-
// Both publish a cheap-tier row, so nothing separates them on capability, and one of them costs the user
|
|
44
|
-
// nothing while the other eats headroom they watch. A background helper should not quietly bill the Claude plan.
|
|
45
|
-
expect(head([CLAUDE, GOOGLE], [])).toEqual({ provider: `gemini`, model: `gemini-3-flash-lite` });
|
|
46
|
-
});
|
|
47
|
-
|
|
48
|
-
test("puts tier ahead of cost: a free frontier model is still the wrong tool for a commit message", () => {
|
|
49
|
-
// Google connected but publishing only its Pro line. Ordering on price first would seat a flagship here,
|
|
50
|
-
// which is the exact outcome the feature exists to avoid.
|
|
51
|
-
const proOnly: ModelSource = { provider: `gemini`, ready: true, models: [`gemini-3-pro`] };
|
|
52
|
-
|
|
53
|
-
expect(head([CLAUDE, proOnly], [])).toEqual({ provider: `claude`, model: `claude-haiku-4-5-20251001` });
|
|
54
|
-
});
|
|
55
|
-
|
|
56
|
-
test("uses stable provider order when two subscriptions offer the same tier", () => {
|
|
57
|
-
const kimiCheap: ModelSource = { provider: `kimi`, ready: true, models: [`kimi-k2-mini`] };
|
|
58
|
-
const claudeCheap: ModelSource = { provider: `claude`, ready: true, models: [`claude-haiku-4-5`] };
|
|
59
|
-
|
|
60
|
-
expect(head([kimiCheap, claudeCheap], [])?.provider).toBe(`claude`);
|
|
61
|
-
});
|
|
62
|
-
|
|
63
|
-
test("answers the same thing however the connected providers happen to be listed", () => {
|
|
64
|
-
// The daemon assembles these from live stores and the browser from its own refs; neither order is a fact.
|
|
65
|
-
const answers = [head([CLAUDE, GOOGLE, CODEX], []), head([CODEX, CLAUDE, GOOGLE], []), head([GOOGLE, CODEX, CLAUDE], [])];
|
|
66
|
-
|
|
67
|
-
expect(new Set(answers.map((answer) => modelPinKey(answer!))).size).toBe(1);
|
|
68
|
-
});
|
|
32
|
+
// The model that answers when nothing goes wrong: the head of the chain, which is what every surface naming
|
|
33
|
+
// the spend up front reads.
|
|
34
|
+
const head = (sources: readonly ModelSource[], pinned: readonly string[]): ModelPin | undefined => readyChain(sources, pinned.map(pin))[0];
|
|
69
35
|
|
|
70
36
|
test("honours a pinned model verbatim, including an id no catalog lists yet", () => {
|
|
71
37
|
expect(head([CLAUDE, GOOGLE], [`claude:claude-opus-5`])).toEqual({ provider: `claude`, model: `claude-opus-5` });
|
|
@@ -74,14 +40,6 @@ test("honours a pinned model verbatim, including an id no catalog lists yet", ()
|
|
|
74
40
|
expect(head([CLAUDE], [`claude:claude-haiku-9`])).toEqual({ provider: `claude`, model: `claude-haiku-9` });
|
|
75
41
|
});
|
|
76
42
|
|
|
77
|
-
test("falls back to Auto when the pinned provider is no longer connected", () => {
|
|
78
|
-
// Rather than failing every click with a credential error while the sandbox can plainly still answer.
|
|
79
|
-
expect(head([offline(CLAUDE), GOOGLE], [`claude:claude-haiku-4-5-20251001`])).toEqual({
|
|
80
|
-
provider: `gemini`,
|
|
81
|
-
model: `gemini-3-flash-lite`,
|
|
82
|
-
});
|
|
83
|
-
});
|
|
84
|
-
|
|
85
43
|
/* A MALFORMED KEY IS REFUSED WHERE KEYS STILL EXIST. A role's list holds PINS, whose two halves are separate
|
|
86
44
|
* fields the schema requires (ModelPinSchema), so "claude with an empty model" is no longer a shape the
|
|
87
45
|
* resolver can be handed — it is rejected at the settings boundary instead. What still travels as a key is
|
|
@@ -93,53 +51,35 @@ test("refuses a malformed key rather than reading half a pin out of it", () => {
|
|
|
93
51
|
expect(parsePinned(`claude:claude-haiku-4-5`)).toEqual({ provider: `claude`, model: `claude-haiku-4-5` });
|
|
94
52
|
});
|
|
95
53
|
|
|
96
|
-
/* THE
|
|
97
|
-
*
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
expect(
|
|
102
|
-
|
|
103
|
-
{ provider: `claude`, model: `claude-haiku-4-5-20251001` },
|
|
104
|
-
]);
|
|
105
|
-
// A pipeline fix is a whole session billed whole, and nothing here can judge what one is worth — so the
|
|
106
|
-
// caller's own floor (the owner's composer pick) answers instead of a ladder this file invented.
|
|
107
|
-
expect(resolveRoleModels([CLAUDE, GOOGLE], [], RUN)).toEqual([]);
|
|
54
|
+
/* THE CLAIM THIS FILE EXISTS FOR NOW. An unset job resolves to NOTHING however much is connected, and it does
|
|
55
|
+
* so for both kinds of role: a resolver that reached for the cheapest rung here is the whole feature that was
|
|
56
|
+
* removed, and it would come back invisibly — the daemon would simply start spending again and the settings
|
|
57
|
+
* row would look unchanged. */
|
|
58
|
+
test("resolves an unset job to nothing, however many accounts are connected", () => {
|
|
59
|
+
expect(readyChain([CLAUDE, GOOGLE, KIMI], [])).toEqual([]);
|
|
60
|
+
expect(head([CLAUDE, GOOGLE], [])).toBeUndefined();
|
|
108
61
|
});
|
|
109
62
|
|
|
110
|
-
test("carries an entry's run settings through untouched
|
|
63
|
+
test("carries an entry's run settings through untouched", () => {
|
|
111
64
|
// The resolver picks WHICH entry; how that entry runs is the entry's own business and rides along whole,
|
|
112
65
|
// because the turn (or the one-shot) is composed from all of it.
|
|
113
66
|
const configured: ModelPin = { provider: `claude`, model: `claude-opus-5`, effort: `max`, thinking: true, harness: `claude-code` };
|
|
114
67
|
|
|
115
|
-
expect(
|
|
116
|
-
expect(resolveRoleModels([CLAUDE], [configured], HELPER)).toEqual([configured]);
|
|
117
|
-
});
|
|
118
|
-
|
|
119
|
-
test("serves the newest of a catalog that publishes no cheap tier at all", () => {
|
|
120
|
-
// Kimi names no tier word anywhere. There is no cheaper rung to find, so the newest row is the honest answer.
|
|
121
|
-
expect(head([KIMI], [])).toEqual({ provider: `kimi`, model: `kimi-k3` });
|
|
68
|
+
expect(readyChain([CLAUDE], [configured])).toEqual([configured]);
|
|
122
69
|
});
|
|
123
70
|
|
|
124
|
-
test("reports nothing when
|
|
125
|
-
|
|
126
|
-
expect(
|
|
71
|
+
test("reports nothing when the accounts a job named have all gone", () => {
|
|
72
|
+
// A live button that fails on click is the failure this prevents: the caller renders the state instead.
|
|
73
|
+
expect(head([offline(CLAUDE), GOOGLE], [`claude:claude-haiku-4-5-20251001`])).toBeUndefined();
|
|
127
74
|
expect(head([], [`claude:claude-haiku-4-5`])).toBeUndefined();
|
|
128
75
|
});
|
|
129
76
|
|
|
130
|
-
test("skips a connected provider whose catalog has not loaded yet", () => {
|
|
131
|
-
const unloaded: ModelSource = { provider: `grok`, ready: true, models: [] };
|
|
132
|
-
|
|
133
|
-
expect(head([unloaded, CLAUDE], [])).toEqual({ provider: `claude`, model: `claude-haiku-4-5-20251001` });
|
|
134
|
-
expect(head([unloaded], [])).toBeUndefined();
|
|
135
|
-
});
|
|
136
|
-
|
|
137
77
|
/* THE CHAIN: what the daemon walks when the model at the top of it refuses. A spent allowance is the ordinary
|
|
138
78
|
* case, not the exotic one: the account a helper shares with the chat runs out mid-afternoon, and the whole
|
|
139
79
|
* point of the list is that the click still lands on the next rung down. */
|
|
140
80
|
|
|
141
81
|
test("keeps the pinned models in the order they were written", () => {
|
|
142
|
-
expect(
|
|
82
|
+
expect(readyChain([CLAUDE, GOOGLE, CODEX], [`codex:gpt-5.6`, `gemini:gemini-3-flash`, `claude:claude-haiku-4-5-20251001`].map(pin))).toEqual([
|
|
143
83
|
{ provider: `codex`, model: `gpt-5.6` },
|
|
144
84
|
{ provider: `gemini`, model: `gemini-3-flash` },
|
|
145
85
|
{ provider: `claude`, model: `claude-haiku-4-5-20251001` },
|
|
@@ -147,7 +87,7 @@ test("keeps the pinned models in the order they were written", () => {
|
|
|
147
87
|
});
|
|
148
88
|
|
|
149
89
|
test("drops a pin whose provider went away and keeps the rest of the order intact", () => {
|
|
150
|
-
expect(
|
|
90
|
+
expect(readyChain([CLAUDE, offline(GOOGLE), CODEX], [`codex:gpt-5.6`, `gemini:gemini-3-flash`, `claude:claude-haiku-4-5`].map(pin))).toEqual([
|
|
151
91
|
{ provider: `codex`, model: `gpt-5.6` },
|
|
152
92
|
{ provider: `claude`, model: `claude-haiku-4-5` },
|
|
153
93
|
]);
|
|
@@ -156,24 +96,17 @@ test("drops a pin whose provider went away and keeps the rest of the order intac
|
|
|
156
96
|
test("stops at the end of a pinned list rather than reaching for an account the user left out", () => {
|
|
157
97
|
// Google and Kimi are connected and cheaper. The user wrote down one model, so one model is what this may
|
|
158
98
|
// spend: a pin exists precisely to keep a helper off the accounts it does not name.
|
|
159
|
-
expect(
|
|
99
|
+
expect(readyChain([CLAUDE, GOOGLE, KIMI], [`claude:claude-haiku-4-5`].map(pin))).toEqual([{ provider: `claude`, model: `claude-haiku-4-5` }]);
|
|
160
100
|
});
|
|
161
101
|
|
|
162
102
|
test("names each model once, however many times the list repeats it", () => {
|
|
163
|
-
// The list is edited
|
|
164
|
-
|
|
103
|
+
// The list is hand-edited and the settings page can write one pin across many jobs, so a duplicate is a
|
|
104
|
+
// real state; unchecked it would spend a second attempt proving the same account is out.
|
|
105
|
+
expect(readyChain([CLAUDE], [`claude:claude-haiku-4-5`, `claude:claude-haiku-4-5`].map(pin))).toEqual([
|
|
165
106
|
{ provider: `claude`, model: `claude-haiku-4-5` },
|
|
166
107
|
]);
|
|
167
108
|
});
|
|
168
109
|
|
|
169
|
-
test("Auto is a ladder too, every connected provider's cheap rung, best first", () => {
|
|
170
|
-
expect(resolveRoleModels([CLAUDE, GOOGLE, KIMI], [], HELPER)).toEqual([
|
|
171
|
-
{ provider: `gemini`, model: `gemini-3-flash-lite` },
|
|
172
|
-
{ provider: `claude`, model: `claude-haiku-4-5-20251001` },
|
|
173
|
-
{ provider: `kimi`, model: `kimi-k3` },
|
|
174
|
-
]);
|
|
175
|
-
});
|
|
176
|
-
|
|
177
110
|
/* A MODEL ENDPOINT the user configured is a provider like any other here, and the reason it has to be is the
|
|
178
111
|
* settings row: its options are built from the same picker catalog, so a pin naming one that this resolver
|
|
179
112
|
* dropped would print one model's name in the settings row and spend a different account entirely. */
|
|
@@ -186,17 +119,3 @@ test("honours a pin on a configured endpoint: the whole id, not the half before
|
|
|
186
119
|
// as "endpoint" and the model as "ollama:qwen3-coder": a pin that silently resolves to nothing.
|
|
187
120
|
expect(modelPinKey({ provider: `endpoint/ollama`, model: `qwen3-coder` })).toBe(`endpoint/ollama:qwen3-coder`);
|
|
188
121
|
});
|
|
189
|
-
|
|
190
|
-
test("leaves Auto to the providers whose price is known, rather than reaching for someone's own server", () => {
|
|
191
|
-
// Claude publishes a Haiku-class row; the endpoint's ids carry no tier word at all, so they are UNRANKED and
|
|
192
|
-
// lose on tier. What a turn on a user's own model API costs is not a fact this repo holds, and Auto should
|
|
193
|
-
// not be asserting one.
|
|
194
|
-
expect(head([CLAUDE, OLLAMA], [])).toEqual({ provider: `claude`, model: `claude-haiku-4-5-20251001` });
|
|
195
|
-
});
|
|
196
|
-
|
|
197
|
-
test("still answers from an endpoint when it is the only thing configured", () => {
|
|
198
|
-
// No tier word in either id, so the shared id-derived ordering decides between them exactly as it does for
|
|
199
|
-
// Kimi above: the point here is that a sandbox whose only model API is its owner's still gets an answer
|
|
200
|
-
// rather than the disabled "nothing connected" button.
|
|
201
|
-
expect(head([offline(CLAUDE), OLLAMA], [])).toEqual({ provider: `endpoint/ollama`, model: `qwen3-coder` });
|
|
202
|
-
});
|
package/src/model-pins.ts
CHANGED
|
@@ -1,7 +1,4 @@
|
|
|
1
|
-
import {
|
|
2
|
-
import { ACCESS_COST } from "./provider-specs.js";
|
|
3
|
-
import { compareCheapestFirst, familyOf, tierRankOf } from "./model-order.js";
|
|
4
|
-
import { type ModelRole, modelRole } from "./model-roles.js";
|
|
1
|
+
import { modelsFor } from "./agent-catalog.js";
|
|
5
2
|
import type { AgentProvider, ModelPin } from "./schemas/agent.js";
|
|
6
3
|
|
|
7
4
|
/* WHICH MODELS A ROLE MAY RUN, IN THE ORDER TO TRY THEM. One resolver over every list in
|
|
@@ -15,30 +12,41 @@ import type { AgentProvider, ModelPin } from "./schemas/agent.js";
|
|
|
15
12
|
* this side says what the running order is.
|
|
16
13
|
*
|
|
17
14
|
* THE RULE LIVES IN THE CONTRACT because both sides need the same answer for different jobs: the daemon runs
|
|
18
|
-
* the model, and the browser has to NAME it, in
|
|
15
|
+
* the model, and the browser has to NAME it, in that job's settings row, before anything has run. Two
|
|
19
16
|
* implementations would drift precisely where it matters most, since a row promising Haiku while the daemon
|
|
20
17
|
* bills Opus is worse than no row at all.
|
|
21
18
|
*
|
|
22
|
-
*
|
|
23
|
-
*
|
|
24
|
-
*
|
|
19
|
+
* AN EMPTY LIST RESOLVES TO NOTHING, FOR EVERY ROLE, and this file no longer derives a floor for any of them.
|
|
20
|
+
* It used to: a `helper` role with no list got an "Auto ladder" worked out from whatever was connected —
|
|
21
|
+
* every provider's cheapest row, best-first — so an owner who had never opened the settings page still got
|
|
22
|
+
* commit messages, session titles and safety verdicts from a model this file picked. That was the wrong
|
|
23
|
+
* default, and the settings row saying "Auto: Gemini 3 Flash Lite, then Claude Haiku 4.5, then …" was the
|
|
24
|
+
* tell: a recommendation nobody asked for, over accounts they had connected for something else, changing
|
|
25
|
+
* under them whenever an account was added. NOT SET NOW MEANS NOT SET. Nothing is auto-selected and nothing
|
|
26
|
+
* is recommended: the owner names the models for a job or the job does not run, which is a state they can
|
|
27
|
+
* read off the row and a bill they cannot be surprised by.
|
|
28
|
+
*
|
|
29
|
+
* The two kinds of role (model-roles.ts) still differ in what the CALLER does with an empty answer — a
|
|
30
|
+
* `helper` is simply off, a `run` falls to the model the owner picked for their own chat — but that is the
|
|
31
|
+
* caller's business, and nothing here has to know which kind it is holding. */
|
|
25
32
|
|
|
26
33
|
/* One provider's standing in the decision: whether a turn on it can be sent at all, and what its catalog holds.
|
|
27
34
|
*
|
|
28
35
|
* ACP agents are deliberately not expressible here — an ACP row's model id is empty because the agent owns its
|
|
29
36
|
* own model, so there is no rung to point it at. `endpoint/<id>` providers ARE, and have to be: their models
|
|
30
37
|
* appear in the same picker the settings rows build their options from, so a pin naming one has to hold rather
|
|
31
|
-
* than
|
|
38
|
+
* than drop out and leave the job running on an account the user was deliberately steering away from. */
|
|
32
39
|
export interface ModelSource {
|
|
33
|
-
// AgentProvider, not NativeProvider: an endpoint's id is user-created and cannot be in a fixed union
|
|
34
|
-
//
|
|
35
|
-
//
|
|
36
|
-
// connected, while a PIN on one holds. Both are the right answers: what a turn on someone's own model server
|
|
37
|
-
// costs is not a fact this repo can know, so it is not one Auto should be asserting.
|
|
40
|
+
// AgentProvider, not NativeProvider: an endpoint's id is user-created and cannot be in a fixed union, and
|
|
41
|
+
// what a turn on somebody's own model server costs is not a fact this repo can know — which is fine here,
|
|
42
|
+
// because nothing on this side ranks anything. A pin either names a provider that can run it or it does not.
|
|
38
43
|
readonly provider: AgentProvider;
|
|
39
44
|
// The same connection predicate every other surface gates on (access.ts web-side, the daemon's own account
|
|
40
45
|
// stores daemon-side). A catalog is never empty by construction, so "has rows" says nothing about "can send".
|
|
41
46
|
readonly ready: boolean;
|
|
47
|
+
// What the provider publishes. NOTHING IN THIS FILE READS IT any more: it was the input to the derived Auto
|
|
48
|
+
// ladder, and a pin is taken verbatim. Kept because it is what a source IS, and the daemon's helper walk
|
|
49
|
+
// still gathers it (role-model.ts); a caller with no catalog to hand passes an empty list and loses nothing.
|
|
42
50
|
readonly models: readonly string[];
|
|
43
51
|
}
|
|
44
52
|
|
|
@@ -72,97 +80,43 @@ export const parsePinned = (pinned: string): ModelChoice | undefined => {
|
|
|
72
80
|
export const pinnedModelLabel = (choice: ModelChoice): string =>
|
|
73
81
|
modelsFor(choice.provider).find((option) => option.value === choice.model)?.label ?? choice.model;
|
|
74
82
|
|
|
75
|
-
// The cheapest row a provider publishes, its whole catalog read from the cheap end. Undefined for a catalog
|
|
76
|
-
// that hasn't loaded yet, which is a real state: every provider serves a floor, but only once something has
|
|
77
|
-
// asked it.
|
|
78
|
-
const cheapestOf = (source: ModelSource): string | undefined => source.models.toSorted(compareCheapestFirst)[0];
|
|
79
|
-
|
|
80
|
-
// Where a provider's cheapest row sits on the shared tier scale, and therefore how well it answers the question
|
|
81
|
-
// Auto asks. UNRANKED (-1) is a genuine last place: it means the id carries no tier word we know, so the row is
|
|
82
|
-
// the provider's base line rather than its budget one.
|
|
83
|
-
const tierOf = (model: string): number => tierRankOf(familyOf(model));
|
|
84
|
-
|
|
85
|
-
// PROVIDERS order, as the final tiebreak. Arbitrary, but the SAME arbitrary answer on every read, the property
|
|
86
|
-
// compareUnrankedModelIds exists to guarantee, and the one a default actually needs. An endpoint is in no fixed
|
|
87
|
-
// list, so it reads -1 and leads the tiebreak; unreachable in practice, since it can never tie on cost.
|
|
88
|
-
const providerOrder = (provider: AgentProvider): number => PROVIDERS.findIndex((entry) => entry.value === provider);
|
|
89
|
-
|
|
90
|
-
/* WHAT AN ENDPOINT COSTS, one rung past every provider's, and the reason it is a number here rather than a
|
|
91
|
-
* member of AccessKind. That axis describes the providers this repo ships, and every one of them is unlocked by
|
|
92
|
-
* signing in to something the user already holds, so none of them is metered per call. An endpoint is the
|
|
93
|
-
* opposite: whatever gateway somebody pointed us at, whose bill this repo cannot see. Reading it as dearer than
|
|
94
|
-
* anything on the table is the conservative answer, and it is what keeps Auto from reaching for a paid gateway
|
|
95
|
-
* on its own initiative. */
|
|
96
|
-
const METERED_COST = Math.max(...Object.values(ACCESS_COST)) + 1;
|
|
97
|
-
|
|
98
|
-
// How much a call on this provider costs at the margin. Every native provider declares an access kind; an
|
|
99
|
-
// endpoint declares none, and takes the metered rung above.
|
|
100
|
-
const costOf = (provider: AgentProvider): number => {
|
|
101
|
-
const access = accessFor(provider);
|
|
102
|
-
return access === undefined ? METERED_COST : ACCESS_COST[access.kind];
|
|
103
|
-
};
|
|
104
|
-
|
|
105
|
-
/* AUTO, every connected provider's cheapest row, best-first, as a ladder rather than a winner. The floor under
|
|
106
|
-
* every `helper` role, and under nothing else.
|
|
107
|
-
*
|
|
108
|
-
* Ranked on TIER FIRST, then cost. That order is the point: a helper's Auto exists to not be the frontier
|
|
109
|
-
* model, so a free flagship is still the wrong tool, while a free Haiku-class row and a subscription
|
|
110
|
-
* Haiku-class row differ only in whose quota they spend. Cost then breaks that tie towards the channel the user
|
|
111
|
-
* is not paying per token for, and against the one they are.
|
|
112
|
-
*
|
|
113
|
-
* NOTE WHAT AUTO IS AND IS NOT AN ARGUMENT FOR. It is the answer for an owner who has said nothing, not a claim
|
|
114
|
-
* that cheap is right: an owner who pins Opus to commit messages is not being talked out of it, which is the
|
|
115
|
-
* whole reason these lists are per role. Auto is what a row says while it is empty.
|
|
116
|
-
*
|
|
117
|
-
* The whole ladder, not just its head, because the same ranking that picks the best answer also states the best
|
|
118
|
-
* SECOND answer, and a sandbox with three accounts connected should not lose its commit messages for six hours
|
|
119
|
-
* because one of them is spent. */
|
|
120
|
-
export const autoLadder = (sources: readonly ModelSource[]): readonly ModelPin[] =>
|
|
121
|
-
sources
|
|
122
|
-
.filter((source) => source.ready)
|
|
123
|
-
.flatMap((source) => {
|
|
124
|
-
const model = cheapestOf(source);
|
|
125
|
-
return model === undefined ? [] : [{ provider: source.provider, model }];
|
|
126
|
-
})
|
|
127
|
-
.toSorted(
|
|
128
|
-
(left, right) =>
|
|
129
|
-
tierOf(right.model) - tierOf(left.model) ||
|
|
130
|
-
costOf(left.provider) - costOf(right.provider) ||
|
|
131
|
-
providerOrder(left.provider) - providerOrder(right.provider),
|
|
132
|
-
);
|
|
133
|
-
|
|
134
83
|
/* WHICH MODELS THIS ROLE MAY RUN, IN THE ORDER TO TRY THEM, given what this sandbox has connected.
|
|
135
|
-
* `pinned` is the stored setting: settings.modelRoles[role], an ordered list of pins
|
|
136
|
-
* floor.
|
|
84
|
+
* `pinned` is the stored setting: settings.modelRoles[role], an ordered list of pins.
|
|
137
85
|
*
|
|
138
86
|
* A pin only holds while its provider is READY: an account the user disconnected would otherwise sit at the
|
|
139
|
-
* head of the chain failing on a credential error
|
|
140
|
-
*
|
|
141
|
-
*
|
|
142
|
-
* from view would look like the app had eaten it.
|
|
87
|
+
* head of the chain failing on a credential error while the sandbox can plainly still answer from the rung
|
|
88
|
+
* below. It stays on SCREEN, greyed — the settings row renders the stored list, not this one — because a
|
|
89
|
+
* setting that vanished from view would look like the app had eaten it.
|
|
143
90
|
*
|
|
144
|
-
* THE PINNED LIST IS THE WHOLE ANSWER
|
|
145
|
-
*
|
|
146
|
-
* spend
|
|
147
|
-
*
|
|
148
|
-
* stopped saying anything about this sandbox, so the floor takes over rather than leaving a dead button.
|
|
91
|
+
* THE PINNED LIST IS THE WHOLE ANSWER, and there is nothing underneath it. A user who writes down three models
|
|
92
|
+
* has said which accounts this job may spend, and reaching for a fourth when all three are out is exactly the
|
|
93
|
+
* "spend an account they were steering away from" failure a pin exists to prevent. A user who writes down none
|
|
94
|
+
* has said the job picks no model at all.
|
|
149
95
|
*
|
|
150
96
|
* THE WHOLE PIN SURVIVES, not the pair inside it: an entry's effort, thinking, speed and harness are what the
|
|
151
97
|
* work is composed from, so a resolver handing back a bare (provider, model) would silently run the head of the
|
|
152
98
|
* list at the provider's defaults. Nothing here reads or judges those fields, which is the point of carrying
|
|
153
99
|
* them whole.
|
|
154
100
|
*
|
|
155
|
-
*
|
|
156
|
-
*
|
|
157
|
-
*
|
|
158
|
-
|
|
101
|
+
* EMPTY OUT MEANS THE LIST HAS NOTHING IT MAY REACH, from two different causes the caller can tell apart by
|
|
102
|
+
* looking at `pinned`: an empty list is an owner who set no model, and a full list that survives none of the
|
|
103
|
+
* readiness filter is an owner whose accounts have gone. The first is the job being switched off, the second
|
|
104
|
+
* is worth a sentence about the accounts.
|
|
105
|
+
*
|
|
106
|
+
* ONE FUNCTION FOR BOTH KINDS OF LADDER, and it is `resolveRoleModels` that stopped existing rather than this
|
|
107
|
+
* one arriving to replace it. A role's list used to add the role's own floor beneath the ready chain, which is
|
|
108
|
+
* the only thing it did that a persona card's list (schemas/personas.ts `personaModels`) did not — so the walk
|
|
109
|
+
* was split out to be shared. With the floor gone there is no difference left to share around: a role's list
|
|
110
|
+
* and a card's list are the same question over the same sources, and two names for it would be two places to
|
|
111
|
+
* read before believing they agree. */
|
|
112
|
+
export const readyChain = (sources: readonly ModelSource[], pinned: readonly ModelPin[]): readonly ModelPin[] => {
|
|
159
113
|
const ready = new Set(sources.filter((source) => source.ready).map((source) => source.provider));
|
|
160
114
|
// Taken verbatim, unvalidated against the catalog on purpose: the picker offers a custom-id escape hatch for
|
|
161
115
|
// a model a catalog hasn't caught up with, and second-guessing the user's own id here would silently run a
|
|
162
116
|
// different model than the settings row names.
|
|
163
117
|
const requested = pinned.filter((pin) => ready.has(pin.provider));
|
|
164
118
|
/* The same model twice would spend two attempts proving one account is out — a real state, since the list is
|
|
165
|
-
* hand-edited and
|
|
119
|
+
* hand-edited and the bulk editor writes one pin across many jobs.
|
|
166
120
|
*
|
|
167
121
|
* THE FIRST OF A PAIR WINS, WHOLE. Two entries can name one model and differ in their knobs (the same Sonnet
|
|
168
122
|
* at Max and again at Low, written while reordering the list), and the one the user reads first is the one
|
|
@@ -174,10 +128,5 @@ export const resolveRoleModels = (sources: readonly ModelSource[], pinned: reado
|
|
|
174
128
|
chain.push(pin);
|
|
175
129
|
}
|
|
176
130
|
}
|
|
177
|
-
|
|
178
|
-
return chain;
|
|
179
|
-
}
|
|
180
|
-
// The floor, which the role declares. An id outside the table has no floor to fall to, and answering with
|
|
181
|
-
// the cheapest connected model for it would be this file inventing a job.
|
|
182
|
-
return modelRole(role)?.kind === `helper` ? autoLadder(sources) : [];
|
|
131
|
+
return chain;
|
|
183
132
|
};
|
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
import { expect, test } from "vitest";
|
|
2
|
+
import { MODEL_ROLE_BLOCKS, MODEL_ROLES, type ModelRoleSpec } from "./model-roles.js";
|
|
3
|
+
|
|
4
|
+
/* THE CATALOG DRAWS THE SETTINGS PAGE, so the properties the page relies on have to be true of the TABLE rather
|
|
5
|
+
* than remembered by whoever last added a row. Eighteen jobs in one unbroken list is what the blocks exist to
|
|
6
|
+
* break up, and the failure they replace is a silent one: a role that belongs to no block is simply missing
|
|
7
|
+
* from Sandbox ▸ Agent ▸ Models, with no error anywhere and a page that looks completely normal. */
|
|
8
|
+
|
|
9
|
+
// The table at its declared width rather than as the literal tuple: `trigger` is absent from a helper's literal
|
|
10
|
+
// type, and what is under test is exactly whether it is absent where it should be.
|
|
11
|
+
const roles: readonly ModelRoleSpec[] = MODEL_ROLES;
|
|
12
|
+
|
|
13
|
+
/* WHAT STARTS A RUN IS THE BLOCK IT IS READ IN, and asserting the two against each other is what stops them
|
|
14
|
+
* drifting: a run whose trigger says one thing while the page draws it under the other heading is a row telling
|
|
15
|
+
* an owner that a session nobody is watching is one they start. A helper answers with its own kind, because it
|
|
16
|
+
* has no trigger and its block is the kind itself. */
|
|
17
|
+
test("every whole session says what starts it, and it is the block it is drawn in", () => {
|
|
18
|
+
const blockOf = new Map(MODEL_ROLE_BLOCKS.flatMap((block) => block.roles.map((role) => [role.id, block.id] as const)));
|
|
19
|
+
|
|
20
|
+
for (const role of roles) {
|
|
21
|
+
expect(role.kind === `run` ? role.trigger : role.kind, role.id).toBe(blockOf.get(role.id));
|
|
22
|
+
}
|
|
23
|
+
});
|
|
24
|
+
|
|
25
|
+
test("a one-shot declares no trigger at all: nothing presses a commit subject into being", () => {
|
|
26
|
+
for (const helper of roles.filter((role) => role.kind === `helper`)) {
|
|
27
|
+
expect(Object.keys(helper), helper.id).not.toContain(`trigger`);
|
|
28
|
+
}
|
|
29
|
+
});
|
|
30
|
+
|
|
31
|
+
/* THE BLOCKS PARTITION THE TABLE: every role in exactly one, nothing invented. Asserted as a partition rather
|
|
32
|
+
* than block by block, because the two ways to get this wrong are opposites and only one of them is visible —
|
|
33
|
+
* a role in no block vanishes from the page, a role in two is drawn twice and would be caught by eye. */
|
|
34
|
+
test("the blocks hold every role once, in the table's own order", () => {
|
|
35
|
+
const blocked = MODEL_ROLE_BLOCKS.flatMap((block) => block.roles.map((role) => role.id));
|
|
36
|
+
|
|
37
|
+
expect(blocked.toSorted()).toEqual(MODEL_ROLES.map((role) => role.id).toSorted());
|
|
38
|
+
/* AND THE ORDER SURVIVES THE SPLIT. Reading the blocks in order is reading the table in order: the table
|
|
39
|
+
* argues its sequence is REACH (jobs nobody picked a model for, then sessions a click starts, then sessions
|
|
40
|
+
* that start without one), and a block re-sorted here would leave that argument describing a page it no
|
|
41
|
+
* longer matches. */
|
|
42
|
+
expect(blocked).toEqual(MODEL_ROLES.map((role) => role.id));
|
|
43
|
+
});
|
|
44
|
+
|
|
45
|
+
test("each block says what it is, so the page keeps no headings of its own", () => {
|
|
46
|
+
for (const block of MODEL_ROLE_BLOCKS) {
|
|
47
|
+
expect(block.label, block.id).toMatch(/\S/);
|
|
48
|
+
expect(block.caption, block.id).toMatch(/\S/);
|
|
49
|
+
// A block with nothing in it would draw a heading over an empty surface.
|
|
50
|
+
expect(block.roles.length, block.id).toBeGreaterThan(0);
|
|
51
|
+
}
|
|
52
|
+
});
|