@intentic/sandbox-contract 1.246.1 → 1.248.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (106) hide show
  1. package/dist/batch-runs.d.ts +2 -0
  2. package/dist/batch-runs.d.ts.map +1 -1
  3. package/dist/batch-runs.js +1 -0
  4. package/dist/batch-runs.js.map +1 -1
  5. package/dist/contracts/agent.contract.d.ts +20 -0
  6. package/dist/contracts/agent.contract.d.ts.map +1 -1
  7. package/dist/contracts/personas.contract.d.ts +38 -4
  8. package/dist/contracts/personas.contract.d.ts.map +1 -1
  9. package/dist/contracts/personas.contract.js +10 -1
  10. package/dist/contracts/personas.contract.js.map +1 -1
  11. package/dist/contracts/runner.contract.d.ts +93 -93
  12. package/dist/contracts/settings.contract.d.ts +109 -12
  13. package/dist/contracts/settings.contract.d.ts.map +1 -1
  14. package/dist/contracts/usage.contract.d.ts +22 -0
  15. package/dist/contracts/usage.contract.d.ts.map +1 -1
  16. package/dist/contracts/usage.contract.js +19 -0
  17. package/dist/contracts/usage.contract.js.map +1 -1
  18. package/dist/definition.d.ts +20 -24
  19. package/dist/definition.d.ts.map +1 -1
  20. package/dist/fast-tier.js +1 -1
  21. package/dist/fast-tier.js.map +1 -1
  22. package/dist/index.d.ts +191 -19
  23. package/dist/index.d.ts.map +1 -1
  24. package/dist/index.js +2 -3
  25. package/dist/index.js.map +1 -1
  26. package/dist/model-pins.d.ts +15 -0
  27. package/dist/model-pins.d.ts.map +1 -0
  28. package/dist/model-pins.js +22 -0
  29. package/dist/model-pins.js.map +1 -0
  30. package/dist/model-roles.d.ts +172 -0
  31. package/dist/model-roles.d.ts.map +1 -0
  32. package/dist/model-roles.js +167 -0
  33. package/dist/model-roles.js.map +1 -0
  34. package/dist/schemas/agent.d.ts +22 -2
  35. package/dist/schemas/agent.d.ts.map +1 -1
  36. package/dist/schemas/agent.js +4 -2
  37. package/dist/schemas/agent.js.map +1 -1
  38. package/dist/schemas/automations.d.ts.map +1 -1
  39. package/dist/schemas/automations.js.map +1 -1
  40. package/dist/schemas/personas.d.ts +48 -4
  41. package/dist/schemas/personas.d.ts.map +1 -1
  42. package/dist/schemas/personas.js +28 -5
  43. package/dist/schemas/personas.js.map +1 -1
  44. package/dist/schemas/plan-limits.d.ts +20 -0
  45. package/dist/schemas/plan-limits.d.ts.map +1 -1
  46. package/dist/schemas/plan-limits.js +21 -0
  47. package/dist/schemas/plan-limits.js.map +1 -1
  48. package/dist/schemas/settings.d.ts +88 -6
  49. package/dist/schemas/settings.d.ts.map +1 -1
  50. package/dist/schemas/settings.js +19 -23
  51. package/dist/schemas/settings.js.map +1 -1
  52. package/dist/schemas/usage.d.ts +5 -0
  53. package/dist/schemas/usage.d.ts.map +1 -1
  54. package/dist/schemas/usage.js +5 -0
  55. package/dist/schemas/usage.js.map +1 -1
  56. package/dist/workspace-state.d.ts +0 -5
  57. package/dist/workspace-state.d.ts.map +1 -1
  58. package/dist/workspace-state.js +0 -1
  59. package/dist/workspace-state.js.map +1 -1
  60. package/package.json +4 -4
  61. package/src/agent-catalog.ts +1 -1
  62. package/src/batch-runs.test.ts +10 -5
  63. package/src/batch-runs.ts +10 -3
  64. package/src/chores/chores.ts +1 -1
  65. package/src/chores/verdict.test.ts +2 -2
  66. package/src/chores/verdict.ts +3 -3
  67. package/src/contracts/personas.contract.ts +17 -0
  68. package/src/contracts/usage.contract.ts +31 -0
  69. package/src/events.ts +1 -1
  70. package/src/fast-tier.test.ts +1 -1
  71. package/src/fast-tier.ts +5 -5
  72. package/src/index.ts +2 -3
  73. package/src/model-pins.test.ts +121 -0
  74. package/src/model-pins.ts +132 -0
  75. package/src/model-roles.test.ts +52 -0
  76. package/src/model-roles.ts +320 -0
  77. package/src/plan-pools.ts +1 -1
  78. package/src/prompt-complexity.test.ts +1 -1
  79. package/src/prompt-complexity.ts +2 -2
  80. package/src/provider-specs.test.ts +1 -1
  81. package/src/schemas/agent.ts +42 -17
  82. package/src/schemas/agents.ts +2 -2
  83. package/src/schemas/automations.ts +6 -2
  84. package/src/schemas/personas.ts +85 -18
  85. package/src/schemas/plan-limits.ts +50 -0
  86. package/src/schemas/settings.ts +92 -98
  87. package/src/schemas/usage.ts +59 -0
  88. package/src/workspace-state.test.ts +0 -1
  89. package/src/workspace-state.ts +0 -6
  90. package/dist/agent-run-model.d.ts +0 -4
  91. package/dist/agent-run-model.d.ts.map +0 -1
  92. package/dist/agent-run-model.js +0 -13
  93. package/dist/agent-run-model.js.map +0 -1
  94. package/dist/quick-model.d.ts +0 -15
  95. package/dist/quick-model.d.ts.map +0 -1
  96. package/dist/quick-model.js +0 -39
  97. package/dist/quick-model.js.map +0 -1
  98. package/dist/schemas/context.d.ts +0 -30
  99. package/dist/schemas/context.d.ts.map +0 -1
  100. package/dist/schemas/context.js +0 -34
  101. package/dist/schemas/context.js.map +0 -1
  102. package/src/agent-run-model.test.ts +0 -76
  103. package/src/agent-run-model.ts +0 -65
  104. package/src/quick-model.test.ts +0 -158
  105. package/src/quick-model.ts +0 -162
  106. package/src/schemas/context.ts +0 -87
@@ -1,76 +0,0 @@
1
- import { expect, test } from "vitest";
2
- import { resolveAgentRunModels } from "./agent-run-model.js";
3
- import type { QuickModelSource } from "./quick-model.js";
4
- import type { AgentRunPin } from "./schemas/agent.js";
5
-
6
- /* Which model a run somebody's BUTTON started opens on. The rule answers the same two surfaces its quick-model
7
- * sibling does: the daemon walks it, the settings row names it, so what these pin is the pair of properties
8
- * that separate the two: an account this sandbox cannot reach never sits at the head of the chain, and an empty
9
- * answer stays empty rather than being filled in with a tier nobody chose.
10
- *
11
- * And one property neither of those covers, new with the pins being objects: an entry's own knobs are the
12
- * entry's, so what survives the walk is the WHOLE pin. A resolver that handed back the pair inside it would run
13
- * the fallback at the head's effort, which is a tier that appears nowhere on the user's screen. */
14
-
15
- const CLAUDE: QuickModelSource = { provider: `claude`, ready: true, models: [`claude-opus-5`, `claude-sonnet-5`] };
16
- const CODEX: QuickModelSource = { provider: `codex`, ready: true, models: [`gpt-5.6`] };
17
- const GOOGLE: QuickModelSource = { provider: `gemini`, ready: true, models: [`gemini-3-pro`] };
18
-
19
- const offline = (source: QuickModelSource): QuickModelSource => ({ ...source, ready: false });
20
-
21
- const pin = (provider: string, model: string, knobs: Partial<AgentRunPin> = {}): AgentRunPin => ({ provider, model, ...knobs });
22
-
23
- test("keeps the user's own order, this list is read, never ranked", () => {
24
- // The opposite of the quick chain, which sorts by tier and cost. Here the order IS the setting: someone who
25
- // put Opus above GPT wants Opus first, and a resolver that knew better would spend the wrong account.
26
- expect(resolveAgentRunModels([CLAUDE, CODEX], [pin(`codex`, `gpt-5.6`), pin(`claude`, `claude-opus-5`)])).toEqual([
27
- { provider: `codex`, model: `gpt-5.6` },
28
- { provider: `claude`, model: `claude-opus-5` },
29
- ]);
30
- });
31
-
32
- test("steps over a provider this sandbox has no credential for", () => {
33
- // The whole reason the setting is a list. With Claude disconnected the head would otherwise be an account
34
- // that fails every Fix with agent, while a perfectly good Codex sits underneath it.
35
- expect(resolveAgentRunModels([offline(CLAUDE), CODEX], [pin(`claude`, `claude-opus-5`), pin(`codex`, `gpt-5.6`)])).toEqual([
36
- { provider: `codex`, model: `gpt-5.6` },
37
- ]);
38
- });
39
-
40
- test("each surviving entry keeps its own knobs, not the head's", () => {
41
- // The effort used to be one setting beside the list, so a fallback ran at whatever the head was set to. It
42
- // is now a property of the entry that actually answers, which is the only place it was ever true.
43
- expect(
44
- resolveAgentRunModels(
45
- [offline(CODEX), CLAUDE],
46
- [pin(`codex`, `gpt-5.6`, { effort: `low` }), pin(`claude`, `claude-opus-5`, { effort: `max`, thinking: true })],
47
- ),
48
- ).toEqual([{ provider: `claude`, model: `claude-opus-5`, effort: `max`, thinking: true }]);
49
- });
50
-
51
- test("resolves to nothing when no pin is reachable: it does NOT fall back to whatever is connected", () => {
52
- // The deliberate difference from resolveQuickModels, which lands on its Auto ladder here. An agent run is
53
- // billed in whole sessions, so an unreachable list hands the choice back to the caller's floor (the user's
54
- // own composer pick) rather than spending an account they never pointed at.
55
- expect(resolveAgentRunModels([offline(CLAUDE), GOOGLE], [pin(`claude`, `claude-opus-5`)])).toEqual([]);
56
- });
57
-
58
- test("an empty list resolves to nothing even with accounts connected", () => {
59
- expect(resolveAgentRunModels([CLAUDE, CODEX, GOOGLE], [])).toEqual([]);
60
- });
61
-
62
- test("drops a duplicate rather than spending two attempts proving one account is out, and the first one's knobs are the ones kept", () => {
63
- // Two entries can now name one model and disagree about how hard it thinks, which is what reordering a list
64
- // by hand produces. The one the user reads first is the one they meant.
65
- expect(
66
- resolveAgentRunModels([CLAUDE], [pin(`claude`, `claude-opus-5`, { effort: `max` }), pin(`claude`, `claude-opus-5`, { effort: `low` })]),
67
- ).toEqual([{ provider: `claude`, model: `claude-opus-5`, effort: `max` }]);
68
- });
69
-
70
- test("carries a model id the static catalog has never heard of", () => {
71
- // The picker offers a custom-id escape hatch, so a pin can name a model released after this build. Second-
72
- // guessing it here would quietly run something other than what the settings row says.
73
- expect(resolveAgentRunModels([CLAUDE], [pin(`claude`, `claude-opus-9-preview`)])).toEqual([
74
- { provider: `claude`, model: `claude-opus-9-preview` },
75
- ]);
76
- });
@@ -1,65 +0,0 @@
1
- import { quickModelKey, type QuickModelSource } from "./quick-model.js";
2
- import type { AgentRunPin } from "./schemas/agent.js";
3
-
4
- /* WHAT A SURFACE-STARTED AGENT RUN OPENS ON, the resolver for `agentRunModels`, sibling to resolveQuickModels
5
- * and deliberately not the same function.
6
- *
7
- * BOTH ARE ORDERED LISTS, FOR THE SAME REASON. One connected account whose allowance went on the chat this
8
- * morning is enough to take every one of these down: the user presses Fix with agent on a red pipeline, an
9
- * isolated session opens, and it dies on a credential error they cannot see from the row. Written in order, the
10
- * next entry catches it.
11
- *
12
- * THEY DIFFER ON WHAT AN EMPTY LIST MEANS, and that difference is the whole reason this is its own file rather
13
- * than a flag on the other one. A quick helper exists to stay OFF the frontier tier, so "work it out from what
14
- * is connected" is a good answer and quickModel's empty list resolves to a derived Auto ladder. An agent run is
15
- * a full session with a worktree, billed whole: nothing here can judge whether a job is worth the frontier tier,
16
- * so an empty list resolves to NOTHING and the caller falls back to the model the user picked for their own
17
- * chat, a choice they made, rather than one this file guessed for them. For the same reason there is no Auto
18
- * ladder underneath a list that has been emptied by disconnection: it would spend an account the user never
19
- * pointed at, on the most expensive kind of run this app starts.
20
- *
21
- * WHAT "STEPPED OVER" MEANS HERE IS NARROWER than the quick chain's, and worth being exact about. The quick
22
- * chain re-asks the next rung when a call comes back refused, because a one-shot that failed has cost nothing
23
- * and can simply be run again. An agent session cannot be replayed that way, by the time a provider refuses
24
- * mid-turn the agent may have already edited files, so this list is read ONCE, at the moment the turn is
25
- * composed, and steps over exactly one thing: an account that is not connected. A model that accepts the turn
26
- * and fails later is a failed run the user reads on the card, like any other. */
27
-
28
- /* The pins that could actually be started right now, in the user's own order.
29
- *
30
- * `sources` is the same readiness view resolveQuickModels takes, so both settings rows agree about which
31
- * accounts this sandbox can send to, a pin greyed as "Not connected" in one row and silently spent by the
32
- * other would be the worst of both.
33
- *
34
- * A pin whose provider is gone is DROPPED rather than held: it would otherwise sit at the head of the chain
35
- * failing every run, which is exactly what the list exists to prevent. It stays on SCREEN, greyed, the settings
36
- * row renders the stored list, not this one, because a setting that vanished from view would look like the app
37
- * had eaten it.
38
- *
39
- * THE WHOLE PIN SURVIVES, not the pair inside it: the entry's own effort, harness and cost knobs are what the
40
- * turn is composed from (turn-resume.ts), so a resolver that handed back a bare (provider, model) would silently
41
- * run the head of the list at the provider's defaults. Nothing here reads or judges those fields, which is the
42
- * point of carrying them whole.
43
- *
44
- * Empty out means nobody has pinned anything this sandbox can reach, and the caller's floor takes over. */
45
- export const resolveAgentRunModels = (sources: readonly QuickModelSource[], pinned: readonly AgentRunPin[]): readonly AgentRunPin[] => {
46
- const ready = new Set(sources.filter((source) => source.ready).map((source) => source.provider));
47
- // Taken verbatim, unvalidated against the catalog, the same reading resolveQuickModels gives its keys and
48
- // for the same reason: the picker offers a custom-id escape hatch for a model the static catalog has not
49
- // caught up with, and second-guessing the id here would run a different model than the settings row names.
50
- const requested = pinned.filter((pin) => ready.has(pin.provider));
51
- /* The same model twice would spend two attempts proving the same account is out. Hand-edited list, so this
52
- * is a real state rather than a defensive branch.
53
- *
54
- * THE FIRST OF A PAIR WINS, WHOLE. Two entries can now name one model and differ in their knobs (the same
55
- * Sonnet at Max and again at Low, written while reordering the list), and the one the user reads first is
56
- * the one they meant; keeping the earlier position with the later entry's effort would run a tier that
57
- * appears nowhere the pin does. */
58
- const chain: AgentRunPin[] = [];
59
- for (const pin of requested) {
60
- if (!chain.some((held) => quickModelKey(held) === quickModelKey(pin))) {
61
- chain.push(pin);
62
- }
63
- }
64
- return chain;
65
- };
@@ -1,158 +0,0 @@
1
- import { expect, test } from "vitest";
2
- import { type QuickModelChoice, type QuickModelSource, quickModelKey, resolveQuickModels } from "./quick-model.js";
3
-
4
- /* Which models a small automatic helper spends, and in which order. The rule answers two surfaces at once:
5
- * the daemon walks it, the browser names its head in the settings row, so what these tests pin is that a
6
- * sandbox's connections alone decide it, with no stored id to go stale, and that there is always a rung
7
- * underneath the first one whenever the sandbox has another account to reach for. */
8
-
9
- // Catalogs as their providers actually publish them: Claude's ranked list, the rest in registry order.
10
- const CLAUDE: QuickModelSource = { provider: `claude`, ready: true, models: [`claude-opus-5`, `claude-sonnet-5`, `claude-haiku-4-5-20251001`] };
11
- const GOOGLE: QuickModelSource = { provider: `gemini`, ready: true, models: [`gemini-3-flash`, `gemini-3-flash-lite`, `gemini-3-pro`] };
12
- const CODEX: QuickModelSource = { provider: `codex`, ready: true, models: [`gpt-5.4-mini`, `gpt-5.6`] };
13
- const KIMI: QuickModelSource = { provider: `kimi`, ready: true, models: [`kimi-k2.6`, `kimi-k2.7-code`, `kimi-k3`] };
14
-
15
- const offline = (source: QuickModelSource): QuickModelSource => ({ ...source, ready: false });
16
-
17
- // The model that answers when nothing goes wrong: the head of the chain, which is what most of what follows is
18
- // about and what every surface naming the spend up front reads.
19
- const head = (sources: readonly QuickModelSource[], pinned: readonly string[]): QuickModelChoice | undefined =>
20
- resolveQuickModels(sources, pinned)[0];
21
-
22
- test("reaches for the efficient rung of the one connected provider, never its flagship", () => {
23
- expect(head([CLAUDE], [])).toEqual({ provider: `claude`, model: `claude-haiku-4-5-20251001` });
24
- });
25
-
26
- test("spends the FREE channel over the subscription when both offer the same rung", () => {
27
- // Both publish a cheap-tier row, so nothing separates them on capability, and one of them costs the user
28
- // nothing while the other eats headroom they watch. A background helper should not quietly bill the Claude plan.
29
- expect(head([CLAUDE, GOOGLE], [])).toEqual({ provider: `gemini`, model: `gemini-3-flash-lite` });
30
- });
31
-
32
- test("puts tier ahead of cost: a free frontier model is still the wrong tool for a commit message", () => {
33
- // Google connected but publishing only its Pro line. Ordering on price first would seat a flagship here,
34
- // which is the exact outcome the feature exists to avoid.
35
- const proOnly: QuickModelSource = { provider: `gemini`, ready: true, models: [`gemini-3-pro`] };
36
-
37
- expect(head([CLAUDE, proOnly], [])).toEqual({ provider: `claude`, model: `claude-haiku-4-5-20251001` });
38
- });
39
-
40
- test("uses stable provider order when two subscriptions offer the same tier", () => {
41
- const kimiCheap: QuickModelSource = { provider: `kimi`, ready: true, models: [`kimi-k2-mini`] };
42
- const claudeCheap: QuickModelSource = { provider: `claude`, ready: true, models: [`claude-haiku-4-5`] };
43
-
44
- expect(head([kimiCheap, claudeCheap], [])?.provider).toBe(`claude`);
45
- });
46
-
47
- test("answers the same thing however the connected providers happen to be listed", () => {
48
- // The daemon assembles these from live stores and the browser from its own refs; neither order is a fact.
49
- const answers = [head([CLAUDE, GOOGLE, CODEX], []), head([CODEX, CLAUDE, GOOGLE], []), head([GOOGLE, CODEX, CLAUDE], [])];
50
-
51
- expect(new Set(answers.map((answer) => quickModelKey(answer!))).size).toBe(1);
52
- });
53
-
54
- test("honours a pinned model verbatim, including an id no catalog lists yet", () => {
55
- expect(head([CLAUDE, GOOGLE], [`claude:claude-opus-5`])).toEqual({ provider: `claude`, model: `claude-opus-5` });
56
- // The picker's custom-id escape hatch reaches here too: a catalog can lag a release, and running something
57
- // other than what the settings row names would be the worse failure.
58
- expect(head([CLAUDE], [`claude:claude-haiku-9`])).toEqual({ provider: `claude`, model: `claude-haiku-9` });
59
- });
60
-
61
- test("falls back to Auto when the pinned provider is no longer connected", () => {
62
- // Rather than failing every click with a credential error while the sandbox can plainly still answer.
63
- expect(head([offline(CLAUDE), GOOGLE], [`claude:claude-haiku-4-5-20251001`])).toEqual({
64
- provider: `gemini`,
65
- model: `gemini-3-flash-lite`,
66
- });
67
- });
68
-
69
- test("ignores a malformed pin instead of running an empty model id", () => {
70
- for (const pinned of [`claude`, `claude:`, `:claude-haiku-4-5`, ` `]) {
71
- expect(head([CLAUDE], [pinned])).toEqual({ provider: `claude`, model: `claude-haiku-4-5-20251001` });
72
- }
73
- });
74
-
75
- test("serves the newest of a catalog that publishes no cheap tier at all", () => {
76
- // Kimi names no tier word anywhere. There is no cheaper rung to find, so the newest row is the honest answer.
77
- expect(head([KIMI], [])).toEqual({ provider: `kimi`, model: `kimi-k3` });
78
- });
79
-
80
- test("reports nothing when no account is connected, so the button can say so instead of failing on click", () => {
81
- expect(head([offline(CLAUDE), offline(GOOGLE)], [])).toBeUndefined();
82
- expect(resolveQuickModels([offline(CLAUDE), offline(GOOGLE)], [])).toEqual([]);
83
- expect(head([], [`claude:claude-haiku-4-5`])).toBeUndefined();
84
- });
85
-
86
- test("skips a connected provider whose catalog has not loaded yet", () => {
87
- const unloaded: QuickModelSource = { provider: `grok`, ready: true, models: [] };
88
-
89
- expect(head([unloaded, CLAUDE], [])).toEqual({ provider: `claude`, model: `claude-haiku-4-5-20251001` });
90
- expect(head([unloaded], [])).toBeUndefined();
91
- });
92
-
93
- /* THE CHAIN: what the daemon walks when the model at the top of it refuses. A spent allowance is the ordinary
94
- * case, not the exotic one: the account a helper shares with the chat runs out mid-afternoon, and the whole
95
- * point of the list is that the click still lands on the next rung down. */
96
-
97
- test("keeps the pinned models in the order they were written", () => {
98
- expect(resolveQuickModels([CLAUDE, GOOGLE, CODEX], [`codex:gpt-5.6`, `gemini:gemini-3-flash`, `claude:claude-haiku-4-5-20251001`])).toEqual([
99
- { provider: `codex`, model: `gpt-5.6` },
100
- { provider: `gemini`, model: `gemini-3-flash` },
101
- { provider: `claude`, model: `claude-haiku-4-5-20251001` },
102
- ]);
103
- });
104
-
105
- test("drops a pin whose provider went away and keeps the rest of the order intact", () => {
106
- expect(resolveQuickModels([CLAUDE, offline(GOOGLE), CODEX], [`codex:gpt-5.6`, `gemini:gemini-3-flash`, `claude:claude-haiku-4-5`])).toEqual([
107
- { provider: `codex`, model: `gpt-5.6` },
108
- { provider: `claude`, model: `claude-haiku-4-5` },
109
- ]);
110
- });
111
-
112
- test("stops at the end of a pinned list rather than reaching for an account the user left out", () => {
113
- // Google and Kimi are connected and cheaper. The user wrote down one model, so one model is what this may
114
- // spend: a pin exists precisely to keep a helper off the accounts it does not name.
115
- expect(resolveQuickModels([CLAUDE, GOOGLE, KIMI], [`claude:claude-haiku-4-5`])).toEqual([{ provider: `claude`, model: `claude-haiku-4-5` }]);
116
- });
117
-
118
- test("names each model once, however many times the list repeats it", () => {
119
- // The list is edited by hand; a duplicate would spend a second attempt proving the same account is out.
120
- expect(resolveQuickModels([CLAUDE], [`claude:claude-haiku-4-5`, `claude:claude-haiku-4-5`])).toEqual([
121
- { provider: `claude`, model: `claude-haiku-4-5` },
122
- ]);
123
- });
124
-
125
- test("Auto is a ladder too, every connected provider's cheap rung, best first", () => {
126
- expect(resolveQuickModels([CLAUDE, GOOGLE, KIMI], [])).toEqual([
127
- { provider: `gemini`, model: `gemini-3-flash-lite` },
128
- { provider: `claude`, model: `claude-haiku-4-5-20251001` },
129
- { provider: `kimi`, model: `kimi-k3` },
130
- ]);
131
- });
132
-
133
- /* A MODEL ENDPOINT the user configured is a provider like any other here, and the reason it has to be is the
134
- * settings row: its options are built from the same picker catalog, so a pin naming one that this resolver
135
- * dropped would print one model's name in the settings row and spend a different account entirely. */
136
- const OLLAMA: QuickModelSource = { provider: `endpoint/ollama`, ready: true, models: [`qwen3-coder`, `gemma3-27b`] };
137
-
138
- test("honours a pin on a configured endpoint: the whole id, not the half before its slash", () => {
139
- expect(head([CLAUDE, OLLAMA], [`endpoint/ollama:qwen3-coder`])).toEqual({ provider: `endpoint/ollama`, model: `qwen3-coder` });
140
- // And it round-trips through the key shape the picker mints, which is where the slash-not-colon rule earns
141
- // itself: parsePinned splits on the FIRST colon, so an `endpoint:ollama` id would have parsed the provider
142
- // as "endpoint" and the model as "ollama:qwen3-coder": a pin that silently resolves to nothing.
143
- expect(quickModelKey({ provider: `endpoint/ollama`, model: `qwen3-coder` })).toBe(`endpoint/ollama:qwen3-coder`);
144
- });
145
-
146
- test("leaves Auto to the providers whose price is known, rather than reaching for someone's own server", () => {
147
- // Claude publishes a Haiku-class row; the endpoint's ids carry no tier word at all, so they are UNRANKED and
148
- // lose on tier. What a turn on a user's own model API costs is not a fact this repo holds, and Auto should
149
- // not be asserting one.
150
- expect(head([CLAUDE, OLLAMA], [])).toEqual({ provider: `claude`, model: `claude-haiku-4-5-20251001` });
151
- });
152
-
153
- test("still answers from an endpoint when it is the only thing configured", () => {
154
- // No tier word in either id, so the shared id-derived ordering decides between them exactly as it does for
155
- // Kimi above: the point here is that a sandbox whose only model API is its owner's still gets an answer
156
- // rather than the disabled "nothing connected" button.
157
- expect(head([offline(CLAUDE), OLLAMA], [])).toEqual({ provider: `endpoint/ollama`, model: `qwen3-coder` });
158
- });
@@ -1,162 +0,0 @@
1
- import { accessFor, modelsFor, PROVIDERS } from "./agent-catalog.js";
2
- import { ACCESS_COST } from "./provider-specs.js";
3
- import { compareCheapestFirst, familyOf, tierRankOf } from "./model-order.js";
4
- import type { AgentProvider } from "./schemas/agent.js";
5
-
6
- /* THE QUICK MODEL, the cheap, fast model a small automatic job spends instead of the frontier model the chat
7
- * runs on. Today that is the commit message written when an agent's work lands; anything else of that shape (a
8
- * branch name, a PR description) reads the same answer, which is the reason this is a `quickModel` setting
9
- * rather than a commit-message one.
10
- *
11
- * IT IS AN ORDER, NOT A MODEL, and that is the whole shape of this file. A single pick is a single point of
12
- * failure: the account it names spends its allowance on the chat all morning, and every job for the rest of the
13
- * day fails on a limit while three other connected providers sit idle. So the setting is a LIST
14
- * read top to bottom, the resolver hands back the whole ladder, and the daemon walks it until one answers.
15
- * Nothing here decides WHICH failures are worth stepping over, that is the daemon's, since only it has run
16
- * the call, this side only says what the running order is.
17
- *
18
- * The rule lives in the contract because BOTH sides need the same answer for different jobs: the daemon runs
19
- * the model, and the browser has to NAME it, in the settings row's "Auto (…)" label, before anything has been
20
- * run. Two implementations would drift precisely where it matters most, since a label promising Haiku while the
21
- * daemon bills Opus is worse than no label.
22
- *
23
- * The default is DERIVED, NEVER STORED. `quickModel` ships EMPTY and that means "work it out from whatever is
24
- * connected right now", so connecting a Google account tomorrow improves the default by itself and
25
- * disconnecting a pinned provider degrades to Auto instead of to a dead button. Same instinct as the rest of
26
- * this repo's model handling: model-order.ts derives tier and recency from the id and curates nothing, and the
27
- * web's defaultModelFor reads the live catalog rather than naming an id that a release will falsify. */
28
-
29
- /* One provider's standing in the decision: whether a turn on it can be sent at all, and what its catalog holds.
30
- *
31
- * ACP agents are deliberately not expressible here, an ACP row's model id is empty because the agent owns its
32
- * own model, so there is no cheap rung to point it at. `endpoint/<id>` providers ARE, and have to be: their
33
- * models appear in the same picker the settings row builds its options from, so a pin naming one has to hold
34
- * rather than fall silently back to Auto and spend an account the user was deliberately steering away from. */
35
- export interface QuickModelSource {
36
- // AgentProvider, not NativeProvider: an endpoint's id is user-created and cannot be in a fixed union. Auto's
37
- // ranking degrades gracefully for one, costOf falls to the metered rung and an id with no tier word is
38
- // UNRANKED, which is genuine last place, so an endpoint effectively only wins Auto when nothing else is
39
- // connected, while a PIN on one holds. Both are the right answers: what a turn on someone's own model server
40
- // costs is not a fact this repo can know, so it is not one Auto should be asserting.
41
- readonly provider: AgentProvider;
42
- // The same connection predicate every other surface gates on (access.ts web-side, the daemon's own account
43
- // stores daemon-side). A catalog is never empty by construction, so "has rows" says nothing about "can send".
44
- readonly ready: boolean;
45
- readonly models: readonly string[];
46
- }
47
-
48
- export interface QuickModelChoice {
49
- readonly provider: AgentProvider;
50
- readonly model: string;
51
- }
52
-
53
- // A pinned selection on the wire: `${provider}:${modelId}`, the same key shape the model picker already mints
54
- // for its entries (PickerEntry.key). An empty LIST of these ⇒ Auto.
55
- export const quickModelKey = (choice: QuickModelChoice): string => `${choice.provider}:${choice.model}`;
56
-
57
- /* Split on the FIRST colon only: a provider id never contains one and a model id might. Exported because the
58
- * key is what several surfaces carry a pinned pair AS: `autoFastModels` stores the same keys, the settings rows
59
- * read one back to draw the model they name, and a session composed from a pin travels as one (composeSession).
60
- *
61
- * `agentRunModels` is the one list that does NOT: an agent-run entry is an object, because it carries how the
62
- * model is to be run beside which model it is (AgentRunPinSchema), and a key with knobs spelled into it would
63
- * be a second encoding of the same thing for nobody's benefit. */
64
- export const parsePinned = (pinned: string): QuickModelChoice | undefined => {
65
- const separator = pinned.indexOf(`:`);
66
- if (separator <= 0 || separator === pinned.length - 1) {
67
- return undefined;
68
- }
69
- return { provider: pinned.slice(0, separator), model: pinned.slice(separator + 1) };
70
- };
71
-
72
- /* A pin as a person reads it: the catalog's own label for the id, or the id itself for one the static catalog
73
- * has not caught up with (the picker offers a custom-id escape hatch, so this is a real case rather than a
74
- * defensive branch). Beside parsePinned because the two are always wanted together, by any surface that has to
75
- * name what a click is about to spend BEFORE it spends it, and the two loudest of those are extensions that
76
- * share no other code with each other. */
77
- export const pinnedModelLabel = (choice: QuickModelChoice): string =>
78
- modelsFor(choice.provider).find((option) => option.value === choice.model)?.label ?? choice.model;
79
-
80
- // The cheapest row a provider publishes, its whole catalog read from the cheap end. Undefined for a catalog
81
- // that hasn't loaded yet, which is a real state: every provider serves a floor, but only once something has
82
- // asked it.
83
- const cheapestOf = (source: QuickModelSource): string | undefined => source.models.toSorted(compareCheapestFirst)[0];
84
-
85
- // Where a provider's cheapest row sits on the shared tier scale, and therefore how well it answers the question
86
- // this whole module asks. UNRANKED (-1) is a genuine last place: it means the id carries no tier word we know,
87
- // so the row is the provider's base line rather than its budget one.
88
- const tierOf = (model: string): number => tierRankOf(familyOf(model));
89
-
90
- // PROVIDERS order, as the final tiebreak. Arbitrary, but the SAME arbitrary answer on every read, the property
91
- // compareUnrankedModelIds exists to guarantee, and the one a default actually needs. An endpoint is in no fixed
92
- // list, so it reads -1 and leads the tiebreak; unreachable in practice, since it can never tie on cost.
93
- const providerOrder = (provider: AgentProvider): number => PROVIDERS.findIndex((entry) => entry.value === provider);
94
-
95
- /* WHAT AN ENDPOINT COSTS, one rung past every provider's, and the reason it is a number here rather than a
96
- * member of AccessKind. That axis describes the providers this repo ships, and every one of them is unlocked by
97
- * signing in to something the user already holds, so none of them is metered per call. An endpoint is the
98
- * opposite: whatever gateway somebody pointed us at, whose bill this repo cannot see. Reading it as dearer than
99
- * anything on the table is the conservative answer, and it is what keeps Auto from reaching for a paid gateway
100
- * on its own initiative. */
101
- const METERED_COST = Math.max(...Object.values(ACCESS_COST)) + 1;
102
-
103
- // How much a call on this provider costs at the margin. Every native provider declares an access kind; an
104
- // endpoint declares none, and takes the metered rung above.
105
- const costOf = (provider: AgentProvider): number => {
106
- const access = accessFor(provider);
107
- return access === undefined ? METERED_COST : ACCESS_COST[access.kind];
108
- };
109
-
110
- /* AUTO, every connected provider's cheapest row, best-first, as a ladder rather than a winner.
111
- *
112
- * Ranked on TIER FIRST, then cost. That order is the point of the feature: the helper exists to not be the
113
- * frontier model, so a free flagship is still the wrong tool, while a free Haiku-class row and a subscription
114
- * Haiku-class row differ only in whose quota they spend. Cost then breaks that tie towards the channel the user
115
- * is not paying per token for, and against the one they are.
116
- *
117
- * The whole ladder, not just its head, because the same ranking that picks the best answer also states the best
118
- * SECOND answer, and a sandbox with three accounts connected should not lose its commit messages for six hours
119
- * because one of them is spent. */
120
- const autoLadder = (sources: readonly QuickModelSource[]): readonly QuickModelChoice[] =>
121
- sources
122
- .filter((source) => source.ready)
123
- .flatMap((source) => {
124
- const model = cheapestOf(source);
125
- return model === undefined ? [] : [{ provider: source.provider, model }];
126
- })
127
- .toSorted(
128
- (left, right) =>
129
- tierOf(right.model) - tierOf(left.model) ||
130
- costOf(left.provider) - costOf(right.provider) ||
131
- providerOrder(left.provider) - providerOrder(right.provider),
132
- );
133
-
134
- /* WHICH MODELS A QUICK HELPER MAY RUN, IN THE ORDER IT SHOULD TRY THEM, given what this sandbox has connected.
135
- * `pinned` is the stored setting: an ordered list of `${provider}:${model}` keys, empty for Auto.
136
- *
137
- * A pin only holds while its provider is READY: an account the user disconnected would otherwise sit at the
138
- * head of the chain failing on a credential error, when the sandbox can plainly still answer. Dropping it is
139
- * the same move the composer already makes when a live catalog stops offering the selected model.
140
- *
141
- * THE PINNED LIST IS THE WHOLE ANSWER whenever any of it survives that filter. Auto does NOT get appended
142
- * underneath, and that is deliberate: a user who writes down three models has said which accounts this feature
143
- * may spend, and quietly reaching for a fourth when all three are out is exactly the "spend an account they
144
- * were steering away from" failure a pin exists to prevent. When NONE of the pins is connected any more the
145
- * list has stopped saying anything about this sandbox, so Auto takes over rather than leaving a dead button.
146
- *
147
- * Empty when nothing is connected: the caller renders a control that says so, rather than a live button that
148
- * fails on click. */
149
- export const resolveQuickModels = (sources: readonly QuickModelSource[], pinned: readonly string[]): readonly QuickModelChoice[] => {
150
- const ready = new Set(sources.filter((source) => source.ready).map((source) => source.provider));
151
- const requested = pinned.flatMap((key) => {
152
- // Taken verbatim, unvalidated against the catalog on purpose: the picker already offers a custom-id
153
- // escape hatch for a model a catalog hasn't caught up with, and second-guessing the user's own id here
154
- // would silently run a different model than the settings row names.
155
- const choice = parsePinned(key);
156
- return choice === undefined || !ready.has(choice.provider) ? [] : [choice];
157
- });
158
- // The same model twice would spend two attempts proving the same account is out, a real state, since the
159
- // list is edited by hand and Auto's ladder can rank a provider the user has also pinned.
160
- const chain = [...new Map(requested.map((choice) => [quickModelKey(choice), choice])).values()];
161
- return chain.length > 0 ? chain : autoLadder(sources);
162
- };
@@ -1,87 +0,0 @@
1
- // context: which part of the workspace a conversation carries
2
- import { z } from "zod";
3
- import { entryId } from "./internal.js";
4
-
5
- /* A CONVERSATION SEES A CHOSEN PART OF THE WORKSPACE, and these three schemas are how the choice is spelled.
6
- *
7
- * A workspace grows into more than any one session should open on: dozens of repositories, hundreds of skills,
8
- * a shelf of reference clones. Every conversation used to get all of it, one worktree per repository, frozen at
9
- * its first turn (sandbox agents/worktrees.ts). The cost is paid twice, in checkouts nobody reads and in a
10
- * project map and a skill listing that describe everything to a session that needs a corner of it
11
- * (docs/context-composition-plan.md at the workspace root measures both).
12
- *
13
- * THREE WORDS. A SHELF is what the owner writes: the items a session MAY take, in the order they are loaded and,
14
- * when a cap bites, shed from the tail. A COMPOSITION is the pick for one conversation, a list of item ids in
15
- * shelf order, stored on its registry entry. An ITEM ID names one thing in the workspace, `<kind>:<name>`.
16
- *
17
- * ONE KIND TODAY, `repo:`, which is the whole of what a composition can make real: a repository is in the
18
- * conversation (a worktree of it exists) or it is not (the directory does not exist). The grammar takes a kind
19
- * prefix so the next ones (a directory inside a repository through a sparse cone, a skill folder, a reference
20
- * shelf entry, a capability) widen this list and change nothing else. The root repository is never an item: it
21
- * is the workspace itself and every conversation stands in it. */
22
- export const CONTEXT_ITEM_KINDS = ["repo"] as const;
23
- export type ContextItemKind = (typeof CONTEXT_ITEM_KINDS)[number];
24
-
25
- /* `<kind>:<name>`. The name is a workspace-relative path or an id: safe segments, no `..`, no empty segment, so
26
- * joining it under a root can never escape, the same shape repo-discovery.ts accepts for a repo id. */
27
- const ITEM = /^(repo):[a-zA-Z0-9][a-zA-Z0-9._-]*(\/[a-zA-Z0-9][a-zA-Z0-9._-]*)*$/;
28
- export const ContextItemIdSchema = z
29
- .string()
30
- .min(1)
31
- .max(200)
32
- .regex(ITEM)
33
- .describe("One thing in the workspace a conversation can carry, as `<kind>:<name>`. Today the kind is `repo` and the name is a repository's workspace-relative path.");
34
- export type ContextItemId = z.infer<typeof ContextItemIdSchema>;
35
-
36
- // The two halves of an id, for the code that has to act on the name. The schema above already proved the shape.
37
- export const contextItem = (id: ContextItemId): { readonly kind: ContextItemKind; readonly name: string } => {
38
- const at = id.indexOf(":");
39
- return { kind: id.slice(0, at) as ContextItemKind, name: id.slice(at + 1) };
40
- };
41
-
42
- /* HOW MANY OF EACH KIND A COMPOSITION MAY HOLD. Counted per kind because the kinds cost differently: a repository
43
- * is a checkout, a skill is a line in every prompt. Absent means no limit. A pinned item is never shed to meet
44
- * a cap: `pinned` is the owner saying "always", and a cap that could override it would make two fields disagree
45
- * about the same item. */
46
- export const ContextCapsSchema = z.object({
47
- repos: z.number().int().min(0).optional().describe("How many repositories a composition may hold. Absent means as many as the shelf allows."),
48
- });
49
- export type ContextCaps = z.infer<typeof ContextCapsSchema>;
50
-
51
- /* WHAT A SESSION MAY TAKE, in the owner's order. One file per shelf under `.intentic/config/context/<id>.json`,
52
- * tracked and carried like a persona card (workspace-state.ts), for the same reason: a shelf is a list of
53
- * names, holds no credential, and belongs in a pull request.
54
- *
55
- * `allowed` IS ORDERED, and the order is the only priority there is: it is what a curator is shown, the order
56
- * items are loaded, and the order they are shed from the tail when a cap is hit. The owner fixes the order and
57
- * the ceiling; whoever picks (a model, a person on the card, the shelf itself when nobody picks) decides
58
- * membership and nothing else.
59
- *
60
- * `pinned` is always in, whether or not it is also listed in `allowed`. `denied` wins over everything, including
61
- * a pinned item, so a shelf that inherits a list it did not write can still take one thing off it. */
62
- export const ContextShelfSchema = z.object({
63
- id: entryId.describe("The shelf's id, which is also its file name."),
64
- label: z.string().max(60).optional().describe("What to call it on screen. Absent falls back to the id."),
65
- pinned: z.array(ContextItemIdSchema).max(200).default([]).describe("Items every composition from this shelf carries."),
66
- allowed: z
67
- .array(ContextItemIdSchema)
68
- .max(500)
69
- .default([])
70
- .describe("Items a composition may carry, in the order they are loaded and shed. The order is the priority; nothing else is."),
71
- denied: z.array(ContextItemIdSchema).max(200).default([]).describe("Items no composition from this shelf carries, whatever else says so."),
72
- caps: ContextCapsSchema.optional().describe("How many of each kind a composition may hold."),
73
- });
74
- export type ContextShelf = z.infer<typeof ContextShelfSchema>;
75
-
76
- /* THE PICK FOR ONE CONVERSATION, on its registry entry beside the worktree composition it decides.
77
- *
78
- * `items` is exactly what the conversation carries, in shelf order, and the root repository besides. A
79
- * conversation whose entry has NO composition carries everything the workspace has, which is what every
80
- * conversation did before shelves existed and what a conversation opened with no shelf still does. Those two
81
- * are one state on purpose: the absence is the answer, and a stored "everything" would be a second spelling of
82
- * it that could disagree with the first. */
83
- export const ContextCompositionSchema = z.object({
84
- shelf: entryId.optional().describe("Which shelf this pick was made from. Absent when the items were set without one."),
85
- items: z.array(ContextItemIdSchema).max(500).describe("What the conversation carries, in shelf order. The root repository is always carried and never listed."),
86
- });
87
- export type ContextComposition = z.infer<typeof ContextCompositionSchema>;