@intentic/sandbox-contract 1.245.0 → 1.247.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (247) hide show
  1. package/README.md +17 -1
  2. package/dist/batch-runs.d.ts +2 -0
  3. package/dist/batch-runs.d.ts.map +1 -1
  4. package/dist/batch-runs.js +1 -0
  5. package/dist/batch-runs.js.map +1 -1
  6. package/dist/command-classes.d.ts +6 -3
  7. package/dist/command-classes.d.ts.map +1 -1
  8. package/dist/command-classes.js +43 -18
  9. package/dist/command-classes.js.map +1 -1
  10. package/dist/contracts/{cursor.contract.d.ts → accounts.contract.d.ts} +102 -3
  11. package/dist/contracts/accounts.contract.d.ts.map +1 -0
  12. package/dist/contracts/accounts.contract.js +61 -0
  13. package/dist/contracts/accounts.contract.js.map +1 -0
  14. package/dist/contracts/agent.contract.d.ts +19 -0
  15. package/dist/contracts/agent.contract.d.ts.map +1 -1
  16. package/dist/contracts/agents.contract.d.ts +121 -0
  17. package/dist/contracts/agents.contract.d.ts.map +1 -1
  18. package/dist/contracts/agents.contract.js +4 -4
  19. package/dist/contracts/agents.contract.js.map +1 -1
  20. package/dist/contracts/ci.contract.d.ts +2 -0
  21. package/dist/contracts/ci.contract.d.ts.map +1 -1
  22. package/dist/contracts/host.contract.d.ts +35 -0
  23. package/dist/contracts/host.contract.d.ts.map +1 -1
  24. package/dist/contracts/host.contract.js +3 -2
  25. package/dist/contracts/host.contract.js.map +1 -1
  26. package/dist/contracts/personas.contract.d.ts +4 -2
  27. package/dist/contracts/personas.contract.d.ts.map +1 -1
  28. package/dist/contracts/runner.contract.d.ts +2 -2
  29. package/dist/contracts/settings.contract.d.ts +58 -20
  30. package/dist/contracts/settings.contract.d.ts.map +1 -1
  31. package/dist/contracts/system.contract.d.ts +80 -30
  32. package/dist/contracts/system.contract.d.ts.map +1 -1
  33. package/dist/contracts/system.contract.js +26 -17
  34. package/dist/contracts/system.contract.js.map +1 -1
  35. package/dist/contracts/usage.contract.d.ts +22 -0
  36. package/dist/contracts/usage.contract.d.ts.map +1 -1
  37. package/dist/contracts/usage.contract.js +19 -0
  38. package/dist/contracts/usage.contract.js.map +1 -1
  39. package/dist/definition.d.ts +20 -28
  40. package/dist/definition.d.ts.map +1 -1
  41. package/dist/documents.d.ts +0 -1
  42. package/dist/documents.d.ts.map +1 -1
  43. package/dist/documents.js +1 -2
  44. package/dist/documents.js.map +1 -1
  45. package/dist/embed.d.ts +23 -0
  46. package/dist/embed.d.ts.map +1 -0
  47. package/dist/embed.js +84 -0
  48. package/dist/embed.js.map +1 -0
  49. package/dist/events.d.ts +21 -0
  50. package/dist/events.d.ts.map +1 -1
  51. package/dist/events.js +5 -2
  52. package/dist/events.js.map +1 -1
  53. package/dist/fast-tier.js +1 -1
  54. package/dist/fast-tier.js.map +1 -1
  55. package/dist/history-state.d.ts.map +1 -1
  56. package/dist/history-state.js +2 -0
  57. package/dist/history-state.js.map +1 -1
  58. package/dist/index.d.ts +453 -306
  59. package/dist/index.d.ts.map +1 -1
  60. package/dist/index.js +7 -15
  61. package/dist/index.js.map +1 -1
  62. package/dist/model-pins.d.ts +17 -0
  63. package/dist/model-pins.d.ts.map +1 -0
  64. package/dist/{quick-model.js → model-pins.js} +18 -11
  65. package/dist/model-pins.js.map +1 -0
  66. package/dist/model-roles.d.ts +144 -0
  67. package/dist/model-roles.d.ts.map +1 -0
  68. package/dist/model-roles.js +129 -0
  69. package/dist/model-roles.js.map +1 -0
  70. package/dist/peer-dial.d.ts +33 -0
  71. package/dist/peer-dial.d.ts.map +1 -0
  72. package/dist/peer-dial.js +79 -0
  73. package/dist/peer-dial.js.map +1 -0
  74. package/dist/peer-mcp-server.d.ts +36 -0
  75. package/dist/peer-mcp-server.d.ts.map +1 -0
  76. package/dist/peer-mcp-server.js +71 -0
  77. package/dist/peer-mcp-server.js.map +1 -0
  78. package/dist/provider-specs.d.ts +38 -20
  79. package/dist/provider-specs.d.ts.map +1 -1
  80. package/dist/provider-specs.js +39 -13
  81. package/dist/provider-specs.js.map +1 -1
  82. package/dist/runtime-state.d.ts +1 -1
  83. package/dist/runtime-state.js +1 -1
  84. package/dist/runtime-state.js.map +1 -1
  85. package/dist/safety-policy.d.ts +12 -3
  86. package/dist/safety-policy.d.ts.map +1 -1
  87. package/dist/safety-policy.js +30 -5
  88. package/dist/safety-policy.js.map +1 -1
  89. package/dist/schemas/agent.d.ts +27 -8
  90. package/dist/schemas/agent.d.ts.map +1 -1
  91. package/dist/schemas/agent.js +10 -4
  92. package/dist/schemas/agent.js.map +1 -1
  93. package/dist/schemas/agents.d.ts +42 -0
  94. package/dist/schemas/agents.d.ts.map +1 -1
  95. package/dist/schemas/agents.js +25 -4
  96. package/dist/schemas/agents.js.map +1 -1
  97. package/dist/schemas/automations.d.ts +11 -2
  98. package/dist/schemas/automations.d.ts.map +1 -1
  99. package/dist/schemas/automations.js +1 -1
  100. package/dist/schemas/automations.js.map +1 -1
  101. package/dist/schemas/ci.d.ts +6 -0
  102. package/dist/schemas/ci.d.ts.map +1 -1
  103. package/dist/schemas/ci.js +3 -2
  104. package/dist/schemas/ci.js.map +1 -1
  105. package/dist/schemas/context.d.ts +30 -0
  106. package/dist/schemas/context.d.ts.map +1 -0
  107. package/dist/schemas/context.js +34 -0
  108. package/dist/schemas/context.js.map +1 -0
  109. package/dist/schemas/{computers.d.ts → devices.d.ts} +154 -60
  110. package/dist/schemas/devices.d.ts.map +1 -0
  111. package/dist/schemas/devices.js +157 -0
  112. package/dist/schemas/devices.js.map +1 -0
  113. package/dist/schemas/hosts.d.ts +12 -0
  114. package/dist/schemas/hosts.d.ts.map +1 -1
  115. package/dist/schemas/hosts.js +1 -0
  116. package/dist/schemas/hosts.js.map +1 -1
  117. package/dist/schemas/issues.d.ts +0 -5
  118. package/dist/schemas/issues.d.ts.map +1 -1
  119. package/dist/schemas/issues.js +0 -1
  120. package/dist/schemas/issues.js.map +1 -1
  121. package/dist/schemas/personas.d.ts +5 -3
  122. package/dist/schemas/personas.d.ts.map +1 -1
  123. package/dist/schemas/personas.js +3 -2
  124. package/dist/schemas/personas.js.map +1 -1
  125. package/dist/schemas/plan-limits.d.ts +20 -0
  126. package/dist/schemas/plan-limits.d.ts.map +1 -1
  127. package/dist/schemas/plan-limits.js +21 -0
  128. package/dist/schemas/plan-limits.js.map +1 -1
  129. package/dist/schemas/provider-oauth.d.ts +48 -16
  130. package/dist/schemas/provider-oauth.d.ts.map +1 -1
  131. package/dist/schemas/provider-oauth.js +22 -20
  132. package/dist/schemas/provider-oauth.js.map +1 -1
  133. package/dist/schemas/settings.d.ts +42 -16
  134. package/dist/schemas/settings.d.ts.map +1 -1
  135. package/dist/schemas/settings.js +22 -28
  136. package/dist/schemas/settings.js.map +1 -1
  137. package/dist/schemas/terminal.js +9 -9
  138. package/dist/schemas/terminal.js.map +1 -1
  139. package/dist/schemas/usage.d.ts +5 -2
  140. package/dist/schemas/usage.d.ts.map +1 -1
  141. package/dist/schemas/usage.js +5 -2
  142. package/dist/schemas/usage.js.map +1 -1
  143. package/dist/shell-regions.d.ts +4 -0
  144. package/dist/shell-regions.d.ts.map +1 -0
  145. package/dist/shell-regions.js +156 -0
  146. package/dist/shell-regions.js.map +1 -0
  147. package/dist/workspace-state.d.ts +8 -0
  148. package/dist/workspace-state.d.ts.map +1 -1
  149. package/dist/workspace-state.js +13 -5
  150. package/dist/workspace-state.js.map +1 -1
  151. package/package.json +37 -4
  152. package/src/agent-catalog.ts +2 -2
  153. package/src/arrival.ts +3 -3
  154. package/src/batch-runs.test.ts +10 -5
  155. package/src/batch-runs.ts +10 -3
  156. package/src/command-classes.test.ts +195 -71
  157. package/src/command-classes.ts +148 -46
  158. package/src/contracts/accounts.contract.ts +94 -0
  159. package/src/contracts/agents.contract.ts +4 -3
  160. package/src/contracts/exit.contract.ts +2 -2
  161. package/src/contracts/host.contract.ts +17 -5
  162. package/src/contracts/settings.contract.ts +1 -1
  163. package/src/contracts/system.contract.ts +43 -24
  164. package/src/contracts/usage.contract.ts +31 -0
  165. package/src/contracts/vpn.contract.ts +2 -2
  166. package/src/documents.test.ts +2 -1
  167. package/src/documents.ts +7 -11
  168. package/src/embed.test.ts +68 -0
  169. package/src/embed.ts +164 -0
  170. package/src/events.ts +31 -4
  171. package/src/fast-tier.test.ts +1 -1
  172. package/src/fast-tier.ts +5 -5
  173. package/src/history-state.ts +12 -3
  174. package/src/host-protocol.ts +2 -2
  175. package/src/index.ts +8 -16
  176. package/src/model-order.ts +1 -1
  177. package/src/{quick-model.test.ts → model-pins.test.ts} +73 -29
  178. package/src/model-pins.ts +183 -0
  179. package/src/model-roles.ts +224 -0
  180. package/src/peer-dial.test.ts +203 -0
  181. package/src/peer-dial.ts +163 -0
  182. package/src/peer-mcp-server.test.ts +104 -0
  183. package/src/peer-mcp-server.ts +144 -0
  184. package/src/plan-pools.ts +1 -1
  185. package/src/prompt-complexity.test.ts +1 -1
  186. package/src/prompt-complexity.ts +2 -2
  187. package/src/provider-specs.test.ts +45 -18
  188. package/src/provider-specs.ts +147 -67
  189. package/src/routes.test.ts +6 -3
  190. package/src/runner-protocol.ts +1 -1
  191. package/src/runtime-state.ts +2 -2
  192. package/src/safety-policy.test.ts +88 -0
  193. package/src/safety-policy.ts +84 -14
  194. package/src/schemas/agent.ts +83 -30
  195. package/src/schemas/agents.ts +67 -6
  196. package/src/schemas/automations.ts +6 -4
  197. package/src/schemas/capabilities.ts +4 -4
  198. package/src/schemas/ci.ts +23 -6
  199. package/src/schemas/context.ts +87 -0
  200. package/src/schemas/{computers.ts → devices.ts} +190 -107
  201. package/src/schemas/hosts.ts +5 -1
  202. package/src/schemas/issues.ts +0 -4
  203. package/src/schemas/personas.ts +8 -3
  204. package/src/schemas/plan-limits.ts +50 -0
  205. package/src/schemas/provider-oauth.ts +49 -52
  206. package/src/schemas/settings.ts +105 -140
  207. package/src/schemas/terminal.ts +12 -12
  208. package/src/schemas/usage.ts +62 -27
  209. package/src/schemas/version-seam.test.ts +0 -1
  210. package/src/shell-regions.ts +289 -0
  211. package/src/versions.ts +2 -2
  212. package/src/webext-links.ts +2 -2
  213. package/src/webext-protocol.ts +2 -2
  214. package/src/workspace-state.test.ts +48 -1
  215. package/src/workspace-state.ts +48 -11
  216. package/dist/agent-run-model.d.ts +0 -4
  217. package/dist/agent-run-model.d.ts.map +0 -1
  218. package/dist/agent-run-model.js +0 -13
  219. package/dist/agent-run-model.js.map +0 -1
  220. package/dist/contracts/claude.contract.d.ts +0 -91
  221. package/dist/contracts/claude.contract.d.ts.map +0 -1
  222. package/dist/contracts/claude.contract.js +0 -50
  223. package/dist/contracts/claude.contract.js.map +0 -1
  224. package/dist/contracts/cursor.contract.d.ts.map +0 -1
  225. package/dist/contracts/cursor.contract.js +0 -50
  226. package/dist/contracts/cursor.contract.js.map +0 -1
  227. package/dist/contracts/grok.contract.d.ts +0 -36
  228. package/dist/contracts/grok.contract.d.ts.map +0 -1
  229. package/dist/contracts/grok.contract.js +0 -31
  230. package/dist/contracts/grok.contract.js.map +0 -1
  231. package/dist/contracts/keys.contract.d.ts +0 -81
  232. package/dist/contracts/keys.contract.d.ts.map +0 -1
  233. package/dist/contracts/keys.contract.js +0 -51
  234. package/dist/contracts/keys.contract.js.map +0 -1
  235. package/dist/quick-model.d.ts +0 -15
  236. package/dist/quick-model.d.ts.map +0 -1
  237. package/dist/quick-model.js.map +0 -1
  238. package/dist/schemas/computers.d.ts.map +0 -1
  239. package/dist/schemas/computers.js +0 -135
  240. package/dist/schemas/computers.js.map +0 -1
  241. package/src/agent-run-model.test.ts +0 -76
  242. package/src/agent-run-model.ts +0 -65
  243. package/src/contracts/claude.contract.ts +0 -71
  244. package/src/contracts/cursor.contract.ts +0 -74
  245. package/src/contracts/grok.contract.ts +0 -41
  246. package/src/contracts/keys.contract.ts +0 -79
  247. package/src/quick-model.ts +0 -155
@@ -58,6 +58,56 @@ export const AccountUsageSchema = z.object({
58
58
  measuredAt: z.number(),
59
59
  });
60
60
  export type AccountUsage = z.infer<typeof AccountUsageSchema>;
61
+ /* THE ONE WAY PAST A SPENT SESSION WINDOW THAT IS NOT WAITING, which Anthropic grants once a week per account.
62
+ *
63
+ * Every other affordance around a refused turn is about WHEN: arm the appointment, count down to the reset,
64
+ * press when it opens. This one moves the clock. The provider reopens the five-hour window immediately and
65
+ * charges the account one of its weekly resets; the WEEKLY allowance is untouched and still binds, so this
66
+ * buys back the session pool and nothing else. Upstream's own CLI spells it `/limit-reset`.
67
+ *
68
+ * THE ANSWER IS THE PROVIDER'S, NEVER OURS. There is no rule here to re-derive: eligibility turns on the plan
69
+ * tier, how long the account has existed, whether it is actually at the wall, whether another experiment holds
70
+ * it, and whether the week's reset is already spent — all of it decided server-side and none of it visible from
71
+ * a usage reading. So this shape is a transcription of what the endpoint said, and the button exists only while
72
+ * it says `available`. A client must never infer availability from a 100% window.
73
+ *
74
+ * `reason` is the provider's own word for the refusal ("tier", "tenure", "not_at_wall", "weekly_limit",
75
+ * "already_used", …) and is carried rather than translated, because the set is the provider's to extend and a
76
+ * word we don't recognise is still worth showing to somebody asking why the button is not there. */
77
+ export const LimitResetStatusSchema = z.object({
78
+ available: z
79
+ .boolean()
80
+ .describe("Whether the provider will reopen this account's session window right now. The only thing a button may be drawn from."),
81
+ reason: z
82
+ .string()
83
+ .optional()
84
+ .describe("Why not, in the provider's own word, when it gave one. Absent when it is available, or when the provider said nothing."),
85
+ // Both epoch SECONDS, matching every other reset instant on the wire (UsageWindow.resetsAt, limitResetsAt).
86
+ nextAvailableAt: z
87
+ .number()
88
+ .optional()
89
+ .describe("When the next reset may be claimed, in epoch seconds, where the provider publishes it. Absent means unknown, never 'now'."),
90
+ weeklyResetsAt: z.number().optional().describe("When the weekly allowance itself reopens, in epoch seconds, where the provider publishes it."),
91
+ });
92
+ export type LimitResetStatus = z.infer<typeof LimitResetStatusSchema>;
93
+ /* WHAT CLAIMING IT DID, in the provider's own vocabulary plus the two failures that are ours.
94
+ *
95
+ * `reset` is the only outcome that changed anything, and the caller's cue to send the held turn again. The rest
96
+ * are all "nothing happened", and they are kept APART rather than folded into one failure because they are read
97
+ * by somebody who just pressed a button and is owed the difference: `already_used` means come back next week,
98
+ * `not_limited` means the window reopened while they were reading, `ineligible` means this account never had
99
+ * it, and `unavailable`/`error` mean try again. Collapsing them would make every one of those read as a fault.
100
+ *
101
+ * Never throws over the wire: a claim that fails leaves the account exactly as it was, and the honest answer to
102
+ * a press is a word, not a stack trace. */
103
+ export const LimitResetClaimSchema = z.object({
104
+ result: z
105
+ .enum(["reset", "already_used", "not_limited", "ineligible", "unavailable", "error"])
106
+ .describe("What the provider did. Only `reset` reopened the window; every other value means nothing changed."),
107
+ nextAvailableAt: z.number().optional().describe("When another reset may be claimed, in epoch seconds, where the provider published it."),
108
+ detail: z.string().optional().describe("What went wrong, in words, for the two outcomes that are this sandbox's fault rather than the plan's."),
109
+ });
110
+ export type LimitResetClaim = z.infer<typeof LimitResetClaimSchema>;
61
111
  /* THE LAST TIME A PROVIDER ACTUALLY REFUSED A TURN, the other half of "can I run on this", and the half no
62
112
  * meter can supply.
63
113
  *
@@ -64,68 +64,65 @@ export const AccountListQuerySchema = z.object({
64
64
  });
65
65
  // Address one account of a provider (disconnect, and the turn's `account`).
66
66
  export const AccountIdSchema = z.object({ id: z.string().min(1).describe("Which account.") });
67
- // Rename one account of a provider whose credential the sandbox owns (Claude, Kimi). Blank ⇒ the daemon falls
68
- // back to the derived name, so clearing a label restores the sign-in identity rather than leaving a nameless
69
- // row. Grok is absent for the same reason it holds one account: OpenCode owns that credential, not this store.
67
+ // Rename one account of a provider whose credential the sandbox owns. Blank ⇒ the daemon falls back to the
68
+ // derived name, so clearing a label restores the sign-in identity rather than leaving a nameless row.
70
69
  export const AccountRenameSchema = z.object({
71
70
  id: z.string().min(1).describe("Which account."),
72
71
  label: z.string().max(80).describe("The new name. Blank restores the one derived from the sign-in, rather than leaving a nameless row."),
73
72
  });
74
- // The completing calls carry the user-chosen label (blank ⇒ the daemon derives one from the sign-in identity
75
- // or a provider default).
76
- export const OauthExchangeSchema = z.object({
77
- code: z.string().min(1).describe("The code the sign-in handed back."),
78
- verifier: z.string().min(1).describe("The proof from the start of the handshake, which is what stops somebody else's code being redeemed here."),
79
- state: z.string().min(1).describe("The handshake this belongs to. A mismatch is refused."),
80
- label: z.string().optional().describe("What to call the account. Blank derives one from the sign-in."),
81
- });
82
- export const AuthorizeChallengeSchema = z.object({
83
- authorizeUrl: z.string().describe("Where to send somebody to sign in."),
84
- verifier: z.string().describe("Keep this and send it back when finishing. It is what proves the code that comes back belongs to this handshake."),
85
- state: z.string().describe("The handshake's own id, sent back with it."),
86
- });
87
- /* CURSOR'S SIGN-IN START. A third login shape, and the reason it is not one of the two above is where the
88
- * SECRET lives during the handshake.
73
+ /* ONE SIGN-IN SHAPE FOR EVERY ACCOUNT THE SANDBOX ITSELF HOLDS, whatever the vendor's mechanism underneath.
89
74
  *
90
- * Claude's is paste-back: the browser receives a code and the caller hands it plus its verifier to `exchange`,
91
- * so the handshake's proof has to travel on the wire and AuthorizeChallengeSchema carries it. Cursor's PKCE
92
- * verifier must never leave the process that generated it, anyone holding it can redeem the login and mint a
93
- * durable key, so the daemon starts the whole flow, keeps the verifier in memory, polls Cursor itself, and
94
- * writes the account when it lands. Nothing redeemable is on this shape at all.
95
- *
96
- * Which makes it behave like a DEVICE flow from the caller's side (open the page, then watch the account list),
97
- * except that there is no one-time code to display: the login page is addressed to this handshake already. So
98
- * DeviceStartSchema's `code` would be a permanently blank field on every card, and TranslatorStartSchema's
99
- * `state` a value nothing sends back. `handshake` is neither, it is a cancellation handle. */
100
- export const CursorLoginStartSchema = z.object({
101
- url: z.string().describe("The page to open and sign in on. It is already addressed to this attempt, so there is no code to type."),
75
+ * Four handshakes used to answer four shapes, and the shapes differed in where the SECRET lived during the
76
+ * handshake. Anthropic's is paste-back: the browser shows a code and the user brings it here. Cursor's PKCE
77
+ * verifier must never leave the process that generated it (anyone holding it can redeem the login and mint a
78
+ * durable key), so the daemon runs the whole flow and the caller only watches the account list. xAI's is a
79
+ * device code. Meta's and Z.ai's mint the vendor's own key, by a device poll or by a redirect that dead-ends in
80
+ * the user's address bar. Every one of them is now the daemon's to hold: nothing redeemable is on this shape,
81
+ * and the one thing the caller can hand back (a code off a page, or the address a redirect landed on) goes to
82
+ * `complete` with the attempt's handshake. `flow` says how the attempt ENDS, which is the only thing a card
83
+ * cannot infer from the fields: a device sign-in finishes upstream and the account appears on its own; a
84
+ * redirect needs the landing address brought back; a paste needs the code the page showed. */
85
+ export const LoginFlowSchema = z.enum(["device", "redirect", "paste"]);
86
+ export type LoginFlow = z.infer<typeof LoginFlowSchema>;
87
+ export const LoginStartSchema = z.object({
88
+ url: z.string().describe("The page to open and sign in on."),
89
+ code: z.string().describe("The one-time code the page will ask for, where the vendor issues one. Blank when the page is already addressed to this attempt."),
90
+ state: z
91
+ .string()
92
+ .describe("For a redirect sign-in, the marker in the address the browser lands on, so a pasted URL can be recognised as this attempt's. Blank otherwise."),
93
+ flow: LoginFlowSchema.describe(
94
+ "How this attempt ends. A device sign-in finishes by itself and you watch the account list; a redirect needs the address it landed on handed back; a paste needs the code the page showed.",
95
+ ),
96
+ variant: z.string().describe("Which of the provider's estates this attempt signs in to. Blank for a provider with one."),
102
97
  handshake: z
103
98
  .string()
104
- .describe(
105
- "This attempt's id, for abandoning it. Not a credential and not redeemable: the proof that finishes the sign-in never leaves the sandbox.",
106
- ),
99
+ .describe("This attempt's id, for finishing or abandoning it. Not a credential and not redeemable: the proof that completes the sign-in never leaves the sandbox."),
107
100
  expiresAt: z.number().describe("When this attempt stops being answerable, in milliseconds, so a card can stop waiting instead of spinning."),
108
101
  });
109
- export type CursorLoginStart = z.infer<typeof CursorLoginStartSchema>;
110
- // Abandon a sign-in nobody completed, so the daemon stops polling Cursor for it. Ordinary tidiness rather than
111
- // a security boundary: an unanswered attempt also times out on its own (see `expiresAt`).
112
- export const CursorLoginCancelSchema = z.object({ handshake: z.string().min(1).describe("Which attempt to stop waiting on.") });
113
- // xAI Grok (via OpenCode) uses subscription OAuth via the headless device-code method. `start` returns the
114
- // `url` the user opens (xAI's verification_uri_complete, which pre-fills the code) and `code`, the same
115
- // one-time code, surfaced so the card matches x.ai exactly. There is no paste-back: OpenCode polls to
116
- // completion and the UI polls `/grok/accounts`.
117
- // ponytail: OpenCode holds one xAI auth per data dir, so Grok stays single-account, the list is 0 or 1. Per
118
- // account would need an OpenCode server per data dir; add when there's demand.
119
- // A device-code login start: the verification URL + the one-time code the user enters there. The native Grok
120
- // flow (via OpenCode), see TranslatorStartSchema for the routed-provider connect, which adds `state`.
121
- export const DeviceStartSchema = z.object({
122
- url: z.string().describe("The page to open, which already has the code in it."),
123
- code: z
124
- .string()
125
- .describe(
126
- "The one-time code, shown as well so the page and the card say the same thing. Nothing is pasted back: the sandbox waits for the sign-in to complete on its own.",
127
- ),
102
+ export type LoginStart = z.infer<typeof LoginStartSchema>;
103
+ // The estate to sign in to, where a provider has more than one (Z.ai's international and mainland plans). Absent
104
+ // takes the provider's default, which is what a provider with a single estate always sends.
105
+ export const LoginRequestSchema = z.object({
106
+ variant: z.string().min(1).optional().describe("Which estate to sign in to. Absent takes the provider's default."),
107
+ });
108
+ /* The half a person brings back, for the two flows that have one: the code the page showed (a paste), or the
109
+ * whole address a redirect landed on. Not the handshake's state, unlike the translator's version below: that
110
+ * one addresses a session CLIProxyAPI holds, so the state has to travel, while this handshake is held right
111
+ * here and taking the caller's word for its own state would be checking a claim against itself. */
112
+ export const LoginCompleteSchema = z.object({
113
+ handshake: z.string().min(1).describe("Which attempt this belongs to."),
114
+ code: z.string().optional().describe("The code the sign-in page showed, for a paste sign-in."),
115
+ redirectUrl: z.string().optional().describe("The address the browser was sent to, whole, for a redirect sign-in. The grant is inside it."),
116
+ label: z.string().optional().describe("What to call the account. Blank derives one from the sign-in."),
117
+ });
118
+ // What finishing hands back: the account, where the exchange answers with one at once (a paste); absent where
119
+ // the daemon still has a mint to do behind the answer and the row lands in the account list minutes later.
120
+ export const LoginCompletedSchema = z.object({
121
+ account: OauthAccountSchema.optional().describe("The account it connected, where the sign-in ends here. Absent means keep watching the account list."),
128
122
  });
123
+ // Abandon a sign-in nobody completed, so the daemon stops polling the vendor for it. Ordinary tidiness rather
124
+ // than a security boundary: an unanswered attempt also times out on its own (see `expiresAt`).
125
+ export const LoginCancelSchema = z.object({ handshake: z.string().min(1).describe("Which attempt to stop waiting on.") });
129
126
  // A routed-provider subscription login start (codex/grok/kimi/gemini via CLIProxyAPI). Device flows poll to
130
127
  // completion after the user approves upstream; redirect flows need the browser's landing URL pasted back. The
131
128
  // explicit flow discriminator matters even when a provider's verification URL already embeds its optional code.
@@ -1,7 +1,8 @@
1
1
  // settings: per-sandbox agent settings (.intentic/config/settings.json)
2
2
  import { z } from "zod";
3
3
  import { CommandJudgeModeSchema } from "../safety-policy.js";
4
- import { AdmissionPolicySchema, AdmissionRuleSchema, AgentRunPinSchema } from "./agent.js";
4
+ import { ModelRoleSchema } from "../model-roles.js";
5
+ import { AdmissionPolicySchema, AdmissionRuleSchema, ModelPinSchema } from "./agent.js";
5
6
  // Which prompt the agent is, before this turn composes anything on top. Two built-in bases and an escape
6
7
  // hatch: Intentic's own (the default), Claude Code's preset, or the owner's text. Declared out here rather
7
8
  // than inline in the settings object because both sides of the wire branch on it, the daemon to build the
@@ -251,8 +252,6 @@ export const SkillRemoveSchema = z.object({
251
252
  // tool is one daemon-side registry entry, not a new settings field.
252
253
  // hashlineEdits , swaps the native Read/Edit/Write for hash-anchored edits on the Claude path (stale-file
253
254
  // guard + fewer output tokens); off ⇒ the native file tools.
254
- // terseOutput , appends a concise-response steer to the end of the system prompt (a stable suffix, so it
255
- // composes with stableSystemPrompt) to cut the model's OWN output tokens.
256
255
  // systemPromptMode , which base the agent's prompt is: "intentic" (default), "claude", or "custom".
257
256
  // systemPrompt , the owner's own prompt text, used only by "custom" mode, where it is the ENTIRE system
258
257
  // prompt and nothing the daemon would otherwise append rides with it, see its own note.
@@ -264,6 +263,8 @@ export const SkillRemoveSchema = z.object({
264
263
  // workspaceMap , computes an AREA index of the project a run starts in and prepends it to the
265
264
  // conversation's opening message, so the turn does not have to buy its own orientation
266
265
  // with a directory listing. Generated from the filesystem every time, never stored.
266
+ // workspaceMapHoldout, conversation-level measurement control for workspaceMap (UsageTurn.mapArm), judged on
267
+ // the directory listings the opening turn ran rather than on its searches.
267
268
  // sidecars , the background pass converging a markdown shadow of every binary workspace file
268
269
  // (docx/pdf/images/audio → .intentic/local/cache/derived/) the moment it lands, via
269
270
  // the baked fileq CLI, so reasoning-time reads are pre-derived. The CLI itself is
@@ -306,28 +307,24 @@ export const SandboxSettingsSchema = z.object({
306
307
  "Keep the instructions identical between turns so the provider can cache them, moving anything that varies into the message instead. Cheaper, at the cost of some flexibility.",
307
308
  ),
308
309
  skills: z.array(z.string()).default(["lsp", "fileq"]).describe("Which skills are switched on."),
310
+ /* WHICH CONTEXT SHELF A CONVERSATION OPENS ON when nothing closer to it says (schemas/context.ts). A persona
311
+ * card's own `context` wins where a turn wears one; this is the sandbox's answer for the turns that do not.
312
+ * Empty means no shelf, so a conversation carries every repository the workspace has, which is what every
313
+ * conversation did before shelves existed. Empty rather than optional because a settings object is parsed
314
+ * from `{}` until the owner first changes something, and every field here has to answer to that. */
315
+ contextShelf: z
316
+ .string()
317
+ .max(60)
318
+ .default("")
319
+ .describe(
320
+ "Which context shelf a conversation opens on when its persona names none: the part of the workspace it carries. Empty means every repository, as before shelves existed.",
321
+ ),
309
322
  hashlineEdits: z
310
323
  .boolean()
311
324
  .default(false)
312
325
  .describe(
313
326
  "Have the agent edit files by line number rather than by quoting the text it wants replaced. Cheaper on large files, and less forgiving of a stale read.",
314
327
  ),
315
- terseOutput: z.boolean().default(false).describe("Ask the agent to say less. It changes how much it narrates, not how much it does."),
316
- /* Measurement control for the terse steer, at TURN level, the same trick `outputHoldout` plays over
317
- * commands, one layer up. A fraction [0,1] of otherwise-eligible turns run WITHOUT the steer and record
318
- * which arm they ran on (UsageTurn.terse), so the savings report can compare two real populations.
319
- *
320
- * It has to be an experiment: unlike a cleaned command, which yields its own raw baseline in the same
321
- * event, a turn cannot be re-run to see what it would have said unsteered. 0 ⇒ no measurement (every
322
- * eligible turn is steered), which is the default because the control costs the very tokens it measures. */
323
- terseHoldout: z
324
- .number()
325
- .min(0)
326
- .max(1)
327
- .default(0)
328
- .describe(
329
- "What share of turns to run without that instruction, so the two can be compared honestly. It has to be measured this way, because a turn cannot be re-run to see what it would have said. Zero means no measurement, which is the default, since the comparison costs the very tokens it is measuring.",
330
- ),
331
328
  /* WHICH SYSTEM PROMPT THE AGENT RUNS ON, the base, before anything this turn composes.
332
329
  *
333
330
  * intentic. Intentic's own prompt, tuned for this harness (intentic-prompt.ts). The default.
@@ -337,15 +334,14 @@ export const SandboxSettingsSchema = z.object({
337
334
  *
338
335
  * The first two are peers: both get the harness's own guidance appended (the AskUserQuestion/plan blocks
339
336
  * the chat's cards need, the checklist guidance behind the todo panel, the browser-tool guidance), plus the
340
- * delegation note and the terse steer. `custom` is the one that does not, by the owner's explicit choice,
341
- * see the field below. */
337
+ * delegation note. `custom` is the one that does not, by the owner's explicit choice, see the field below. */
342
338
  systemPromptMode: SystemPromptModeSchema.default("intentic").describe(
343
339
  "Which instructions the agent starts from: intentic's own, the ones the installed Claude Code carries, or your own. The first two both get this product's own guidance added on top; your own gets nothing added, which is the point of it.",
344
340
  ),
345
341
  /* The owner's own prompt, used only when `systemPromptMode` is "custom". Then it is the ENTIRE system
346
- * prompt: both built-in bases are gone and so is everything the daemon would otherwise append, the widget
347
- * guidance the chat's cards are driven by, and the terse-output steer (whose toggle goes inert). That is
348
- * the price of total control, and the UI states it at the moment of the edit rather than letting the
342
+ * prompt: both built-in bases are gone and so is everything the daemon would otherwise append, including
343
+ * the widget guidance the chat's cards are driven by. That is the price of total control, and the UI states
344
+ * it at the moment of the edit rather than letting the
349
345
  * widgets go quietly dark. Only the cross-provider delegation note survives, because it has a home outside
350
346
  * the system prompt already (the user-message preamble stableSystemPrompt puts it in).
351
347
  *
@@ -395,6 +391,21 @@ export const SandboxSettingsSchema = z.object({
395
391
  .describe(
396
392
  "Open every conversation with a map of the project it starts in: what is in it, what each part is for, and where the agent is standing. Worked out fresh each time rather than written down anywhere, because a written layout is wrong within a fortnight. Off by default, since it spends tokens on the first message of every conversation.",
397
393
  ),
394
+ /* Measurement control for the map, at CONVERSATION level for the plainest of reasons: the note is sent on
395
+ * the opening message, so every later turn of a mapped conversation has a map in its transcript and a
396
+ * per-turn flip would call eleven treated turns controls.
397
+ *
398
+ * Judged on `UsageTurn.openingListings` and read on each conversation's opening turn (usage/turn-experiments.ts),
399
+ * because that is the turn the note was sent to and averaging it across a long conversation divides the
400
+ * effect by the conversation's length. 0 ⇒ no measurement and every conversation receives the map. */
401
+ workspaceMapHoldout: z
402
+ .number()
403
+ .min(0)
404
+ .max(1)
405
+ .default(0)
406
+ .describe(
407
+ "What share of conversations to open without the map, so the two can be compared. Whole conversations rather than individual turns, because the map is sent once and stays in the conversation's history afterwards.",
408
+ ),
398
409
  /* THE MARKDOWN SHADOWS OF BINARY FILES, the eager half of fileq (_sandbox/fileq). The lazy half — the
399
410
  * `fileq` CLI an agent runs mid-task — is always on PATH and gated only by its skill; this switch is
400
411
  * about the BACKGROUND pass: the daemon watching /work and converging a sidecar under
@@ -448,26 +459,39 @@ export const SandboxSettingsSchema = z.object({
448
459
  .max(1)
449
460
  .default(0)
450
461
  .describe("What share of commands to leave untrimmed, so the saving can be measured against a real comparison rather than estimated."),
451
- /* The models behind the small automatic jobs that are not a conversation, today the commit message
452
- * written when an agent's work lands. An ORDERED list of `${provider}:${modelId}`, tried top to bottom, or
453
- * EMPTY for Auto.
454
- *
455
- * A LIST rather than a pick, because the single interesting failure of this feature is a model that is
456
- * connected and simply will not answer today: the account's allowance went on the chat, and one spent
457
- * provider then takes the job down for hours while the others sit idle. Written in order, the daemon steps
458
- * over the spent one and the message still gets written (agent/quick-model.ts walks it).
459
- *
460
- * Empty is the default and still the interesting case: Auto is resolved from whatever accounts are
461
- * connected at the moment it is read (resolveQuickModels), so it can never name a provider this sandbox has
462
- * no credential for, it improves by itself when one is added, and it is a ladder too, the cheapest rung of
463
- * every connected provider, best first. Storing resolved ids here instead would go stale exactly like a
464
- * pinned model does. */
465
- quickModel: z
466
- .array(z.string())
467
- .max(10)
468
- .default([])
462
+ /* WHICH MODEL DOES WHICH JOB, one ordered list per ROLE (model-roles.ts declares them all).
463
+ *
464
+ * ONE KEY RATHER THAN SEVENTEEN, and the record is keyed by the role catalog rather than by free strings: a
465
+ * job that starts choosing a model tomorrow becomes configurable by adding a row to that table, and the
466
+ * settings page, the resolver and the daemon's lookup all follow without a schema change. Seventeen named
467
+ * fields here would be the same table written a fourth time, in the one place where getting it out of step
468
+ * spends somebody's money.
469
+ *
470
+ * IT REPLACED THREE BUNDLED KEYS — `quickModel`, `agentRunModels` and `commandJudgeModels` — and the reason
471
+ * is worth keeping: the first two were grouped by assumed INTENSITY, not by job. "Quick" covered commit
472
+ * messages, session titles and loop verdicts at once, so an owner who wanted better commit subjects could
473
+ * not ask for them without also moving every session title onto the same model; "agent runs" covered a
474
+ * production incident and a documentation sweep with one tier. An intensity is a guess about work its owner
475
+ * knows better, and neither the configuration nor the UI was actually saving anyone anything by making it.
476
+ *
477
+ * EACH LIST IS AN ORDERED LADDER of pins, tried top to bottom, because the interesting failure is a model
478
+ * that is connected and will not answer today: the account's allowance went on the chat, and one spent
479
+ * provider takes that job down for hours while the others sit idle.
480
+ *
481
+ * AN ABSENT OR EMPTY LIST IS THE INTERESTING CASE and means the role's declared floor (resolveRoleModels): a
482
+ * one-shot helper derives an Auto ladder from whatever is connected right now — so it can never name a
483
+ * provider this sandbox has no credential for, and it improves by itself as accounts are added — while a
484
+ * whole session falls to the model the owner picked for their own chat, because nothing here can judge what
485
+ * a session is worth and a wrong guess is billed whole. Storing resolved ids instead would go stale exactly
486
+ * as a pinned model does. */
487
+ // `partialRecord`, not `record`: an exhaustive one would make every role a required key, so a settings file
488
+ // that has never been touched would have to spell out seventeen empty arrays to be valid, and adding a role
489
+ // would invalidate every settings file in existence. An absent key IS the answer "this role has no list".
490
+ modelRoles: z
491
+ .partialRecord(ModelRoleSchema, z.array(ModelPinSchema).max(10))
492
+ .default({})
469
493
  .describe(
470
- "Which models do the small automatic jobs that are not a conversation, such as writing a commit message. A list rather than one pick, tried in order, because the interesting failure is a model that is connected and simply will not answer today. Empty means work it out from whatever is connected, which improves by itself as accounts are added.",
494
+ "Which models do which job, one ordered list per job: commit messages, session titles, the safety judge, pipeline fixes, and every other place this sandbox picks a model for you. Tried in order, so one spent account does not take a job down. A job with no list falls back to its own default: cheapest connected for the one-shot helpers, your own chat model for whole sessions.",
471
495
  ),
472
496
  /* WHICH REPOS KEEP A CHANGELOG, the repos whose commits carry a `Release-Note:` trailer, written by the
473
497
  * same quick model that drafts the subject (git/commit-message.ts) and harvested at release time.
@@ -489,40 +513,6 @@ export const SandboxSettingsSchema = z.object({
489
513
  .describe(
490
514
  "Which repositories keep a changelog, and so get a user-facing note written alongside each merge. A list rather than a switch, and empty by default, because the commit writer's standing rule is to copy the house style rather than impose one, and a repository that has never written such a note gives it nothing to copy.",
491
515
  ),
492
- /* WHAT AN AGENT RUN OPENS ON, the tier above quickModel, and the answer for every turn a SURFACE starts
493
- * rather than a person at a composer: Fix with agent on a pipeline or a deployment, a Maintenance chore, a
494
- * Documentation or Acceptance run, the fix a failed pre-push check proposes. An ORDERED list of PINS, each
495
- * naming a provider and model AND how that one is to be run (AgentRunPinSchema); EMPTY ⇒ whatever the chat
496
- * composer would have started with, which is the honest floor because it is the model the user already
497
- * chose to work with.
498
- *
499
- * EACH ENTRY CARRIES ITS OWN REASONING AND COST KNOBS, which is why these are objects rather than the
500
- * `${provider}:${model}` keys the two lists around them still hold. The effort used to be one field beside
501
- * the list, answering for every model in it, and the entries of this list are the least interchangeable
502
- * things on the page: the head is the tier the owner wants the work done at and what follows it is the
503
- * account that catches it when the first is spent. AgentRunPinSchema has the rest of the argument.
504
- *
505
- * A LIST, for the reason quickModel is one: the account at the head runs out, and every surface-started run
506
- * in the sandbox then fails on a credential the user cannot see from the row they pressed. Written in order,
507
- * the next one down catches it (turn-resume.ts walks it).
508
- *
509
- * PINNED, NOT DERIVED, the deliberate difference from quickModel one line above, and the reason these are
510
- * two settings rather than one. A quick helper exists to stay OFF the frontier tier, so cheapest-connected
511
- * is the right automatic answer and an empty list resolves to Auto. An agent run has to read a failing
512
- * suite, or a container log, or a story, and repair the thing: the tier is a judgement about how much the
513
- * job is worth, nothing here can make it, and a wrong guess is billed in whole sessions rather than in
514
- * tokens. So an empty list here resolves to NOTHING and the composer's own pick answers instead.
515
- *
516
- * The daemon applies this to any turn flagged `unattended` that names no model of its own, one rule, so a
517
- * surface added tomorrow inherits it by saying what it is instead of re-deriving where models come from. A
518
- * surface MAY still name one (the shared run button's caret, Acceptance's per-run pick), and that wins. */
519
- agentRunModels: z
520
- .array(AgentRunPinSchema)
521
- .max(10)
522
- .default([])
523
- .describe(
524
- "Which models run the work a screen starts rather than a person: fixing a red pipeline, a maintenance chore, an acceptance run. Tried in order, so one spent account does not take every such run down, and each entry says how hard that model should think as well as which one it is. Empty falls back to whatever the chat would have used, which is the honest floor because it is the model you already chose to work with.",
525
- ),
526
516
  /* AUTOMATIC TIER SELECTION: may the daemon run an easy-looking turn on a cheaper rung of the provider the
527
517
  * user is already on, instead of on the model they picked?
528
518
  *
@@ -565,9 +555,9 @@ export const SandboxSettingsSchema = z.object({
565
555
  "How readily a turn counts as simple enough for the cheaper model. It moves only the cutoff: at every setting a turn still has to say something positively easy, so nothing here can downgrade a short vague request.",
566
556
  ),
567
557
  /* WHICH CHEAP MODEL A DOWNGRADED TURN LANDS ON, an ordered list of `${provider}:${model}` keys
568
- * (quickModelKey), or EMPTY for Auto.
558
+ * (modelPinKey), or EMPTY for Auto.
569
559
  *
570
- * Empty is the default and the interesting case, exactly as quickModel's is: Auto is the cheapest row the
560
+ * Empty is the default and the interesting case, exactly as a one-shot role's is: Auto is the cheapest row the
571
561
  * turn's own provider publishes, read through the same cheap-end order (compareCheapestFirst), so the two
572
562
  * features can never disagree about which rung is the cheap one, and connecting an account tomorrow
573
563
  * improves the answer by itself.
@@ -731,35 +721,13 @@ export const SandboxSettingsSchema = z.object({
731
721
  * entitled to say so, and before this they could not: the old rulebook could be set to allow everything, and
732
722
  * the redesign quietly made itself the one part of the sandbox you could only opt further into.
733
723
  *
734
- * Judged commands are also the one automatic job whose model choice genuinely differs from the rest of
735
- * `quickModel`'s work, which is why the list below exists rather than a line in the comment above it: a
736
- * commit message written by a model that misread the diff is a sentence somebody edits, and a verdict
737
- * written by a model that misread a command is a card that should not have been raised or, worse, one that
738
- * should have been. */
724
+ * WHICH MODEL judges is not answered here: it is the `safety-judge` role in `modelRoles`, like every other
725
+ * job in this sandbox that picks one. It used to need its own key, on the argument that a verdict is worth a
726
+ * different model than a commit message — true, and the fact that it had to be argued for one job at a time
727
+ * is exactly what the role catalog replaced. */
739
728
  commandJudge: CommandJudgeModeSchema.default("on").describe(
740
729
  "Whether a model reads your safety policy before a flagged command runs. Off judges nothing and asks about nothing; Watch judges everything and records it without ever interrupting you, which is how you find out what your policy actually does before you let it stop anything; On lets the verdict decide. Wiping a disk or deleting under /history asks at every setting — that rule is typed rather than judged, and cannot be turned off.",
741
730
  ),
742
- /* WHICH MODEL READS THE POLICY, an ordered list of `${provider}:${modelId}` keys (quickModelKey) walked top
743
- * to bottom, or EMPTY for whatever the quick model would be.
744
- *
745
- * A LIST, for the reason every other model setting here is one: the head runs out and the whole feature goes
746
- * with it. Falling back to the quick chain rather than to Auto is deliberate — it is what this did before the
747
- * setting existed, so an owner who never opens the row keeps exactly the behaviour they had, and one who
748
- * writes an entry is saying that command verdicts are worth a different model than commit messages.
749
- *
750
- * THE ARGUMENT FOR SETTING IT AT ALL, since the cheapest connected rung is the default: this prompt is the
751
- * one quick job that is genuinely adversarial. Its input includes text the agent is about to run, which may
752
- * have arrived from a stranger's web page, and a small model can be talked round by it (command-judge.ts is
753
- * candid about that). It is also the job where being WRONG is expensive in both directions — a needless card
754
- * teaches the owner to click through the next one. Neither is a reason for us to spend somebody's frontier
755
- * allowance by default; both are reasons for them to be able to. */
756
- commandJudgeModels: z
757
- .array(z.string())
758
- .max(10)
759
- .default([])
760
- .describe(
761
- "Which models decide whether a flagged command should run, tried in order so one spent account does not take the gate down. Empty uses whatever the quick model is, which is what this did before the setting existed.",
762
- ),
763
731
  /* HOW MUCH AN AGENT MAY DELEGATE, the three ceilings the Claude Code harness enforces on its own Agent
764
732
  * tool, surfaced here because their defaults are tuned for a laptop and this is a container the owner sized.
765
733
  *
@@ -777,7 +745,7 @@ export const SandboxSettingsSchema = z.object({
777
745
  * The refusal an agent sees names the env var (`ask them to raise CLAUDE_CODE_MAX_CONCURRENT_SUBAGENTS`),
778
746
  * which is why these three exist as settings at all: without them the only answer to that ask is editing
779
747
  * the container's environment and restarting the daemon. */
780
- subagentsAtOnce: z.number().min(1).max(200).default(20).describe("How many helper agents may work at the same time."),
748
+ subagentsAtOnce: z.number().min(1).max(200).default(20).describe("How many subagents may work at the same time."),
781
749
  subagentsPerTurn: z.number().min(1).max(2000).default(200).describe("How many a single turn may start in total."),
782
750
  // Depth 1 = an agent may delegate, but its children may not. The CLI's own default is 3, and it is the one
783
751
  // of the three whose runaway case is unbounded rather than merely wide, each level multiplies the last.
@@ -786,7 +754,7 @@ export const SandboxSettingsSchema = z.object({
786
754
  .min(1)
787
755
  .max(10)
788
756
  .default(3)
789
- .describe("How many levels deep the delegation may go, since a helper can start helpers of its own."),
757
+ .describe("How many levels deep the delegation may go, since a subagent can start subagents of its own."),
790
758
  });
791
759
  export type SandboxSettings = z.infer<typeof SandboxSettingsSchema>;
792
760
  // One of the two built-in bases, as text: Intentic's own prompt, or Claude Code's preset read out of the CLI
@@ -800,19 +768,9 @@ export const BuiltinPromptTextSchema = z.object({ text: z.string(), version: z.s
800
768
  export type BuiltinPromptText = z.infer<typeof BuiltinPromptTextSchema>;
801
769
  /* ---- savings report: what each token-reduction mechanism actually saved ----
802
770
  *
803
- * TWO FAMILIES, deliberately never one list of bars. They are measured differently, and a chart that ranks
804
- * them side by side claims a confidence and a denominator that only one of them has:
805
- *
806
- * input , shell output the cleaners trimmed before the model ever saw it. Both sides of the comparison come
807
- * off the SAME command (raw in, emitted out), so the counterfactual is observed rather than
808
- * estimated: exact, per command, no sample size to argue about.
809
- * output, the model's own tokens under the terse steer. There is no second run of the same turn to compare
810
- * against, so the only honest number is an experiment: a turn-level holdout, an n per arm, and a
811
- * margin. It is absent entirely until both arms are large enough for the delta to mean anything.
812
- *
813
- * The two are also in different units of value, a saved tool-output token is saved again on every later
814
- * request of that conversation, an output token is saved once but costs several times as much, which is the
815
- * other reason they are separate sections with separate totals rather than one number.
771
+ * Input-side savings: shell output the cleaners trimmed before the model ever saw it. Both sides of the
772
+ * comparison come off the SAME command (raw in, emitted out), so the counterfactual is observed rather than
773
+ * estimated: exact, per command, no sample size to argue about.
816
774
  */
817
775
 
818
776
  // One mechanism's realized saving, biggest first. `savedTokens` is what THIS stage removed from what reached
@@ -851,15 +809,17 @@ export const SavingsArmSchema = z.object({ turns: z.number(), mean: z.number() }
851
809
  * stand behind. An experiment can carry several, see TurnExperimentSchema.
852
810
  *
853
811
  * `metric` says what `mean` counts and what `deltaPct` is a delta in, and choosing it is most of the work.
854
- * proseChars , the terse steer: the thing it steers, and the only part of the model's output that
855
- * responds to being asked to be brief (UsageTurn.proseChars has why output tokens cannot).
856
812
  * searchCalls , the search teaching: the searches a turn ran, which the teaching directly changes.
857
813
  * openingSearches, the same, narrower: the searches before the turn first touched a file.
858
- * Search mechanisms must not be judged on COST. Cost is a whole turn's work, a search mechanism moves one part
859
- * of it, and the part sits inside the noise of the rest exactly as the steer's effect once sat inside its
860
- * tool-call arguments. */
814
+ * openingListings , the project map: the directory listings a turn ran to work out where it was, which is
815
+ * what the map hands over and what its note tells the turn not to go and fetch.
816
+ * callsBeforeTarget, the map again, on value rather than compliance: how far the turn walked before
817
+ * touching a file it went on to edit.
818
+ * Neither mechanism may be judged on COST. Cost is a whole turn's work, both of these move one part of it, and
819
+ * the part sits inside the noise of the rest. The map's whole payload is about 200 tokens, so a cost reading
820
+ * would be measuring a quantity two orders of magnitude under its own error bar. */
861
821
  export const TurnMetricReadingSchema = z.object({
862
- metric: z.enum(["proseChars", "searchCalls", "openingSearches"]),
822
+ metric: z.enum(["searchCalls", "openingSearches", "openingListings", "callsBeforeTarget"]),
863
823
  on: SavingsArmSchema,
864
824
  off: SavingsArmSchema,
865
825
  /* HOW MUCH LONGER, when the margin spans zero and the honest answer is "keep collecting": the additional
@@ -886,23 +846,20 @@ export const TurnMetricReadingSchema = z.object({
886
846
  /* THE CLAIM, present only once there is one. Both together, and only when the margin does NOT span zero.
887
847
  *
888
848
  * A schema that can't express a half-measured experiment is how a 34%-that-becomes-8%-tomorrow never
889
- * reaches the screen, and clearing `minTurns` turned out not to be enough to buy that. The terse steer
890
- * crossed its thirtieth control turn and immediately reported +31.2% ± 35.1pp: a confidence interval
849
+ * reaches the screen, and clearing `minTurns` turned out not to be enough to buy that. An early search
850
+ * teaching run crossed its thirtieth control turn and immediately reported +31.2% ± 35.1pp: a confidence interval
891
851
  * running from −3.4% to +66.7%, which is to say no effect was measured at all, rendered as an alarming
892
852
  * number pointing the wrong way. Thirty turns is where the normal approximation starts to hold, not where
893
853
  * this much per-turn spread resolves an effect; requiring the interval to exclude zero is the same
894
854
  * withhold-until-it-means-something rule applied to the thing that actually decides whether it does.
895
855
  * deltaPct, change in the metric's mean per turn under the mechanism; negative is a saving.
896
856
  * saved , what the delta is worth over the turns that actually ran with it, in this window, in the
897
- * metric's own unit (characters, or searches). */
857
+ * metric's own unit (searches). */
898
858
  deltaPct: z.number().optional(),
899
859
  saved: z.number().optional(),
900
860
  });
901
861
  export type TurnMetricReading = z.infer<typeof TurnMetricReadingSchema>;
902
- /* A turn-level A/B, the one shape both of this sandbox's turn experiments report in, because they differ in
903
- * nothing but which flag flips and what the turns are judged on. Only turns the mechanism was ELIGIBLE for are
904
- * counted: a turn under a custom system prompt drops the terse steer along with everything else the daemon
905
- * appends, so it belongs to neither arm.
862
+ /* A turn-level A/B, the shape the iq search teaching experiment reports in.
906
863
  *
907
864
  * ONE COIN FLIP, SEVERAL READINGS. `metrics` is a list because the search teaching is judged on two, the
908
865
  * searches a turn ran, and the ones it ran before touching a file, and they are two readings of the SAME
@@ -919,17 +876,22 @@ export const TurnExperimentSchema = z.object({
919
876
  // counts toward the daemon's real threshold instead of a number the browser guessed. Shared by every
920
877
  // reading: they are the same turns counted differently, so they clear it together.
921
878
  minTurns: z.number(),
922
- // The randomized unit behind the arm counts. Turn mechanisms default to turns; teaching loaded into a
923
- // provider session randomizes and analyzes whole conversations so repeated turns are not false replicas.
924
- sampleUnit: z.enum(["turns", "conversations"]).optional(),
879
+ /* The randomized unit behind the arm counts. Turn mechanisms default to turns. Teaching loaded into a
880
+ * provider session randomizes and analyzes whole conversations, so repeated turns are not false replicas.
881
+ *
882
+ * "opening turns" is the third case and belongs to a treatment sent ONCE: the project map rides the first
883
+ * message and nothing after it, so the sample is one turn per conversation rather than an average over
884
+ * the conversation's turns. The distinction is not cosmetic. Averaging a first-turn treatment across a
885
+ * twelve-turn conversation divides its effect by twelve and reports the remainder as noise. */
886
+ sampleUnit: z.enum(["turns", "conversations", "opening turns"]).optional(),
925
887
  // Content-addressed treatment version. Present where mixing rows from two instruction revisions would turn
926
888
  // one experiment into two unnamed ones; the reader filters to this (latest) cohort.
927
889
  cohort: z.string().optional(),
928
890
  });
929
891
  export type TurnExperiment = z.infer<typeof TurnExperimentSchema>;
930
- // `output`/`search` are absent when that experiment isn't running at all (its flag off, or no holdout set), a
931
- // section that isn't there reads as "not measured", which is the truth, while zeros would read as "measured,
932
- // worth nothing".
892
+ // `search` is absent when that experiment isn't running at all (its flag off, or no holdout set), a section
893
+ // that isn't there reads as "not measured", which is the truth, while zeros would read as "measured, worth
894
+ // nothing".
933
895
  /* WHAT THE COMPLEXITY JUDGE HAS BEEN SAYING, read back off the spend ledger's tier fields (UsageTurn.tierScore
934
896
  * and friends) over the requested window. The three numbers docs/model-routing-design.md §4 says the feature
935
897
  * cannot be defended without, plus the veto count, and nothing else: no counterfactual "you would have saved
@@ -964,8 +926,11 @@ export const TierReportSchema = z.object({
964
926
  export type TierReport = z.infer<typeof TierReportSchema>;
965
927
  export const SavingsReportSchema = z.object({
966
928
  input: InputSavingsSchema,
967
- output: TurnExperimentSchema.optional(),
968
929
  search: TurnExperimentSchema.optional(),
930
+ /* The project map's arms, read on opening turns (see TurnExperimentSchema.sampleUnit). Absent under the
931
+ * same rule as `search`: the switch is off, or no holdout is set, and a section that is not there reads as
932
+ * "not measured", which is the truth, where zeros would read as "measured, worth nothing". */
933
+ map: TurnExperimentSchema.optional(),
969
934
  // Automatic tier selection's readout, see TierReportSchema. Absent ⇒ nothing was judged in the window.
970
935
  tier: TierReportSchema.optional(),
971
936
  });