@intentic/sandbox-contract 1.245.0 → 1.247.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (247) hide show
  1. package/README.md +17 -1
  2. package/dist/batch-runs.d.ts +2 -0
  3. package/dist/batch-runs.d.ts.map +1 -1
  4. package/dist/batch-runs.js +1 -0
  5. package/dist/batch-runs.js.map +1 -1
  6. package/dist/command-classes.d.ts +6 -3
  7. package/dist/command-classes.d.ts.map +1 -1
  8. package/dist/command-classes.js +43 -18
  9. package/dist/command-classes.js.map +1 -1
  10. package/dist/contracts/{cursor.contract.d.ts → accounts.contract.d.ts} +102 -3
  11. package/dist/contracts/accounts.contract.d.ts.map +1 -0
  12. package/dist/contracts/accounts.contract.js +61 -0
  13. package/dist/contracts/accounts.contract.js.map +1 -0
  14. package/dist/contracts/agent.contract.d.ts +19 -0
  15. package/dist/contracts/agent.contract.d.ts.map +1 -1
  16. package/dist/contracts/agents.contract.d.ts +121 -0
  17. package/dist/contracts/agents.contract.d.ts.map +1 -1
  18. package/dist/contracts/agents.contract.js +4 -4
  19. package/dist/contracts/agents.contract.js.map +1 -1
  20. package/dist/contracts/ci.contract.d.ts +2 -0
  21. package/dist/contracts/ci.contract.d.ts.map +1 -1
  22. package/dist/contracts/host.contract.d.ts +35 -0
  23. package/dist/contracts/host.contract.d.ts.map +1 -1
  24. package/dist/contracts/host.contract.js +3 -2
  25. package/dist/contracts/host.contract.js.map +1 -1
  26. package/dist/contracts/personas.contract.d.ts +4 -2
  27. package/dist/contracts/personas.contract.d.ts.map +1 -1
  28. package/dist/contracts/runner.contract.d.ts +2 -2
  29. package/dist/contracts/settings.contract.d.ts +58 -20
  30. package/dist/contracts/settings.contract.d.ts.map +1 -1
  31. package/dist/contracts/system.contract.d.ts +80 -30
  32. package/dist/contracts/system.contract.d.ts.map +1 -1
  33. package/dist/contracts/system.contract.js +26 -17
  34. package/dist/contracts/system.contract.js.map +1 -1
  35. package/dist/contracts/usage.contract.d.ts +22 -0
  36. package/dist/contracts/usage.contract.d.ts.map +1 -1
  37. package/dist/contracts/usage.contract.js +19 -0
  38. package/dist/contracts/usage.contract.js.map +1 -1
  39. package/dist/definition.d.ts +20 -28
  40. package/dist/definition.d.ts.map +1 -1
  41. package/dist/documents.d.ts +0 -1
  42. package/dist/documents.d.ts.map +1 -1
  43. package/dist/documents.js +1 -2
  44. package/dist/documents.js.map +1 -1
  45. package/dist/embed.d.ts +23 -0
  46. package/dist/embed.d.ts.map +1 -0
  47. package/dist/embed.js +84 -0
  48. package/dist/embed.js.map +1 -0
  49. package/dist/events.d.ts +21 -0
  50. package/dist/events.d.ts.map +1 -1
  51. package/dist/events.js +5 -2
  52. package/dist/events.js.map +1 -1
  53. package/dist/fast-tier.js +1 -1
  54. package/dist/fast-tier.js.map +1 -1
  55. package/dist/history-state.d.ts.map +1 -1
  56. package/dist/history-state.js +2 -0
  57. package/dist/history-state.js.map +1 -1
  58. package/dist/index.d.ts +453 -306
  59. package/dist/index.d.ts.map +1 -1
  60. package/dist/index.js +7 -15
  61. package/dist/index.js.map +1 -1
  62. package/dist/model-pins.d.ts +17 -0
  63. package/dist/model-pins.d.ts.map +1 -0
  64. package/dist/{quick-model.js → model-pins.js} +18 -11
  65. package/dist/model-pins.js.map +1 -0
  66. package/dist/model-roles.d.ts +144 -0
  67. package/dist/model-roles.d.ts.map +1 -0
  68. package/dist/model-roles.js +129 -0
  69. package/dist/model-roles.js.map +1 -0
  70. package/dist/peer-dial.d.ts +33 -0
  71. package/dist/peer-dial.d.ts.map +1 -0
  72. package/dist/peer-dial.js +79 -0
  73. package/dist/peer-dial.js.map +1 -0
  74. package/dist/peer-mcp-server.d.ts +36 -0
  75. package/dist/peer-mcp-server.d.ts.map +1 -0
  76. package/dist/peer-mcp-server.js +71 -0
  77. package/dist/peer-mcp-server.js.map +1 -0
  78. package/dist/provider-specs.d.ts +38 -20
  79. package/dist/provider-specs.d.ts.map +1 -1
  80. package/dist/provider-specs.js +39 -13
  81. package/dist/provider-specs.js.map +1 -1
  82. package/dist/runtime-state.d.ts +1 -1
  83. package/dist/runtime-state.js +1 -1
  84. package/dist/runtime-state.js.map +1 -1
  85. package/dist/safety-policy.d.ts +12 -3
  86. package/dist/safety-policy.d.ts.map +1 -1
  87. package/dist/safety-policy.js +30 -5
  88. package/dist/safety-policy.js.map +1 -1
  89. package/dist/schemas/agent.d.ts +27 -8
  90. package/dist/schemas/agent.d.ts.map +1 -1
  91. package/dist/schemas/agent.js +10 -4
  92. package/dist/schemas/agent.js.map +1 -1
  93. package/dist/schemas/agents.d.ts +42 -0
  94. package/dist/schemas/agents.d.ts.map +1 -1
  95. package/dist/schemas/agents.js +25 -4
  96. package/dist/schemas/agents.js.map +1 -1
  97. package/dist/schemas/automations.d.ts +11 -2
  98. package/dist/schemas/automations.d.ts.map +1 -1
  99. package/dist/schemas/automations.js +1 -1
  100. package/dist/schemas/automations.js.map +1 -1
  101. package/dist/schemas/ci.d.ts +6 -0
  102. package/dist/schemas/ci.d.ts.map +1 -1
  103. package/dist/schemas/ci.js +3 -2
  104. package/dist/schemas/ci.js.map +1 -1
  105. package/dist/schemas/context.d.ts +30 -0
  106. package/dist/schemas/context.d.ts.map +1 -0
  107. package/dist/schemas/context.js +34 -0
  108. package/dist/schemas/context.js.map +1 -0
  109. package/dist/schemas/{computers.d.ts → devices.d.ts} +154 -60
  110. package/dist/schemas/devices.d.ts.map +1 -0
  111. package/dist/schemas/devices.js +157 -0
  112. package/dist/schemas/devices.js.map +1 -0
  113. package/dist/schemas/hosts.d.ts +12 -0
  114. package/dist/schemas/hosts.d.ts.map +1 -1
  115. package/dist/schemas/hosts.js +1 -0
  116. package/dist/schemas/hosts.js.map +1 -1
  117. package/dist/schemas/issues.d.ts +0 -5
  118. package/dist/schemas/issues.d.ts.map +1 -1
  119. package/dist/schemas/issues.js +0 -1
  120. package/dist/schemas/issues.js.map +1 -1
  121. package/dist/schemas/personas.d.ts +5 -3
  122. package/dist/schemas/personas.d.ts.map +1 -1
  123. package/dist/schemas/personas.js +3 -2
  124. package/dist/schemas/personas.js.map +1 -1
  125. package/dist/schemas/plan-limits.d.ts +20 -0
  126. package/dist/schemas/plan-limits.d.ts.map +1 -1
  127. package/dist/schemas/plan-limits.js +21 -0
  128. package/dist/schemas/plan-limits.js.map +1 -1
  129. package/dist/schemas/provider-oauth.d.ts +48 -16
  130. package/dist/schemas/provider-oauth.d.ts.map +1 -1
  131. package/dist/schemas/provider-oauth.js +22 -20
  132. package/dist/schemas/provider-oauth.js.map +1 -1
  133. package/dist/schemas/settings.d.ts +42 -16
  134. package/dist/schemas/settings.d.ts.map +1 -1
  135. package/dist/schemas/settings.js +22 -28
  136. package/dist/schemas/settings.js.map +1 -1
  137. package/dist/schemas/terminal.js +9 -9
  138. package/dist/schemas/terminal.js.map +1 -1
  139. package/dist/schemas/usage.d.ts +5 -2
  140. package/dist/schemas/usage.d.ts.map +1 -1
  141. package/dist/schemas/usage.js +5 -2
  142. package/dist/schemas/usage.js.map +1 -1
  143. package/dist/shell-regions.d.ts +4 -0
  144. package/dist/shell-regions.d.ts.map +1 -0
  145. package/dist/shell-regions.js +156 -0
  146. package/dist/shell-regions.js.map +1 -0
  147. package/dist/workspace-state.d.ts +8 -0
  148. package/dist/workspace-state.d.ts.map +1 -1
  149. package/dist/workspace-state.js +13 -5
  150. package/dist/workspace-state.js.map +1 -1
  151. package/package.json +37 -4
  152. package/src/agent-catalog.ts +2 -2
  153. package/src/arrival.ts +3 -3
  154. package/src/batch-runs.test.ts +10 -5
  155. package/src/batch-runs.ts +10 -3
  156. package/src/command-classes.test.ts +195 -71
  157. package/src/command-classes.ts +148 -46
  158. package/src/contracts/accounts.contract.ts +94 -0
  159. package/src/contracts/agents.contract.ts +4 -3
  160. package/src/contracts/exit.contract.ts +2 -2
  161. package/src/contracts/host.contract.ts +17 -5
  162. package/src/contracts/settings.contract.ts +1 -1
  163. package/src/contracts/system.contract.ts +43 -24
  164. package/src/contracts/usage.contract.ts +31 -0
  165. package/src/contracts/vpn.contract.ts +2 -2
  166. package/src/documents.test.ts +2 -1
  167. package/src/documents.ts +7 -11
  168. package/src/embed.test.ts +68 -0
  169. package/src/embed.ts +164 -0
  170. package/src/events.ts +31 -4
  171. package/src/fast-tier.test.ts +1 -1
  172. package/src/fast-tier.ts +5 -5
  173. package/src/history-state.ts +12 -3
  174. package/src/host-protocol.ts +2 -2
  175. package/src/index.ts +8 -16
  176. package/src/model-order.ts +1 -1
  177. package/src/{quick-model.test.ts → model-pins.test.ts} +73 -29
  178. package/src/model-pins.ts +183 -0
  179. package/src/model-roles.ts +224 -0
  180. package/src/peer-dial.test.ts +203 -0
  181. package/src/peer-dial.ts +163 -0
  182. package/src/peer-mcp-server.test.ts +104 -0
  183. package/src/peer-mcp-server.ts +144 -0
  184. package/src/plan-pools.ts +1 -1
  185. package/src/prompt-complexity.test.ts +1 -1
  186. package/src/prompt-complexity.ts +2 -2
  187. package/src/provider-specs.test.ts +45 -18
  188. package/src/provider-specs.ts +147 -67
  189. package/src/routes.test.ts +6 -3
  190. package/src/runner-protocol.ts +1 -1
  191. package/src/runtime-state.ts +2 -2
  192. package/src/safety-policy.test.ts +88 -0
  193. package/src/safety-policy.ts +84 -14
  194. package/src/schemas/agent.ts +83 -30
  195. package/src/schemas/agents.ts +67 -6
  196. package/src/schemas/automations.ts +6 -4
  197. package/src/schemas/capabilities.ts +4 -4
  198. package/src/schemas/ci.ts +23 -6
  199. package/src/schemas/context.ts +87 -0
  200. package/src/schemas/{computers.ts → devices.ts} +190 -107
  201. package/src/schemas/hosts.ts +5 -1
  202. package/src/schemas/issues.ts +0 -4
  203. package/src/schemas/personas.ts +8 -3
  204. package/src/schemas/plan-limits.ts +50 -0
  205. package/src/schemas/provider-oauth.ts +49 -52
  206. package/src/schemas/settings.ts +105 -140
  207. package/src/schemas/terminal.ts +12 -12
  208. package/src/schemas/usage.ts +62 -27
  209. package/src/schemas/version-seam.test.ts +0 -1
  210. package/src/shell-regions.ts +289 -0
  211. package/src/versions.ts +2 -2
  212. package/src/webext-links.ts +2 -2
  213. package/src/webext-protocol.ts +2 -2
  214. package/src/workspace-state.test.ts +48 -1
  215. package/src/workspace-state.ts +48 -11
  216. package/dist/agent-run-model.d.ts +0 -4
  217. package/dist/agent-run-model.d.ts.map +0 -1
  218. package/dist/agent-run-model.js +0 -13
  219. package/dist/agent-run-model.js.map +0 -1
  220. package/dist/contracts/claude.contract.d.ts +0 -91
  221. package/dist/contracts/claude.contract.d.ts.map +0 -1
  222. package/dist/contracts/claude.contract.js +0 -50
  223. package/dist/contracts/claude.contract.js.map +0 -1
  224. package/dist/contracts/cursor.contract.d.ts.map +0 -1
  225. package/dist/contracts/cursor.contract.js +0 -50
  226. package/dist/contracts/cursor.contract.js.map +0 -1
  227. package/dist/contracts/grok.contract.d.ts +0 -36
  228. package/dist/contracts/grok.contract.d.ts.map +0 -1
  229. package/dist/contracts/grok.contract.js +0 -31
  230. package/dist/contracts/grok.contract.js.map +0 -1
  231. package/dist/contracts/keys.contract.d.ts +0 -81
  232. package/dist/contracts/keys.contract.d.ts.map +0 -1
  233. package/dist/contracts/keys.contract.js +0 -51
  234. package/dist/contracts/keys.contract.js.map +0 -1
  235. package/dist/quick-model.d.ts +0 -15
  236. package/dist/quick-model.d.ts.map +0 -1
  237. package/dist/quick-model.js.map +0 -1
  238. package/dist/schemas/computers.d.ts.map +0 -1
  239. package/dist/schemas/computers.js +0 -135
  240. package/dist/schemas/computers.js.map +0 -1
  241. package/src/agent-run-model.test.ts +0 -76
  242. package/src/agent-run-model.ts +0 -65
  243. package/src/contracts/claude.contract.ts +0 -71
  244. package/src/contracts/cursor.contract.ts +0 -74
  245. package/src/contracts/grok.contract.ts +0 -41
  246. package/src/contracts/keys.contract.ts +0 -79
  247. package/src/quick-model.ts +0 -155
@@ -194,14 +194,14 @@ export type SubagentKind = z.infer<typeof SubagentKindSchema>;
194
194
  // an operator acts on differently from "the child is working".
195
195
  export const SubagentStatusSchema = z.enum(["pending", "running", "blocked", "completed", "failed", "killed", "paused"]);
196
196
  export type SubagentStatus = z.infer<typeof SubagentStatusSchema>;
197
- /* WHETHER ANYTHING CHECKED WHAT THE HELPER DID, carried beside its report rather than left for the reader to
198
- * assume. Computed from the helper's own tool calls, the files it edited against the checks that ran after
197
+ /* WHETHER ANYTHING CHECKED WHAT THE SUBAGENT DID, carried beside its report rather than left for the reader to
198
+ * assume. Computed from the subagent's own tool calls, the files it edited against the checks that ran after
199
199
  * them (the daemon's child-verification.ts), so it holds on every provider rather than only where the Claude
200
200
  * hooks reach.
201
201
  *
202
202
  * The four states are deliberately not two. `verified` and `failing` each name the command that spoke, so a
203
203
  * targeted test is never read as the suite; `unproven` is the one that matters most, work changed and nothing
204
- * ran; and `no-code` says the helper edited nothing, which is the honest answer for a research helper and
204
+ * ran; and `no-code` says the subagent edited nothing, which is the honest answer for a research subagent and
205
205
  * must not be rendered as approval. Absent ⇒ the daemon saw no tool calls from it at all. */
206
206
  export const SubagentVerificationSchema = z.object({
207
207
  state: z
@@ -222,10 +222,10 @@ export const SubagentSessionSchema = z.object({
222
222
  id: z
223
223
  .string()
224
224
  .describe(
225
- "The id of the tool call that started it (an SDK child) or the child's own conversation id (a spawned one); either way both sides already hold it, so a card links to its helper with the id it has and the helper points back the same way.",
225
+ "The id of the tool call that started it (an SDK child) or the child's own conversation id (a spawned one); either way both sides already hold it, so a card links to its subagent with the id it has and the subagent points back the same way.",
226
226
  ),
227
227
  kind: SubagentKindSchema.describe(
228
- "What sort of helper: one the runtime's own Task tool spawned in-process, or a full agent the daemon started for the turn. It changes only how you watch it.",
228
+ "What sort of subagent: one the runtime's own Task tool spawned in-process, or a full child agent the daemon started for the turn. It changes only how you watch it.",
229
229
  ),
230
230
  // The conversation whose turn spawned this, what the area groups its rows by, and the way back to the chat
231
231
  // the card lives in.
@@ -233,19 +233,19 @@ export const SubagentSessionSchema = z.object({
233
233
  // What it is and what it was asked to do: the subagent type (`Explore`, `general-purpose`) or a spawned
234
234
  // child's provider label, and the caller's one-line description. The area's row and the card's title read
235
235
  // as `Explore · Locate claimIndexer definition`.
236
- agentType: z.string().optional().describe("What kind of helper it is."),
236
+ agentType: z.string().optional().describe("What kind of subagent it is."),
237
237
  description: z.string().optional().describe("What it was asked to do, in one line."),
238
238
  model: z.string().optional().describe("Which model it runs on."),
239
239
  // Which provider serves a `spawned` child (its AgentProvider id), so the row can wear the right logo. An
240
240
  // SDK subagent implies its own: it runs on its parent's provider.
241
- provider: z.string().optional().describe("Which provider serves it, for a helper spawned across providers."),
241
+ provider: z.string().optional().describe("Which provider serves it, for a child agent spawned across providers."),
242
242
  // How deep in the spawn tree (1 = spawned by the turn itself). From the SDK's meta.json; a subagent may
243
243
  // itself delegate, and a flat list that cannot say so reads as though the turn started all of them.
244
244
  spawnDepth: z
245
245
  .number()
246
246
  .optional()
247
247
  .describe(
248
- "How deep in the chain it sits, where one means the turn itself started it. A helper can start helpers, and a flat list that could not say so would read as though the turn started all of them.",
248
+ "How deep in the chain it sits, where one means the turn itself started it. A subagent can start subagents, and a flat list that could not say so would read as though the turn started all of them.",
249
249
  ),
250
250
  // Backgrounded: the parent went on working instead of waiting for it. This is the whole reason the list
251
251
  // exists, a backgrounded child used to be invisible until its result landed, sometimes minutes later.
@@ -253,7 +253,7 @@ export const SubagentSessionSchema = z.object({
253
253
  .boolean()
254
254
  .optional()
255
255
  .describe(
256
- "The parent carried on working instead of waiting for it. This is the whole reason the list exists: such a helper used to be invisible until its result landed, sometimes minutes later.",
256
+ "The parent carried on working instead of waiting for it. This is the whole reason the list exists: such a subagent used to be invisible until its result landed, sometimes minutes later.",
257
257
  ),
258
258
  status: SubagentStatusSchema.describe(
259
259
  "How it is going. Blocked means it needs an answer, which a parent and an operator act on differently from it simply working.",
@@ -266,7 +266,7 @@ export const SubagentSessionSchema = z.object({
266
266
  tokens: z
267
267
  .number()
268
268
  .optional()
269
- .describe("What it has spent. Its own, so a parent's cost and the sum of its helpers' are two different true numbers."),
269
+ .describe("What it has spent. Its own, so a parent's cost and the sum of its subagents' are two different true numbers."),
270
270
  toolUses: z.number().optional().describe("How many tools it has used."),
271
271
  lastTool: z.string().optional().describe("The last one it reached for."),
272
272
  // Its report, the last assistant message (SubagentStop) or the task summary. The answer to "what did it
@@ -274,7 +274,7 @@ export const SubagentSessionSchema = z.object({
274
274
  summary: z
275
275
  .string()
276
276
  .optional()
277
- .describe("Its report: what it concluded, without opening its record. The question a finished helper gets read for."),
277
+ .describe("Its report: what it concluded, without opening its record. The question a finished subagent gets read for."),
278
278
  error: z.string().optional().describe("Why it failed, when it did."),
279
279
  // Whether anything checked the work behind that report (SubagentVerificationSchema). Filled once it ends:
280
280
  // a standing read while it is still working would be a verdict on a job half done.
@@ -282,7 +282,7 @@ export const SubagentSessionSchema = z.object({
282
282
  });
283
283
  export type SubagentSession = z.infer<typeof SubagentSessionSchema>;
284
284
  export const SubagentsListSchema = z.object({
285
- sessions: z.array(SubagentSessionSchema).describe("Every helper this sandbox's conversations have started."),
285
+ sessions: z.array(SubagentSessionSchema).describe("Every subagent and child agent this sandbox's conversations have started."),
286
286
  });
287
287
  export type SubagentsList = z.infer<typeof SubagentsListSchema>;
288
288
  export const SubagentIdParamSchema = z.object({ id: z.string() });
@@ -83,14 +83,6 @@ export const UsageTurnSchema = z.object({
83
83
  cacheCreationTokens: z.number().describe("Tokens written to cache, which cost more up front and less afterwards."),
84
84
  costUsd: z.number().describe("What it cost, in dollars."),
85
85
  durationMs: z.number().describe("How long it took, in milliseconds."),
86
- /* Which arm of the terse experiment this turn ran on (settings.terseHoldout), the only record of it, and
87
- * the reason the savings report can say what the steer is worth instead of guessing.
88
- *
89
- * ABSENT means "not part of the experiment", not "off": a turn under a custom system prompt drops the
90
- * steer along with everything else the daemon appends, and a turn run with the experiment switched off has
91
- * no control to be compared against. Pooling those into the off-arm would compare steered turns against a
92
- * population selected by something other than the coin flip, which is not a control at all. */
93
- terse: z.boolean().optional(),
94
86
  /* Which arm of the iq SEARCH-TEACHING experiment this conversation runs on
95
87
  * (settings.iqSearchHoldout). Stable for every turn in one conversation: the treatment is instruction
96
88
  * loaded into a provider session, so flipping it per turn would call a remembered treatment a control.
@@ -99,25 +91,9 @@ export const UsageTurnSchema = z.object({
99
91
  // Hash of the plugin nudge + skill body used for this arm. Control turns carry it too, so a report can keep
100
92
  // both sides of one treatment revision together and exclude older wording after an upgrade.
101
93
  iqSearchCohort: z.string().optional(),
102
- /* Characters of the model's own PROSE this turn, the `delta` frames only, so no tool-call arguments and no
103
- * thinking. What the terse steer is judged on, and the reason it can be judged at all.
104
- *
105
- * `outputTokens` cannot serve: measured over a day of real turns it is 91.6% tool-call arguments (an Edit's
106
- * old_string and new_string, a Write's whole file body) and 7.8% prose. The steer moves prose. So a fifth
107
- * off the model's narration moves the total by 1.6%, against a margin of ±35 points, which is to say the
108
- * experiment was structurally unable to see its own treatment, and the number it printed instead was
109
- * whichever arm happened to draw the bigger tasks.
110
- *
111
- * CHARACTERS, not tokens, because the provider bills a total and never breaks it down, a token figure here
112
- * would be chars÷4 wearing a unit it had not earned. For a comparison of two arms the constant cancels
113
- * anyway, and the honest unit is the one actually counted.
114
- *
115
- * Absent ⇒ the turn predates this being measured; `armOf` drops it from the population rather than reading
116
- * it as a silent turn. */
117
- proseChars: z.number().optional(),
118
94
  /* SEARCHES THIS TURN RAN, every tool call that went looking for code, the dedicated search tools and the
119
95
  * CLI searches alike (isSearchCall owns the rule; `iq q` is Bash and would otherwise not be counted at all).
120
- * What the search teaching is judged on, and the same correction `proseChars` is to the terse steer.
96
+ * What the search teaching is judged on.
121
97
  *
122
98
  * COST PER TURN CANNOT SERVE: cost is a whole turn's worth of work, a search mechanism touches one part of
123
99
  * it, and the part lives inside the noise of the rest, exactly the shape that made output tokens unable to
@@ -128,8 +104,8 @@ export const UsageTurnSchema = z.object({
128
104
  * rather than being filtered out, they dilute both arms equally, while selecting on "did it search" would
129
105
  * select on the treatment itself.
130
106
  *
131
- * Absent ⇒ the turn predates this being measured; `armOf` drops it rather than reading it as a turn that
132
- * searched nothing. */
107
+ * Absent ⇒ the turn predates this being measured; the arm's mean drops it (turn-experiments.ts, the filter
108
+ * inside `conversationExperimentOf`) rather than reading it as a turn that searched nothing. */
133
109
  searchCalls: z.number().optional(),
134
110
  /* …and how many of them came BEFORE the turn first opened or changed a file, the orientation burst. A turn
135
111
  * that already knows where to look starts working; one that doesn't goes hunting first.
@@ -143,6 +119,65 @@ export const UsageTurnSchema = z.object({
143
119
  *
144
120
  * Absent ⇒ as for `searchCalls`. */
145
121
  openingSearches: z.number().optional(),
122
+ /* THE LISTINGS THIS TURN RAN TO WORK OUT WHERE IT WAS, `ls`, `ls /work`, `tree intentic`, the LS tool
123
+ * aimed at the same places (isRootListing owns the rule). What the project map is judged on, and the
124
+ * reason it could not be judged before.
125
+ *
126
+ * `searchCalls` counts a directory listing and a ripgrep as one event, deliberately, so that a taxonomy
127
+ * cannot report whichever spelling of a search the model happened to reach for. That is right for the
128
+ * search teaching and blind to the map: measured over 468 mapped sessions of this workspace against 497
129
+ * unmapped ones, searches before the first file moved +7.6% with a margin of ±17.6pp, while the share of
130
+ * sessions opening with a directory listing fell from 46.3% to 32.1%. The map does not shorten the
131
+ * orientation burst, it changes what the burst is made of, and only this counts the difference.
132
+ *
133
+ * UP TO THE FIRST FILE, exactly like `openingSearches`, which took the corpus to settle. Counted over the
134
+ * whole turn instead, the same sessions give 66.4% against 73.3%, a gap of 6.9pp where the orientation
135
+ * window gives 14.2pp. The dilution is not noise: a turn already at work lists the directory it has
136
+ * narrowed to, and no map could have answered that. The shape of the listing cannot tell the two apart,
137
+ * because it is the same shape; only when it happened can.
138
+ *
139
+ * Absent ⇒ the turn predates this being measured, never a turn that listed nothing. */
140
+ openingListings: z.number().optional(),
141
+ /* TOOL CALLS BEFORE THE TURN FIRST TOUCHED A FILE IT WENT ON TO EDIT: how far it walked before reaching
142
+ * the thing it turned out to be looking for. The corpus study this map was designed from measured the
143
+ * same quantity by hand (the workspace's docs/agent-exploration-patterns.md §4: median 4, mean 8.2) and called it the one
144
+ * honest reading of whether orientation got better.
145
+ *
146
+ * IT CAN ONLY BE KNOWN AT THE END, which is why it is recorded here and computed nowhere else: whether a
147
+ * file was the target is a fact about the turn's edits, and the turn has to finish before that is
148
+ * decided. The route keeps the first call index per path and intersects it with the edit ledger.
149
+ *
150
+ * Absent on a turn that edited nothing, which is most short turns, and NOT zero: a turn with no target
151
+ * never reached one, and averaging that in as "reached it immediately" would report the turns that did
152
+ * no work as the best targeted. That absence costs the reading three quarters of the population, so it
153
+ * is a metric to accumulate for months rather than a gate to wait on. */
154
+ callsBeforeTarget: z.number().optional(),
155
+ /* WHICH ARM OF THE PROJECT MAP EXPERIMENT this conversation ran (settings.workspaceMapHoldout), and how
156
+ * many characters the note actually cost when it was sent.
157
+ *
158
+ * CONVERSATION-STABLE, and for a plainer reason than the search teaching's: the map is sent once, on the
159
+ * opening message, so every later turn of a mapped conversation is a turn whose transcript holds a map.
160
+ * A per-turn flip would label eleven treated turns as controls.
161
+ *
162
+ * `mapChars` rather than a token estimate, because characters are what the renderer's budget is in
163
+ * (workspace-map.ts caps the note at 2,800) and a tokenizer's answer would vary by model. Present only on
164
+ * the turn that actually sent one, so the ledger says what the feature costs instead of assuming it: the
165
+ * median note over this workspace's own corpus is 795 characters against a ceiling of 2,800.
166
+ *
167
+ * Absent ⇒ no measurement (the holdout is zero, or the row predates it); true/false ⇒ mapped/unmapped. */
168
+ mapArm: z.boolean().optional(),
169
+ mapChars: z.number().optional(),
170
+ /* WHICH TURN OF ITS CONVERSATION THIS WAS, counting from zero, so an opening turn can be recognised from
171
+ * one row instead of inferred from the rows around it.
172
+ *
173
+ * The inference it replaces is wrong at exactly one place and it is the place that matters: a reader
174
+ * windowed to the last seven days would take each conversation's earliest row IN THE WINDOW as its
175
+ * opening turn, so every conversation that started the week before would contribute a mid-conversation
176
+ * turn to a reading about first turns. The project map is sent on turn zero and nowhere else, so that
177
+ * misreading is not an edge case for it, it is the measurement.
178
+ *
179
+ * Absent ⇒ the row predates this, or the turn belonged to no conversation at all. */
180
+ turnIndex: z.number().optional(),
146
181
  /* DID THIS TURN FINISH, OR DID IT STOP TALKING. The fields that tell the two apart, and the reason
147
182
  * `outcome` alone could never.
148
183
  *
@@ -15,7 +15,6 @@ test("a payload from a build that predates a toggle parses, with the new toggle
15
15
  stableSystemPrompt: false,
16
16
  skills: [],
17
17
  hashlineEdits: false,
18
- terseOutput: true,
19
18
  iqSearch: true,
20
19
  outputCleaners: "-cap",
21
20
  outputHoldout: 0.1,
@@ -0,0 +1,289 @@
1
+ /* WHICH PARTS OF A COMMAND ARE TEXT RATHER THAN A PROGRAM, so the one tier that cannot be argued with stops
2
+ * firing on a mention of a dangerous verb.
3
+ *
4
+ * WHAT THIS IS FOR, precisely, because it bounds how good it has to be. The triage catalog next door
5
+ * (command-classes.ts) is deliberately over-inclusive: a match only means a judge should look, and a false
6
+ * positive there costs one model call. That economy holds for every class except the hard-ruled one, where a
7
+ * match is an interruption no policy and no verdict can waive (safety-policy.ts hardRuleClasses). So
8
+ * `echo "rm -rf /" >> notes.md` and `rg 'rm -rf /'` raised un-waivable cards over a string being written to a
9
+ * file and a search of the tree — the exact failure the judge redesign was built to end, surviving in the one
10
+ * tier the judge cannot reach.
11
+ *
12
+ * This says where a fragment sits. A fragment inside a region below is still REPORTED (the class holds, the
13
+ * judge still reads it, the card still marks it); it just does not trip the hard rule.
14
+ *
15
+ * WHICH IS WHY A REGEX-LEVEL SCANNER IS ENOUGH, and this is the design argument rather than an excuse. Both
16
+ * ways of being wrong are cheap:
17
+ *
18
+ * · MISS a region (call text a program) ⇒ one judge call. Exactly today's behaviour, which is the floor.
19
+ * · INVENT a region (call a program text) ⇒ the class is still reported and the judge still rules on it.
20
+ * Only the un-waivable tier is skipped, and the judge is the tier that reads the owner's policy.
21
+ *
22
+ * Precision has to be good enough to skip tier 1½, never good enough to skip tier 1. Nothing here is a
23
+ * boundary, for the same reason nothing in command-classes.ts is: `sh -c "$CMD"` and a path assembled from a
24
+ * variable walk past all of it. The boundaries are structural and elsewhere.
25
+ *
26
+ * THREE KINDS OF REGION, and each is a place a shell will not run what it holds:
27
+ *
28
+ * 1 A COMMENT, `#` to end of line.
29
+ * 2 A HEREDOC BODY, the usual way an agent writes a script it is not running yet.
30
+ * 3 A QUOTED ARGUMENT OF A VERB THAT CANNOT EXECUTE ONE — echo, printf, and the searchers. Plus a quoted
31
+ * commit message after -m, whatever the verb, because message text never runs.
32
+ *
33
+ * `sed`, `awk` and `perl` are deliberately NOT in that verb list, though they are the obvious next entries:
34
+ * each can run a shell out of its own quoted program (`awk 'BEGIN{system("…")}'`, `perl -e`, GNU sed's `s///e`),
35
+ * so their quoted argument is a program and calling it text would be wrong rather than merely imprecise.
36
+ */
37
+
38
+ import type { CommandSpan } from "./command-classes.js";
39
+
40
+ /* Verbs whose quoted arguments this scanner will call text. Every one of them either prints its argument or
41
+ * matches with it, and none has a documented way to execute it. Widening this list is safe in the sense the
42
+ * header sets out, but each entry should be able to answer "how would this run its argument?" with "it cannot". */
43
+ const QUOTING_VERBS: ReadonlySet<string> = new Set([
44
+ "echo",
45
+ "printf",
46
+ "rg",
47
+ "grep",
48
+ "egrep",
49
+ "fgrep",
50
+ "ack",
51
+ "ag",
52
+ "ripgrep",
53
+ ]);
54
+
55
+ /* A flag whose value is a message: git's -m, and the long spelling. A quoted string here is prose that reaches
56
+ * a commit, a tag or a PR, and `git commit -m "rm -rf the old build dir"` is one of the more ordinary ways to
57
+ * write a dangerous-looking command that is not one. Read on the WORD BEFORE a quoted argument, so it applies
58
+ * whatever the verb is. */
59
+ const MESSAGE_FLAGS: ReadonlySet<string> = new Set(["-m", "--message", "-am", "--body", "-b"]);
60
+
61
+ // Where an unquoted word ends: whitespace, a pipeline or list operator, a redirect, a subshell paren.
62
+ const WORD_END = /[\s;|&<>()]/;
63
+
64
+ // A word that is a variable assignment prefixing a command (`FOO=bar cmd …`), skipped when looking for the verb.
65
+ const ASSIGNMENT = /^[A-Za-z_][A-Za-z0-9_]*=/;
66
+
67
+ /* `<<EOF`, `<<-EOF`, `<<'EOF'`, `<<"EOF"`. The delimiter's quoting only decides whether the body expands, which
68
+ * changes nothing here: an expanded body is still a body being written somewhere rather than run. */
69
+ const HEREDOC_OPEN = /<<-?\s*(['"]?)([A-Za-z_][A-Za-z0-9_]*)\1/g;
70
+
71
+ /* The heredoc bodies in a command, as spans. Found first and in their own pass, because a body is line-oriented
72
+ * and can hold anything at all — an odd number of quotes in it would otherwise throw the word scanner out of
73
+ * step for the rest of the command. */
74
+ const heredocBodies = (command: string): CommandSpan[] => {
75
+ const bodies: CommandSpan[] = [];
76
+ for (const open of command.matchAll(HEREDOC_OPEN)) {
77
+ const indented = command.slice(open.index, open.index + 3).startsWith("<<-");
78
+ const newline = command.indexOf("\n", open.index + open[0].length);
79
+ if (newline === -1) {
80
+ // `cat <<EOF` with nothing after it: an unterminated heredoc, so there is no body to mark.
81
+ continue;
82
+ }
83
+ const start = newline + 1;
84
+ const terminator = new RegExp(`^${indented ? "[ \\t]*" : ""}${open[2] as string}[ \\t]*$`, "m");
85
+ const rest = command.slice(start);
86
+ const end = terminator.exec(rest)?.index;
87
+ // An unterminated body runs to the end of the command, which is what the shell would read too.
88
+ bodies.push({ start, end: end === undefined ? command.length : start + end });
89
+ }
90
+ return bodies;
91
+ };
92
+
93
+ // Is this offset inside one of the spans already found? Heredoc bodies are skipped wholesale by the word scan.
94
+ const within = (spans: readonly CommandSpan[], offset: number): CommandSpan | undefined =>
95
+ spans.find((span) => offset >= span.start && offset < span.end);
96
+
97
+ // A substitution inside double quotes, the one thing that makes a quoted argument a program again.
98
+ const EXPANDS = /\$\(|`/;
99
+
100
+ /* One quoted segment inside a word: the span covers the quotes as well as what is between them, so a region
101
+ * handed back from here contains the whole `"rm -rf /"` and a match on any part of it reads as contained. */
102
+ interface QuotedSegment {
103
+ readonly span: CommandSpan;
104
+ /* Does a shell expand anything in here? `"$(rm -rf /)"` inside an echo is a real delete whose output is
105
+ * printed, so a double-quoted segment carrying a substitution is NOT text and never becomes a region. Single
106
+ * quotes expand nothing, so they never set this. */
107
+ readonly expands: boolean;
108
+ }
109
+
110
+ interface ShellWord {
111
+ readonly start: number;
112
+ // The word with its quoting removed, which is what a verb name and a flag are compared against.
113
+ readonly text: string;
114
+ readonly quoted: readonly QuotedSegment[];
115
+ // Does this word begin a simple command? True for the first word after a separator or at the start.
116
+ readonly opensCommand: boolean;
117
+ }
118
+
119
+ // A verb by its bare name: `/bin/echo` and `./echo` are echo, and the path in front of it says nothing new.
120
+ const bareVerb = (text: string): string => text.slice(Math.max(text.lastIndexOf("/"), text.lastIndexOf("\\")) + 1);
121
+
122
+ // Every list and pipeline operator: past one of these the next word is a verb again.
123
+ const SEPARATORS = new Set(["\n", ";", "|", "&", "(", ")"]);
124
+ const BLANKS = new Set([" ", "\t", "\r"]);
125
+
126
+ /* One quoted run, from its opening quote to its closing one. An unbalanced quote takes the rest of the
127
+ * command, which is what a shell waiting for more input would do and keeps the caller advancing. */
128
+ const readQuoted = (command: string, open: number): QuotedSegment & { readonly text: string; readonly end: number } => {
129
+ const quote = command[open] as string;
130
+ const close = command.indexOf(quote, open + 1);
131
+ const end = close === -1 ? command.length : close + 1;
132
+ const text = command.slice(open + 1, close === -1 ? command.length : close);
133
+ return { span: { start: open, end }, expands: quote === '"' && EXPANDS.test(text), text, end };
134
+ };
135
+
136
+ /* One word, from `start` to whatever ends it, with the quoted runs inside it kept as spans. `end === start`
137
+ * cannot happen: the caller only enters here on a character that is neither blank nor a separator. */
138
+ const readWord = (command: string, start: number): { readonly word: Omit<ShellWord, "opensCommand">; readonly end: number } => {
139
+ const quoted: QuotedSegment[] = [];
140
+ let text = "";
141
+ let index = start;
142
+ while (index < command.length && !WORD_END.test(command[index] as string)) {
143
+ const char = command[index] as string;
144
+ if (char === "'" || char === '"') {
145
+ const segment = readQuoted(command, index);
146
+ quoted.push({ span: segment.span, expands: segment.expands });
147
+ text += segment.text;
148
+ index = segment.end;
149
+ continue;
150
+ }
151
+ if (char === "\\") {
152
+ text += command[index + 1] ?? "";
153
+ index += 2;
154
+ continue;
155
+ }
156
+ text += char;
157
+ index += 1;
158
+ }
159
+ return { word: { start, text, quoted }, end: index };
160
+ };
161
+
162
+ /* The command taken apart into words, with each word's quoted segments and whether it opens a simple command.
163
+ * Deliberately a scanner rather than a parser: it tracks quoting and the separators that start a new command,
164
+ * and it knows nothing about control flow, functions or expansion. Everything it gets wrong is bounded by the
165
+ * header's argument. */
166
+ const scanWords = (command: string, skip: readonly CommandSpan[]): ShellWord[] => {
167
+ const words: ShellWord[] = [];
168
+ let index = 0;
169
+ let opensCommand = true;
170
+ while (index < command.length) {
171
+ const skipped = within(skip, index);
172
+ if (skipped !== undefined) {
173
+ index = skipped.end;
174
+ continue;
175
+ }
176
+ const char = command[index] as string;
177
+ if (SEPARATORS.has(char)) {
178
+ opensCommand = true;
179
+ index += 1;
180
+ continue;
181
+ }
182
+ if (BLANKS.has(char)) {
183
+ index += 1;
184
+ continue;
185
+ }
186
+ if (char === "#") {
187
+ // A comment runs to the end of the line. Only reached at a word boundary, so `a#b` is one word.
188
+ const newline = command.indexOf("\n", index);
189
+ index = newline === -1 ? command.length : newline;
190
+ continue;
191
+ }
192
+ const { word, end } = readWord(command, index);
193
+ // A character that is neither a separator nor part of a word (a stray redirect): step over it so the
194
+ // loop always advances.
195
+ index = end === index ? index + 1 : end;
196
+ if (end !== word.start) {
197
+ words.push({ ...word, opensCommand });
198
+ opensCommand = false;
199
+ }
200
+ }
201
+ return words;
202
+ };
203
+
204
+ /* WHERE A COMMAND HOLDS TEXT RATHER THAN A PROGRAM, as spans over the command, unsorted and possibly
205
+ * overlapping — callers ask containment questions of them rather than rendering them.
206
+ *
207
+ * Exported for the classifier, which asks it once per command and hands the answer to every table. */
208
+ export const inertRegions = (command: string): CommandSpan[] => {
209
+ const bodies = heredocBodies(command);
210
+ const regions: CommandSpan[] = [...bodies];
211
+ const words = scanWords(command, bodies);
212
+ // Comments are consumed by the scanner rather than reported, so they are found again here: the scan is
213
+ // where the quoting state lives, and a `#` inside a quoted string is not a comment.
214
+ let quotingVerb = false;
215
+ let previous: ShellWord | undefined;
216
+ for (const word of words) {
217
+ if (word.opensCommand) {
218
+ quotingVerb = QUOTING_VERBS.has(bareVerb(word.text));
219
+ previous = undefined;
220
+ }
221
+ /* An env assignment in front of the verb (`LC_ALL=C grep …`) is not the verb. Re-read the next word as
222
+ * one instead of giving up on the command. */
223
+ if (word.opensCommand && ASSIGNMENT.test(word.text) && word.quoted.length === 0) {
224
+ quotingVerb = false;
225
+ previous = undefined;
226
+ continue;
227
+ }
228
+ const afterMessageFlag = previous !== undefined && MESSAGE_FLAGS.has(previous.text);
229
+ if (quotingVerb || afterMessageFlag) {
230
+ regions.push(...word.quoted.filter((segment) => !segment.expands).map((segment) => segment.span));
231
+ }
232
+ previous = word;
233
+ }
234
+ regions.push(...commentRegions(command, bodies));
235
+ return regions;
236
+ };
237
+
238
+ /* Comments, on the same scan discipline as the words: a `#` only opens one where a word could have started, so
239
+ * `sha#1` is not a comment, and a quoted `"# …"` is not one either.
240
+ *
241
+ * Its own walk rather than a by-product of scanWords, because the two want different things from a quote — the
242
+ * word scan needs what is INSIDE one, this needs only to be past it. Sharing readQuoted keeps them agreeing on
243
+ * where one ends, which is the only fact they both depend on. */
244
+ const commentRegions = (command: string, skip: readonly CommandSpan[]): CommandSpan[] => {
245
+ const comments: CommandSpan[] = [];
246
+ let index = 0;
247
+ while (index < command.length) {
248
+ const skipped = within(skip, index);
249
+ if (skipped !== undefined) {
250
+ index = skipped.end;
251
+ continue;
252
+ }
253
+ const char = command[index] as string;
254
+ if (char === "'" || char === '"') {
255
+ index = readQuoted(command, index).end;
256
+ continue;
257
+ }
258
+ if (char === "\\") {
259
+ index += 2;
260
+ continue;
261
+ }
262
+ // A word boundary in front is what makes it a comment rather than part of a word.
263
+ if (char === "#" && (index === 0 || WORD_END.test(command[index - 1] as string))) {
264
+ const newline = command.indexOf("\n", index);
265
+ const end = newline === -1 ? command.length : newline;
266
+ comments.push({ start: index, end });
267
+ index = end;
268
+ continue;
269
+ }
270
+ index += 1;
271
+ }
272
+ return comments;
273
+ };
274
+
275
+ /* Is this fragment somewhere a shell would RUN it? The question the hard rule asks of every span triage matched.
276
+ *
277
+ * JUDGED ON WHERE THE FRAGMENT STARTS, not on whether a region contains the whole of it, and the difference is
278
+ * not a relaxation — it is the only reading that answers the question asked. Every pattern in the catalog is
279
+ * written to BEGIN at the verb (`rm`, `docker volume rm`, `mkfs`, `git push`), so the span's first character is
280
+ * where the dangerous thing was found. Whether the match then ran on past a closing quote says something about
281
+ * the regex, not about the command: `rm`'s own parser reads to the end of the invocation and `>` is not a
282
+ * terminator, so `echo "rm -rf /" >> notes.md` produces a span covering `rm -rf /" >> notes.md`. Requiring
283
+ * containment made that command live — which is exactly the card this whole change exists to stop raising.
284
+ *
285
+ * The conservative direction is preserved where it matters: a fragment whose verb sits OUTSIDE every region is
286
+ * live no matter what it runs into afterwards, so `echo "tidying" && rm -rf /` and `rm -rf "$(cat list)" # ok`
287
+ * both stay live. */
288
+ export const isLive = (span: CommandSpan, regions: readonly CommandSpan[]): boolean =>
289
+ !regions.some((region) => span.start >= region.start && span.start < region.end);
package/src/versions.ts CHANGED
@@ -1,7 +1,7 @@
1
1
  /* COMPARING THE VERSIONS THIS SYSTEM STAMPS ON WHAT IT SHIPS, the daemon, the sandbox image, and the two agents
2
- * that run on a user's own computer. One release stamps all of them to the SAME version, so "is this one behind
2
+ * that run on a user's own device. One release stamps all of them to the SAME version, so "is this one behind
3
3
  * that one" is one question with one answer, and it lives here because both ends ask it: the daemon compares its
4
- * own build against the latest published release, and the browser compares a computer's agent against the same.
4
+ * own build against the latest published release, and the browser compares a device's agent against the same.
5
5
  *
6
6
  * Shared rather than copied because the two copies would not disagree until the day it mattered: 1.9.0 against
7
7
  * 1.10.0 is where a hand-rolled comparator goes wrong, and it goes wrong by reporting "up to date". */
@@ -35,7 +35,7 @@ export const webextLendUrl = (sandboxUrl: string): string => `${sandboxUrl.repla
35
35
 
36
36
  /* ---- the pairing code: the one string that travels from the sandbox's card into the extension ----
37
37
  *
38
- * A connected computer is paired by a shell one-liner, which can carry two values in two environment variables
38
+ * A connected device is paired by a shell one-liner, which can carry two values in two environment variables
39
39
  * because a terminal is a place where long strings are normal. A browser extension's popup is not: what a
40
40
  * person will actually do there is paste ONE thing, once, and anything that asks them to copy a URL into one
41
41
  * box and a token into another is a flow that fails on the second box.
@@ -43,7 +43,7 @@ export const webextLendUrl = (sandboxUrl: string): string => `${sandboxUrl.repla
43
43
  * So both halves ride in one code. It is not encryption and does not pretend to be — base64url of two fields,
44
44
  * so that the thing on the clipboard is opaque enough not to be edited by hand, short enough to paste, and
45
45
  * carries its own sandbox address, which is the field a person could not possibly be expected to type. The
46
- * secret in it is the pairing token, which is single-use and expires in ten minutes (webext-store.ts).
46
+ * secret in it is the pairing token, which is single-use and expires in ten minutes (the daemon's peer store).
47
47
  *
48
48
  * The prefix is a version marker, and it is here so that a code from an older sandbox meets a clear "this code
49
49
  * is from a different version" in the extension rather than a JSON parse error. */
@@ -2,13 +2,13 @@ import { z } from "zod";
2
2
 
3
3
  /* The handshake on /system/webext/connect, the ONE message on that socket that is not oRPC.
4
4
  *
5
- * Same two-phase shape as a connected computer's (host-protocol.ts) and for the same reason: the daemon has
5
+ * Same two-phase shape as a connected device's (host-protocol.ts) and for the same reason: the daemon has
6
6
  * nothing to call until it knows whose socket this is, so the proof cannot itself be an oRPC call. The browser
7
7
  * extension's first frame carries its enrollment token, the daemon resolves WHICH capability it belongs to, and
8
8
  * from that message on every byte is `webextContract` with the EXTENSION serving.
9
9
  *
10
10
  * WHY A SEPARATE PROTOCOL FROM host's, when the frame is the same two fields: because the thing on the other
11
- * end is not a computer. It has no shell, no filesystem and no screen; what it has is tabs, origins the person
11
+ * end is not a device. It has no shell, no filesystem and no screen; what it has is tabs, origins the person
12
12
  * granted one at a time, and a human watching every click. Sharing the host's schema would have meant a card of
13
13
  * switches that mean nothing (`roots`, `sandboxRemove`) and an agent told about a home directory it cannot
14
14
  * reach. The two connectors are siblings, not one connector with a flag. */
@@ -6,6 +6,9 @@ import {
6
6
  isLockedWorkspacePath,
7
7
  isReportedManifest,
8
8
  isReviewableLockedPath,
9
+ LOCKED_STATE_ENTRIES,
10
+ lockedWorkspaceEntry,
11
+ PLAN_DOCUMENTS_DIR,
9
12
  REPORTED_MANIFEST_PATHS,
10
13
  SEARCHABLE_STATE_PATHS,
11
14
  SHARED_STATE_PATHS,
@@ -244,6 +247,47 @@ describe(`isLockedWorkspacePath`, () => {
244
247
  expect(isLockedWorkspacePath(`.intentic\\secrets\\auth\\codex`)).toBe(true);
245
248
  expect(isLockedWorkspacePath(`./.intentic/config/capabilities.json`)).toBe(true);
246
249
  });
250
+
251
+ it(`lets the plan documents out of the session store around them`, () => {
252
+ // The plan a card asks the reader to approve, whose full text that card already renders. Refusing the
253
+ // file left the card's one link into the workspace landing on a padlock about the document it was
254
+ // asking about; the transcripts it sits beside stay locked.
255
+ expect(isLockedWorkspacePath(`${PLAN_DOCUMENTS_DIR}/wiggly-spring.md`)).toBe(false);
256
+ expect(isLockedWorkspacePath(PLAN_DOCUMENTS_DIR)).toBe(false);
257
+ expect(isLockedWorkspacePath(`.intentic/records/sessions/claude/projects/x.jsonl`)).toBe(true);
258
+ expect(isLockedWorkspacePath(`.intentic/records/sessions`)).toBe(true);
259
+ // …and it is the directory that is exempt, not the word: a sibling store named for it is not one.
260
+ expect(isLockedWorkspacePath(`.intentic/records/sessions/claude/plans-backup/x.md`)).toBe(true);
261
+ });
262
+ });
263
+
264
+ /* WHICH entry a locked path belongs to, which is what the refusal screen says a file holds and where to manage
265
+ * it. Split out from the boolean so the browser can key its sentences on the daemon's own list instead of a
266
+ * second copy of the rule — the copy it kept drifted through the state regrouping and stranded every locked
267
+ * file on the generic sentence. */
268
+ describe(`lockedWorkspaceEntry`, () => {
269
+ it(`names the entry a path matched, not the leaf it ends at`, () => {
270
+ // A locked FOLDER is one row in the explorer and never descended, so the name worth reporting is the
271
+ // folder's: "Cookies is kept private" is true of something the reader has never heard of.
272
+ expect(lockedWorkspaceEntry(`.intentic/local/browser/Default/Cookies`)).toBe(`local/browser`);
273
+ expect(lockedWorkspaceEntry(`.intentic/secrets/auth/codex/auth.json`)).toBe(`secrets/auth`);
274
+ expect(lockedWorkspaceEntry(`.intentic/config/capabilities.json`)).toBe(`config/capabilities.json`);
275
+ expect(lockedWorkspaceEntry(`.git/config`)).toBe(`.git`);
276
+ });
277
+
278
+ it(`answers undefined for everything the lock does not hold`, () => {
279
+ expect(lockedWorkspaceEntry(`src/app.ts`)).toBeUndefined();
280
+ expect(lockedWorkspaceEntry(`.intentic/config/settings.json`)).toBeUndefined();
281
+ expect(lockedWorkspaceEntry(`${PLAN_DOCUMENTS_DIR}/wiggly-spring.md`)).toBeUndefined();
282
+ });
283
+
284
+ it(`answers for every entry the lock declares`, () => {
285
+ // The set is the daemon's; this is what makes it addressable from the browser. An entry nobody can
286
+ // resolve back out is one the refusal screen could only describe generically.
287
+ for (const entry of LOCKED_STATE_ENTRIES) {
288
+ expect([entry, lockedWorkspaceEntry(`${STATE_DIR}/${entry}`)]).toEqual([entry, entry]);
289
+ }
290
+ });
247
291
  });
248
292
 
249
293
  /* The carve-out the diff routes ask for, and the reason it is derived: a locked entry the root repo TRACKS has
@@ -330,10 +374,13 @@ describe(`VERSIONED_STATE_PATHS`, () => {
330
374
  /* The connections themselves, and the entry that took the longest to earn its place: it was classed
331
375
  * `secret` on the strength of holding each capability's credential, which stopped being true when the
332
376
  * vault took the values out and left the shape behind. Connecting a deployment orchestrator, or
333
- * granting a connected computer shell and screen control, is the largest change made to what this
377
+ * granting a connected device shell and screen control, is the largest change made to what this
334
378
  * sandbox can DO, and it used to leave no diff. */
335
379
  `${STATE_DIR}/config/capabilities.json`,
336
380
  `${STATE_DIR}/config/capability-dismissals.json`,
381
+ // The context shelves: which repositories a conversation opened on one carries. A list of names, and
382
+ // the decision about what a session may see, which is what a review is for.
383
+ `${STATE_DIR}/config/context/`,
337
384
  /* The two entries the AGENT authors on its own initiative, and the reason `versioned` is not read as
338
385
  * config-only. Both are the sandbox acting outward: a draft publishes words under the owner's name,
339
386
  * a workspace extension is code that runs in the app and can serve HTTP with the workspace under