@intentic/sandbox-contract 1.245.0 → 1.247.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +17 -1
- package/dist/batch-runs.d.ts +2 -0
- package/dist/batch-runs.d.ts.map +1 -1
- package/dist/batch-runs.js +1 -0
- package/dist/batch-runs.js.map +1 -1
- package/dist/command-classes.d.ts +6 -3
- package/dist/command-classes.d.ts.map +1 -1
- package/dist/command-classes.js +43 -18
- package/dist/command-classes.js.map +1 -1
- package/dist/contracts/{cursor.contract.d.ts → accounts.contract.d.ts} +102 -3
- package/dist/contracts/accounts.contract.d.ts.map +1 -0
- package/dist/contracts/accounts.contract.js +61 -0
- package/dist/contracts/accounts.contract.js.map +1 -0
- package/dist/contracts/agent.contract.d.ts +19 -0
- package/dist/contracts/agent.contract.d.ts.map +1 -1
- package/dist/contracts/agents.contract.d.ts +121 -0
- package/dist/contracts/agents.contract.d.ts.map +1 -1
- package/dist/contracts/agents.contract.js +4 -4
- package/dist/contracts/agents.contract.js.map +1 -1
- package/dist/contracts/ci.contract.d.ts +2 -0
- package/dist/contracts/ci.contract.d.ts.map +1 -1
- package/dist/contracts/host.contract.d.ts +35 -0
- package/dist/contracts/host.contract.d.ts.map +1 -1
- package/dist/contracts/host.contract.js +3 -2
- package/dist/contracts/host.contract.js.map +1 -1
- package/dist/contracts/personas.contract.d.ts +4 -2
- package/dist/contracts/personas.contract.d.ts.map +1 -1
- package/dist/contracts/runner.contract.d.ts +2 -2
- package/dist/contracts/settings.contract.d.ts +58 -20
- package/dist/contracts/settings.contract.d.ts.map +1 -1
- package/dist/contracts/system.contract.d.ts +80 -30
- package/dist/contracts/system.contract.d.ts.map +1 -1
- package/dist/contracts/system.contract.js +26 -17
- package/dist/contracts/system.contract.js.map +1 -1
- package/dist/contracts/usage.contract.d.ts +22 -0
- package/dist/contracts/usage.contract.d.ts.map +1 -1
- package/dist/contracts/usage.contract.js +19 -0
- package/dist/contracts/usage.contract.js.map +1 -1
- package/dist/definition.d.ts +20 -28
- package/dist/definition.d.ts.map +1 -1
- package/dist/documents.d.ts +0 -1
- package/dist/documents.d.ts.map +1 -1
- package/dist/documents.js +1 -2
- package/dist/documents.js.map +1 -1
- package/dist/embed.d.ts +23 -0
- package/dist/embed.d.ts.map +1 -0
- package/dist/embed.js +84 -0
- package/dist/embed.js.map +1 -0
- package/dist/events.d.ts +21 -0
- package/dist/events.d.ts.map +1 -1
- package/dist/events.js +5 -2
- package/dist/events.js.map +1 -1
- package/dist/fast-tier.js +1 -1
- package/dist/fast-tier.js.map +1 -1
- package/dist/history-state.d.ts.map +1 -1
- package/dist/history-state.js +2 -0
- package/dist/history-state.js.map +1 -1
- package/dist/index.d.ts +453 -306
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +7 -15
- package/dist/index.js.map +1 -1
- package/dist/model-pins.d.ts +17 -0
- package/dist/model-pins.d.ts.map +1 -0
- package/dist/{quick-model.js → model-pins.js} +18 -11
- package/dist/model-pins.js.map +1 -0
- package/dist/model-roles.d.ts +144 -0
- package/dist/model-roles.d.ts.map +1 -0
- package/dist/model-roles.js +129 -0
- package/dist/model-roles.js.map +1 -0
- package/dist/peer-dial.d.ts +33 -0
- package/dist/peer-dial.d.ts.map +1 -0
- package/dist/peer-dial.js +79 -0
- package/dist/peer-dial.js.map +1 -0
- package/dist/peer-mcp-server.d.ts +36 -0
- package/dist/peer-mcp-server.d.ts.map +1 -0
- package/dist/peer-mcp-server.js +71 -0
- package/dist/peer-mcp-server.js.map +1 -0
- package/dist/provider-specs.d.ts +38 -20
- package/dist/provider-specs.d.ts.map +1 -1
- package/dist/provider-specs.js +39 -13
- package/dist/provider-specs.js.map +1 -1
- package/dist/runtime-state.d.ts +1 -1
- package/dist/runtime-state.js +1 -1
- package/dist/runtime-state.js.map +1 -1
- package/dist/safety-policy.d.ts +12 -3
- package/dist/safety-policy.d.ts.map +1 -1
- package/dist/safety-policy.js +30 -5
- package/dist/safety-policy.js.map +1 -1
- package/dist/schemas/agent.d.ts +27 -8
- package/dist/schemas/agent.d.ts.map +1 -1
- package/dist/schemas/agent.js +10 -4
- package/dist/schemas/agent.js.map +1 -1
- package/dist/schemas/agents.d.ts +42 -0
- package/dist/schemas/agents.d.ts.map +1 -1
- package/dist/schemas/agents.js +25 -4
- package/dist/schemas/agents.js.map +1 -1
- package/dist/schemas/automations.d.ts +11 -2
- package/dist/schemas/automations.d.ts.map +1 -1
- package/dist/schemas/automations.js +1 -1
- package/dist/schemas/automations.js.map +1 -1
- package/dist/schemas/ci.d.ts +6 -0
- package/dist/schemas/ci.d.ts.map +1 -1
- package/dist/schemas/ci.js +3 -2
- package/dist/schemas/ci.js.map +1 -1
- package/dist/schemas/context.d.ts +30 -0
- package/dist/schemas/context.d.ts.map +1 -0
- package/dist/schemas/context.js +34 -0
- package/dist/schemas/context.js.map +1 -0
- package/dist/schemas/{computers.d.ts → devices.d.ts} +154 -60
- package/dist/schemas/devices.d.ts.map +1 -0
- package/dist/schemas/devices.js +157 -0
- package/dist/schemas/devices.js.map +1 -0
- package/dist/schemas/hosts.d.ts +12 -0
- package/dist/schemas/hosts.d.ts.map +1 -1
- package/dist/schemas/hosts.js +1 -0
- package/dist/schemas/hosts.js.map +1 -1
- package/dist/schemas/issues.d.ts +0 -5
- package/dist/schemas/issues.d.ts.map +1 -1
- package/dist/schemas/issues.js +0 -1
- package/dist/schemas/issues.js.map +1 -1
- package/dist/schemas/personas.d.ts +5 -3
- package/dist/schemas/personas.d.ts.map +1 -1
- package/dist/schemas/personas.js +3 -2
- package/dist/schemas/personas.js.map +1 -1
- package/dist/schemas/plan-limits.d.ts +20 -0
- package/dist/schemas/plan-limits.d.ts.map +1 -1
- package/dist/schemas/plan-limits.js +21 -0
- package/dist/schemas/plan-limits.js.map +1 -1
- package/dist/schemas/provider-oauth.d.ts +48 -16
- package/dist/schemas/provider-oauth.d.ts.map +1 -1
- package/dist/schemas/provider-oauth.js +22 -20
- package/dist/schemas/provider-oauth.js.map +1 -1
- package/dist/schemas/settings.d.ts +42 -16
- package/dist/schemas/settings.d.ts.map +1 -1
- package/dist/schemas/settings.js +22 -28
- package/dist/schemas/settings.js.map +1 -1
- package/dist/schemas/terminal.js +9 -9
- package/dist/schemas/terminal.js.map +1 -1
- package/dist/schemas/usage.d.ts +5 -2
- package/dist/schemas/usage.d.ts.map +1 -1
- package/dist/schemas/usage.js +5 -2
- package/dist/schemas/usage.js.map +1 -1
- package/dist/shell-regions.d.ts +4 -0
- package/dist/shell-regions.d.ts.map +1 -0
- package/dist/shell-regions.js +156 -0
- package/dist/shell-regions.js.map +1 -0
- package/dist/workspace-state.d.ts +8 -0
- package/dist/workspace-state.d.ts.map +1 -1
- package/dist/workspace-state.js +13 -5
- package/dist/workspace-state.js.map +1 -1
- package/package.json +37 -4
- package/src/agent-catalog.ts +2 -2
- package/src/arrival.ts +3 -3
- package/src/batch-runs.test.ts +10 -5
- package/src/batch-runs.ts +10 -3
- package/src/command-classes.test.ts +195 -71
- package/src/command-classes.ts +148 -46
- package/src/contracts/accounts.contract.ts +94 -0
- package/src/contracts/agents.contract.ts +4 -3
- package/src/contracts/exit.contract.ts +2 -2
- package/src/contracts/host.contract.ts +17 -5
- package/src/contracts/settings.contract.ts +1 -1
- package/src/contracts/system.contract.ts +43 -24
- package/src/contracts/usage.contract.ts +31 -0
- package/src/contracts/vpn.contract.ts +2 -2
- package/src/documents.test.ts +2 -1
- package/src/documents.ts +7 -11
- package/src/embed.test.ts +68 -0
- package/src/embed.ts +164 -0
- package/src/events.ts +31 -4
- package/src/fast-tier.test.ts +1 -1
- package/src/fast-tier.ts +5 -5
- package/src/history-state.ts +12 -3
- package/src/host-protocol.ts +2 -2
- package/src/index.ts +8 -16
- package/src/model-order.ts +1 -1
- package/src/{quick-model.test.ts → model-pins.test.ts} +73 -29
- package/src/model-pins.ts +183 -0
- package/src/model-roles.ts +224 -0
- package/src/peer-dial.test.ts +203 -0
- package/src/peer-dial.ts +163 -0
- package/src/peer-mcp-server.test.ts +104 -0
- package/src/peer-mcp-server.ts +144 -0
- package/src/plan-pools.ts +1 -1
- package/src/prompt-complexity.test.ts +1 -1
- package/src/prompt-complexity.ts +2 -2
- package/src/provider-specs.test.ts +45 -18
- package/src/provider-specs.ts +147 -67
- package/src/routes.test.ts +6 -3
- package/src/runner-protocol.ts +1 -1
- package/src/runtime-state.ts +2 -2
- package/src/safety-policy.test.ts +88 -0
- package/src/safety-policy.ts +84 -14
- package/src/schemas/agent.ts +83 -30
- package/src/schemas/agents.ts +67 -6
- package/src/schemas/automations.ts +6 -4
- package/src/schemas/capabilities.ts +4 -4
- package/src/schemas/ci.ts +23 -6
- package/src/schemas/context.ts +87 -0
- package/src/schemas/{computers.ts → devices.ts} +190 -107
- package/src/schemas/hosts.ts +5 -1
- package/src/schemas/issues.ts +0 -4
- package/src/schemas/personas.ts +8 -3
- package/src/schemas/plan-limits.ts +50 -0
- package/src/schemas/provider-oauth.ts +49 -52
- package/src/schemas/settings.ts +105 -140
- package/src/schemas/terminal.ts +12 -12
- package/src/schemas/usage.ts +62 -27
- package/src/schemas/version-seam.test.ts +0 -1
- package/src/shell-regions.ts +289 -0
- package/src/versions.ts +2 -2
- package/src/webext-links.ts +2 -2
- package/src/webext-protocol.ts +2 -2
- package/src/workspace-state.test.ts +48 -1
- package/src/workspace-state.ts +48 -11
- package/dist/agent-run-model.d.ts +0 -4
- package/dist/agent-run-model.d.ts.map +0 -1
- package/dist/agent-run-model.js +0 -13
- package/dist/agent-run-model.js.map +0 -1
- package/dist/contracts/claude.contract.d.ts +0 -91
- package/dist/contracts/claude.contract.d.ts.map +0 -1
- package/dist/contracts/claude.contract.js +0 -50
- package/dist/contracts/claude.contract.js.map +0 -1
- package/dist/contracts/cursor.contract.d.ts.map +0 -1
- package/dist/contracts/cursor.contract.js +0 -50
- package/dist/contracts/cursor.contract.js.map +0 -1
- package/dist/contracts/grok.contract.d.ts +0 -36
- package/dist/contracts/grok.contract.d.ts.map +0 -1
- package/dist/contracts/grok.contract.js +0 -31
- package/dist/contracts/grok.contract.js.map +0 -1
- package/dist/contracts/keys.contract.d.ts +0 -81
- package/dist/contracts/keys.contract.d.ts.map +0 -1
- package/dist/contracts/keys.contract.js +0 -51
- package/dist/contracts/keys.contract.js.map +0 -1
- package/dist/quick-model.d.ts +0 -15
- package/dist/quick-model.d.ts.map +0 -1
- package/dist/quick-model.js.map +0 -1
- package/dist/schemas/computers.d.ts.map +0 -1
- package/dist/schemas/computers.js +0 -135
- package/dist/schemas/computers.js.map +0 -1
- package/src/agent-run-model.test.ts +0 -76
- package/src/agent-run-model.ts +0 -65
- package/src/contracts/claude.contract.ts +0 -71
- package/src/contracts/cursor.contract.ts +0 -74
- package/src/contracts/grok.contract.ts +0 -41
- package/src/contracts/keys.contract.ts +0 -79
- package/src/quick-model.ts +0 -155
package/src/schemas/terminal.ts
CHANGED
|
@@ -194,14 +194,14 @@ export type SubagentKind = z.infer<typeof SubagentKindSchema>;
|
|
|
194
194
|
// an operator acts on differently from "the child is working".
|
|
195
195
|
export const SubagentStatusSchema = z.enum(["pending", "running", "blocked", "completed", "failed", "killed", "paused"]);
|
|
196
196
|
export type SubagentStatus = z.infer<typeof SubagentStatusSchema>;
|
|
197
|
-
/* WHETHER ANYTHING CHECKED WHAT THE
|
|
198
|
-
* assume. Computed from the
|
|
197
|
+
/* WHETHER ANYTHING CHECKED WHAT THE SUBAGENT DID, carried beside its report rather than left for the reader to
|
|
198
|
+
* assume. Computed from the subagent's own tool calls, the files it edited against the checks that ran after
|
|
199
199
|
* them (the daemon's child-verification.ts), so it holds on every provider rather than only where the Claude
|
|
200
200
|
* hooks reach.
|
|
201
201
|
*
|
|
202
202
|
* The four states are deliberately not two. `verified` and `failing` each name the command that spoke, so a
|
|
203
203
|
* targeted test is never read as the suite; `unproven` is the one that matters most, work changed and nothing
|
|
204
|
-
* ran; and `no-code` says the
|
|
204
|
+
* ran; and `no-code` says the subagent edited nothing, which is the honest answer for a research subagent and
|
|
205
205
|
* must not be rendered as approval. Absent ⇒ the daemon saw no tool calls from it at all. */
|
|
206
206
|
export const SubagentVerificationSchema = z.object({
|
|
207
207
|
state: z
|
|
@@ -222,10 +222,10 @@ export const SubagentSessionSchema = z.object({
|
|
|
222
222
|
id: z
|
|
223
223
|
.string()
|
|
224
224
|
.describe(
|
|
225
|
-
"The id of the tool call that started it (an SDK child) or the child's own conversation id (a spawned one); either way both sides already hold it, so a card links to its
|
|
225
|
+
"The id of the tool call that started it (an SDK child) or the child's own conversation id (a spawned one); either way both sides already hold it, so a card links to its subagent with the id it has and the subagent points back the same way.",
|
|
226
226
|
),
|
|
227
227
|
kind: SubagentKindSchema.describe(
|
|
228
|
-
"What sort of
|
|
228
|
+
"What sort of subagent: one the runtime's own Task tool spawned in-process, or a full child agent the daemon started for the turn. It changes only how you watch it.",
|
|
229
229
|
),
|
|
230
230
|
// The conversation whose turn spawned this, what the area groups its rows by, and the way back to the chat
|
|
231
231
|
// the card lives in.
|
|
@@ -233,19 +233,19 @@ export const SubagentSessionSchema = z.object({
|
|
|
233
233
|
// What it is and what it was asked to do: the subagent type (`Explore`, `general-purpose`) or a spawned
|
|
234
234
|
// child's provider label, and the caller's one-line description. The area's row and the card's title read
|
|
235
235
|
// as `Explore · Locate claimIndexer definition`.
|
|
236
|
-
agentType: z.string().optional().describe("What kind of
|
|
236
|
+
agentType: z.string().optional().describe("What kind of subagent it is."),
|
|
237
237
|
description: z.string().optional().describe("What it was asked to do, in one line."),
|
|
238
238
|
model: z.string().optional().describe("Which model it runs on."),
|
|
239
239
|
// Which provider serves a `spawned` child (its AgentProvider id), so the row can wear the right logo. An
|
|
240
240
|
// SDK subagent implies its own: it runs on its parent's provider.
|
|
241
|
-
provider: z.string().optional().describe("Which provider serves it, for a
|
|
241
|
+
provider: z.string().optional().describe("Which provider serves it, for a child agent spawned across providers."),
|
|
242
242
|
// How deep in the spawn tree (1 = spawned by the turn itself). From the SDK's meta.json; a subagent may
|
|
243
243
|
// itself delegate, and a flat list that cannot say so reads as though the turn started all of them.
|
|
244
244
|
spawnDepth: z
|
|
245
245
|
.number()
|
|
246
246
|
.optional()
|
|
247
247
|
.describe(
|
|
248
|
-
"How deep in the chain it sits, where one means the turn itself started it. A
|
|
248
|
+
"How deep in the chain it sits, where one means the turn itself started it. A subagent can start subagents, and a flat list that could not say so would read as though the turn started all of them.",
|
|
249
249
|
),
|
|
250
250
|
// Backgrounded: the parent went on working instead of waiting for it. This is the whole reason the list
|
|
251
251
|
// exists, a backgrounded child used to be invisible until its result landed, sometimes minutes later.
|
|
@@ -253,7 +253,7 @@ export const SubagentSessionSchema = z.object({
|
|
|
253
253
|
.boolean()
|
|
254
254
|
.optional()
|
|
255
255
|
.describe(
|
|
256
|
-
"The parent carried on working instead of waiting for it. This is the whole reason the list exists: such a
|
|
256
|
+
"The parent carried on working instead of waiting for it. This is the whole reason the list exists: such a subagent used to be invisible until its result landed, sometimes minutes later.",
|
|
257
257
|
),
|
|
258
258
|
status: SubagentStatusSchema.describe(
|
|
259
259
|
"How it is going. Blocked means it needs an answer, which a parent and an operator act on differently from it simply working.",
|
|
@@ -266,7 +266,7 @@ export const SubagentSessionSchema = z.object({
|
|
|
266
266
|
tokens: z
|
|
267
267
|
.number()
|
|
268
268
|
.optional()
|
|
269
|
-
.describe("What it has spent. Its own, so a parent's cost and the sum of its
|
|
269
|
+
.describe("What it has spent. Its own, so a parent's cost and the sum of its subagents' are two different true numbers."),
|
|
270
270
|
toolUses: z.number().optional().describe("How many tools it has used."),
|
|
271
271
|
lastTool: z.string().optional().describe("The last one it reached for."),
|
|
272
272
|
// Its report, the last assistant message (SubagentStop) or the task summary. The answer to "what did it
|
|
@@ -274,7 +274,7 @@ export const SubagentSessionSchema = z.object({
|
|
|
274
274
|
summary: z
|
|
275
275
|
.string()
|
|
276
276
|
.optional()
|
|
277
|
-
.describe("Its report: what it concluded, without opening its record. The question a finished
|
|
277
|
+
.describe("Its report: what it concluded, without opening its record. The question a finished subagent gets read for."),
|
|
278
278
|
error: z.string().optional().describe("Why it failed, when it did."),
|
|
279
279
|
// Whether anything checked the work behind that report (SubagentVerificationSchema). Filled once it ends:
|
|
280
280
|
// a standing read while it is still working would be a verdict on a job half done.
|
|
@@ -282,7 +282,7 @@ export const SubagentSessionSchema = z.object({
|
|
|
282
282
|
});
|
|
283
283
|
export type SubagentSession = z.infer<typeof SubagentSessionSchema>;
|
|
284
284
|
export const SubagentsListSchema = z.object({
|
|
285
|
-
sessions: z.array(SubagentSessionSchema).describe("Every
|
|
285
|
+
sessions: z.array(SubagentSessionSchema).describe("Every subagent and child agent this sandbox's conversations have started."),
|
|
286
286
|
});
|
|
287
287
|
export type SubagentsList = z.infer<typeof SubagentsListSchema>;
|
|
288
288
|
export const SubagentIdParamSchema = z.object({ id: z.string() });
|
package/src/schemas/usage.ts
CHANGED
|
@@ -83,14 +83,6 @@ export const UsageTurnSchema = z.object({
|
|
|
83
83
|
cacheCreationTokens: z.number().describe("Tokens written to cache, which cost more up front and less afterwards."),
|
|
84
84
|
costUsd: z.number().describe("What it cost, in dollars."),
|
|
85
85
|
durationMs: z.number().describe("How long it took, in milliseconds."),
|
|
86
|
-
/* Which arm of the terse experiment this turn ran on (settings.terseHoldout), the only record of it, and
|
|
87
|
-
* the reason the savings report can say what the steer is worth instead of guessing.
|
|
88
|
-
*
|
|
89
|
-
* ABSENT means "not part of the experiment", not "off": a turn under a custom system prompt drops the
|
|
90
|
-
* steer along with everything else the daemon appends, and a turn run with the experiment switched off has
|
|
91
|
-
* no control to be compared against. Pooling those into the off-arm would compare steered turns against a
|
|
92
|
-
* population selected by something other than the coin flip, which is not a control at all. */
|
|
93
|
-
terse: z.boolean().optional(),
|
|
94
86
|
/* Which arm of the iq SEARCH-TEACHING experiment this conversation runs on
|
|
95
87
|
* (settings.iqSearchHoldout). Stable for every turn in one conversation: the treatment is instruction
|
|
96
88
|
* loaded into a provider session, so flipping it per turn would call a remembered treatment a control.
|
|
@@ -99,25 +91,9 @@ export const UsageTurnSchema = z.object({
|
|
|
99
91
|
// Hash of the plugin nudge + skill body used for this arm. Control turns carry it too, so a report can keep
|
|
100
92
|
// both sides of one treatment revision together and exclude older wording after an upgrade.
|
|
101
93
|
iqSearchCohort: z.string().optional(),
|
|
102
|
-
/* Characters of the model's own PROSE this turn, the `delta` frames only, so no tool-call arguments and no
|
|
103
|
-
* thinking. What the terse steer is judged on, and the reason it can be judged at all.
|
|
104
|
-
*
|
|
105
|
-
* `outputTokens` cannot serve: measured over a day of real turns it is 91.6% tool-call arguments (an Edit's
|
|
106
|
-
* old_string and new_string, a Write's whole file body) and 7.8% prose. The steer moves prose. So a fifth
|
|
107
|
-
* off the model's narration moves the total by 1.6%, against a margin of ±35 points, which is to say the
|
|
108
|
-
* experiment was structurally unable to see its own treatment, and the number it printed instead was
|
|
109
|
-
* whichever arm happened to draw the bigger tasks.
|
|
110
|
-
*
|
|
111
|
-
* CHARACTERS, not tokens, because the provider bills a total and never breaks it down, a token figure here
|
|
112
|
-
* would be chars÷4 wearing a unit it had not earned. For a comparison of two arms the constant cancels
|
|
113
|
-
* anyway, and the honest unit is the one actually counted.
|
|
114
|
-
*
|
|
115
|
-
* Absent ⇒ the turn predates this being measured; `armOf` drops it from the population rather than reading
|
|
116
|
-
* it as a silent turn. */
|
|
117
|
-
proseChars: z.number().optional(),
|
|
118
94
|
/* SEARCHES THIS TURN RAN, every tool call that went looking for code, the dedicated search tools and the
|
|
119
95
|
* CLI searches alike (isSearchCall owns the rule; `iq q` is Bash and would otherwise not be counted at all).
|
|
120
|
-
* What the search teaching is judged on
|
|
96
|
+
* What the search teaching is judged on.
|
|
121
97
|
*
|
|
122
98
|
* COST PER TURN CANNOT SERVE: cost is a whole turn's worth of work, a search mechanism touches one part of
|
|
123
99
|
* it, and the part lives inside the noise of the rest, exactly the shape that made output tokens unable to
|
|
@@ -128,8 +104,8 @@ export const UsageTurnSchema = z.object({
|
|
|
128
104
|
* rather than being filtered out, they dilute both arms equally, while selecting on "did it search" would
|
|
129
105
|
* select on the treatment itself.
|
|
130
106
|
*
|
|
131
|
-
* Absent ⇒ the turn predates this being measured;
|
|
132
|
-
* searched nothing. */
|
|
107
|
+
* Absent ⇒ the turn predates this being measured; the arm's mean drops it (turn-experiments.ts, the filter
|
|
108
|
+
* inside `conversationExperimentOf`) rather than reading it as a turn that searched nothing. */
|
|
133
109
|
searchCalls: z.number().optional(),
|
|
134
110
|
/* …and how many of them came BEFORE the turn first opened or changed a file, the orientation burst. A turn
|
|
135
111
|
* that already knows where to look starts working; one that doesn't goes hunting first.
|
|
@@ -143,6 +119,65 @@ export const UsageTurnSchema = z.object({
|
|
|
143
119
|
*
|
|
144
120
|
* Absent ⇒ as for `searchCalls`. */
|
|
145
121
|
openingSearches: z.number().optional(),
|
|
122
|
+
/* THE LISTINGS THIS TURN RAN TO WORK OUT WHERE IT WAS, `ls`, `ls /work`, `tree intentic`, the LS tool
|
|
123
|
+
* aimed at the same places (isRootListing owns the rule). What the project map is judged on, and the
|
|
124
|
+
* reason it could not be judged before.
|
|
125
|
+
*
|
|
126
|
+
* `searchCalls` counts a directory listing and a ripgrep as one event, deliberately, so that a taxonomy
|
|
127
|
+
* cannot report whichever spelling of a search the model happened to reach for. That is right for the
|
|
128
|
+
* search teaching and blind to the map: measured over 468 mapped sessions of this workspace against 497
|
|
129
|
+
* unmapped ones, searches before the first file moved +7.6% with a margin of ±17.6pp, while the share of
|
|
130
|
+
* sessions opening with a directory listing fell from 46.3% to 32.1%. The map does not shorten the
|
|
131
|
+
* orientation burst, it changes what the burst is made of, and only this counts the difference.
|
|
132
|
+
*
|
|
133
|
+
* UP TO THE FIRST FILE, exactly like `openingSearches`, which took the corpus to settle. Counted over the
|
|
134
|
+
* whole turn instead, the same sessions give 66.4% against 73.3%, a gap of 6.9pp where the orientation
|
|
135
|
+
* window gives 14.2pp. The dilution is not noise: a turn already at work lists the directory it has
|
|
136
|
+
* narrowed to, and no map could have answered that. The shape of the listing cannot tell the two apart,
|
|
137
|
+
* because it is the same shape; only when it happened can.
|
|
138
|
+
*
|
|
139
|
+
* Absent ⇒ the turn predates this being measured, never a turn that listed nothing. */
|
|
140
|
+
openingListings: z.number().optional(),
|
|
141
|
+
/* TOOL CALLS BEFORE THE TURN FIRST TOUCHED A FILE IT WENT ON TO EDIT: how far it walked before reaching
|
|
142
|
+
* the thing it turned out to be looking for. The corpus study this map was designed from measured the
|
|
143
|
+
* same quantity by hand (the workspace's docs/agent-exploration-patterns.md §4: median 4, mean 8.2) and called it the one
|
|
144
|
+
* honest reading of whether orientation got better.
|
|
145
|
+
*
|
|
146
|
+
* IT CAN ONLY BE KNOWN AT THE END, which is why it is recorded here and computed nowhere else: whether a
|
|
147
|
+
* file was the target is a fact about the turn's edits, and the turn has to finish before that is
|
|
148
|
+
* decided. The route keeps the first call index per path and intersects it with the edit ledger.
|
|
149
|
+
*
|
|
150
|
+
* Absent on a turn that edited nothing, which is most short turns, and NOT zero: a turn with no target
|
|
151
|
+
* never reached one, and averaging that in as "reached it immediately" would report the turns that did
|
|
152
|
+
* no work as the best targeted. That absence costs the reading three quarters of the population, so it
|
|
153
|
+
* is a metric to accumulate for months rather than a gate to wait on. */
|
|
154
|
+
callsBeforeTarget: z.number().optional(),
|
|
155
|
+
/* WHICH ARM OF THE PROJECT MAP EXPERIMENT this conversation ran (settings.workspaceMapHoldout), and how
|
|
156
|
+
* many characters the note actually cost when it was sent.
|
|
157
|
+
*
|
|
158
|
+
* CONVERSATION-STABLE, and for a plainer reason than the search teaching's: the map is sent once, on the
|
|
159
|
+
* opening message, so every later turn of a mapped conversation is a turn whose transcript holds a map.
|
|
160
|
+
* A per-turn flip would label eleven treated turns as controls.
|
|
161
|
+
*
|
|
162
|
+
* `mapChars` rather than a token estimate, because characters are what the renderer's budget is in
|
|
163
|
+
* (workspace-map.ts caps the note at 2,800) and a tokenizer's answer would vary by model. Present only on
|
|
164
|
+
* the turn that actually sent one, so the ledger says what the feature costs instead of assuming it: the
|
|
165
|
+
* median note over this workspace's own corpus is 795 characters against a ceiling of 2,800.
|
|
166
|
+
*
|
|
167
|
+
* Absent ⇒ no measurement (the holdout is zero, or the row predates it); true/false ⇒ mapped/unmapped. */
|
|
168
|
+
mapArm: z.boolean().optional(),
|
|
169
|
+
mapChars: z.number().optional(),
|
|
170
|
+
/* WHICH TURN OF ITS CONVERSATION THIS WAS, counting from zero, so an opening turn can be recognised from
|
|
171
|
+
* one row instead of inferred from the rows around it.
|
|
172
|
+
*
|
|
173
|
+
* The inference it replaces is wrong at exactly one place and it is the place that matters: a reader
|
|
174
|
+
* windowed to the last seven days would take each conversation's earliest row IN THE WINDOW as its
|
|
175
|
+
* opening turn, so every conversation that started the week before would contribute a mid-conversation
|
|
176
|
+
* turn to a reading about first turns. The project map is sent on turn zero and nowhere else, so that
|
|
177
|
+
* misreading is not an edge case for it, it is the measurement.
|
|
178
|
+
*
|
|
179
|
+
* Absent ⇒ the row predates this, or the turn belonged to no conversation at all. */
|
|
180
|
+
turnIndex: z.number().optional(),
|
|
146
181
|
/* DID THIS TURN FINISH, OR DID IT STOP TALKING. The fields that tell the two apart, and the reason
|
|
147
182
|
* `outcome` alone could never.
|
|
148
183
|
*
|
|
@@ -0,0 +1,289 @@
|
|
|
1
|
+
/* WHICH PARTS OF A COMMAND ARE TEXT RATHER THAN A PROGRAM, so the one tier that cannot be argued with stops
|
|
2
|
+
* firing on a mention of a dangerous verb.
|
|
3
|
+
*
|
|
4
|
+
* WHAT THIS IS FOR, precisely, because it bounds how good it has to be. The triage catalog next door
|
|
5
|
+
* (command-classes.ts) is deliberately over-inclusive: a match only means a judge should look, and a false
|
|
6
|
+
* positive there costs one model call. That economy holds for every class except the hard-ruled one, where a
|
|
7
|
+
* match is an interruption no policy and no verdict can waive (safety-policy.ts hardRuleClasses). So
|
|
8
|
+
* `echo "rm -rf /" >> notes.md` and `rg 'rm -rf /'` raised un-waivable cards over a string being written to a
|
|
9
|
+
* file and a search of the tree — the exact failure the judge redesign was built to end, surviving in the one
|
|
10
|
+
* tier the judge cannot reach.
|
|
11
|
+
*
|
|
12
|
+
* This says where a fragment sits. A fragment inside a region below is still REPORTED (the class holds, the
|
|
13
|
+
* judge still reads it, the card still marks it); it just does not trip the hard rule.
|
|
14
|
+
*
|
|
15
|
+
* WHICH IS WHY A REGEX-LEVEL SCANNER IS ENOUGH, and this is the design argument rather than an excuse. Both
|
|
16
|
+
* ways of being wrong are cheap:
|
|
17
|
+
*
|
|
18
|
+
* · MISS a region (call text a program) ⇒ one judge call. Exactly today's behaviour, which is the floor.
|
|
19
|
+
* · INVENT a region (call a program text) ⇒ the class is still reported and the judge still rules on it.
|
|
20
|
+
* Only the un-waivable tier is skipped, and the judge is the tier that reads the owner's policy.
|
|
21
|
+
*
|
|
22
|
+
* Precision has to be good enough to skip tier 1½, never good enough to skip tier 1. Nothing here is a
|
|
23
|
+
* boundary, for the same reason nothing in command-classes.ts is: `sh -c "$CMD"` and a path assembled from a
|
|
24
|
+
* variable walk past all of it. The boundaries are structural and elsewhere.
|
|
25
|
+
*
|
|
26
|
+
* THREE KINDS OF REGION, and each is a place a shell will not run what it holds:
|
|
27
|
+
*
|
|
28
|
+
* 1 A COMMENT, `#` to end of line.
|
|
29
|
+
* 2 A HEREDOC BODY, the usual way an agent writes a script it is not running yet.
|
|
30
|
+
* 3 A QUOTED ARGUMENT OF A VERB THAT CANNOT EXECUTE ONE — echo, printf, and the searchers. Plus a quoted
|
|
31
|
+
* commit message after -m, whatever the verb, because message text never runs.
|
|
32
|
+
*
|
|
33
|
+
* `sed`, `awk` and `perl` are deliberately NOT in that verb list, though they are the obvious next entries:
|
|
34
|
+
* each can run a shell out of its own quoted program (`awk 'BEGIN{system("…")}'`, `perl -e`, GNU sed's `s///e`),
|
|
35
|
+
* so their quoted argument is a program and calling it text would be wrong rather than merely imprecise.
|
|
36
|
+
*/
|
|
37
|
+
|
|
38
|
+
import type { CommandSpan } from "./command-classes.js";
|
|
39
|
+
|
|
40
|
+
/* Verbs whose quoted arguments this scanner will call text. Every one of them either prints its argument or
|
|
41
|
+
* matches with it, and none has a documented way to execute it. Widening this list is safe in the sense the
|
|
42
|
+
* header sets out, but each entry should be able to answer "how would this run its argument?" with "it cannot". */
|
|
43
|
+
const QUOTING_VERBS: ReadonlySet<string> = new Set([
|
|
44
|
+
"echo",
|
|
45
|
+
"printf",
|
|
46
|
+
"rg",
|
|
47
|
+
"grep",
|
|
48
|
+
"egrep",
|
|
49
|
+
"fgrep",
|
|
50
|
+
"ack",
|
|
51
|
+
"ag",
|
|
52
|
+
"ripgrep",
|
|
53
|
+
]);
|
|
54
|
+
|
|
55
|
+
/* A flag whose value is a message: git's -m, and the long spelling. A quoted string here is prose that reaches
|
|
56
|
+
* a commit, a tag or a PR, and `git commit -m "rm -rf the old build dir"` is one of the more ordinary ways to
|
|
57
|
+
* write a dangerous-looking command that is not one. Read on the WORD BEFORE a quoted argument, so it applies
|
|
58
|
+
* whatever the verb is. */
|
|
59
|
+
const MESSAGE_FLAGS: ReadonlySet<string> = new Set(["-m", "--message", "-am", "--body", "-b"]);
|
|
60
|
+
|
|
61
|
+
// Where an unquoted word ends: whitespace, a pipeline or list operator, a redirect, a subshell paren.
|
|
62
|
+
const WORD_END = /[\s;|&<>()]/;
|
|
63
|
+
|
|
64
|
+
// A word that is a variable assignment prefixing a command (`FOO=bar cmd …`), skipped when looking for the verb.
|
|
65
|
+
const ASSIGNMENT = /^[A-Za-z_][A-Za-z0-9_]*=/;
|
|
66
|
+
|
|
67
|
+
/* `<<EOF`, `<<-EOF`, `<<'EOF'`, `<<"EOF"`. The delimiter's quoting only decides whether the body expands, which
|
|
68
|
+
* changes nothing here: an expanded body is still a body being written somewhere rather than run. */
|
|
69
|
+
const HEREDOC_OPEN = /<<-?\s*(['"]?)([A-Za-z_][A-Za-z0-9_]*)\1/g;
|
|
70
|
+
|
|
71
|
+
/* The heredoc bodies in a command, as spans. Found first and in their own pass, because a body is line-oriented
|
|
72
|
+
* and can hold anything at all — an odd number of quotes in it would otherwise throw the word scanner out of
|
|
73
|
+
* step for the rest of the command. */
|
|
74
|
+
const heredocBodies = (command: string): CommandSpan[] => {
|
|
75
|
+
const bodies: CommandSpan[] = [];
|
|
76
|
+
for (const open of command.matchAll(HEREDOC_OPEN)) {
|
|
77
|
+
const indented = command.slice(open.index, open.index + 3).startsWith("<<-");
|
|
78
|
+
const newline = command.indexOf("\n", open.index + open[0].length);
|
|
79
|
+
if (newline === -1) {
|
|
80
|
+
// `cat <<EOF` with nothing after it: an unterminated heredoc, so there is no body to mark.
|
|
81
|
+
continue;
|
|
82
|
+
}
|
|
83
|
+
const start = newline + 1;
|
|
84
|
+
const terminator = new RegExp(`^${indented ? "[ \\t]*" : ""}${open[2] as string}[ \\t]*$`, "m");
|
|
85
|
+
const rest = command.slice(start);
|
|
86
|
+
const end = terminator.exec(rest)?.index;
|
|
87
|
+
// An unterminated body runs to the end of the command, which is what the shell would read too.
|
|
88
|
+
bodies.push({ start, end: end === undefined ? command.length : start + end });
|
|
89
|
+
}
|
|
90
|
+
return bodies;
|
|
91
|
+
};
|
|
92
|
+
|
|
93
|
+
// Is this offset inside one of the spans already found? Heredoc bodies are skipped wholesale by the word scan.
|
|
94
|
+
const within = (spans: readonly CommandSpan[], offset: number): CommandSpan | undefined =>
|
|
95
|
+
spans.find((span) => offset >= span.start && offset < span.end);
|
|
96
|
+
|
|
97
|
+
// A substitution inside double quotes, the one thing that makes a quoted argument a program again.
|
|
98
|
+
const EXPANDS = /\$\(|`/;
|
|
99
|
+
|
|
100
|
+
/* One quoted segment inside a word: the span covers the quotes as well as what is between them, so a region
|
|
101
|
+
* handed back from here contains the whole `"rm -rf /"` and a match on any part of it reads as contained. */
|
|
102
|
+
interface QuotedSegment {
|
|
103
|
+
readonly span: CommandSpan;
|
|
104
|
+
/* Does a shell expand anything in here? `"$(rm -rf /)"` inside an echo is a real delete whose output is
|
|
105
|
+
* printed, so a double-quoted segment carrying a substitution is NOT text and never becomes a region. Single
|
|
106
|
+
* quotes expand nothing, so they never set this. */
|
|
107
|
+
readonly expands: boolean;
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
interface ShellWord {
|
|
111
|
+
readonly start: number;
|
|
112
|
+
// The word with its quoting removed, which is what a verb name and a flag are compared against.
|
|
113
|
+
readonly text: string;
|
|
114
|
+
readonly quoted: readonly QuotedSegment[];
|
|
115
|
+
// Does this word begin a simple command? True for the first word after a separator or at the start.
|
|
116
|
+
readonly opensCommand: boolean;
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
// A verb by its bare name: `/bin/echo` and `./echo` are echo, and the path in front of it says nothing new.
|
|
120
|
+
const bareVerb = (text: string): string => text.slice(Math.max(text.lastIndexOf("/"), text.lastIndexOf("\\")) + 1);
|
|
121
|
+
|
|
122
|
+
// Every list and pipeline operator: past one of these the next word is a verb again.
|
|
123
|
+
const SEPARATORS = new Set(["\n", ";", "|", "&", "(", ")"]);
|
|
124
|
+
const BLANKS = new Set([" ", "\t", "\r"]);
|
|
125
|
+
|
|
126
|
+
/* One quoted run, from its opening quote to its closing one. An unbalanced quote takes the rest of the
|
|
127
|
+
* command, which is what a shell waiting for more input would do and keeps the caller advancing. */
|
|
128
|
+
const readQuoted = (command: string, open: number): QuotedSegment & { readonly text: string; readonly end: number } => {
|
|
129
|
+
const quote = command[open] as string;
|
|
130
|
+
const close = command.indexOf(quote, open + 1);
|
|
131
|
+
const end = close === -1 ? command.length : close + 1;
|
|
132
|
+
const text = command.slice(open + 1, close === -1 ? command.length : close);
|
|
133
|
+
return { span: { start: open, end }, expands: quote === '"' && EXPANDS.test(text), text, end };
|
|
134
|
+
};
|
|
135
|
+
|
|
136
|
+
/* One word, from `start` to whatever ends it, with the quoted runs inside it kept as spans. `end === start`
|
|
137
|
+
* cannot happen: the caller only enters here on a character that is neither blank nor a separator. */
|
|
138
|
+
const readWord = (command: string, start: number): { readonly word: Omit<ShellWord, "opensCommand">; readonly end: number } => {
|
|
139
|
+
const quoted: QuotedSegment[] = [];
|
|
140
|
+
let text = "";
|
|
141
|
+
let index = start;
|
|
142
|
+
while (index < command.length && !WORD_END.test(command[index] as string)) {
|
|
143
|
+
const char = command[index] as string;
|
|
144
|
+
if (char === "'" || char === '"') {
|
|
145
|
+
const segment = readQuoted(command, index);
|
|
146
|
+
quoted.push({ span: segment.span, expands: segment.expands });
|
|
147
|
+
text += segment.text;
|
|
148
|
+
index = segment.end;
|
|
149
|
+
continue;
|
|
150
|
+
}
|
|
151
|
+
if (char === "\\") {
|
|
152
|
+
text += command[index + 1] ?? "";
|
|
153
|
+
index += 2;
|
|
154
|
+
continue;
|
|
155
|
+
}
|
|
156
|
+
text += char;
|
|
157
|
+
index += 1;
|
|
158
|
+
}
|
|
159
|
+
return { word: { start, text, quoted }, end: index };
|
|
160
|
+
};
|
|
161
|
+
|
|
162
|
+
/* The command taken apart into words, with each word's quoted segments and whether it opens a simple command.
|
|
163
|
+
* Deliberately a scanner rather than a parser: it tracks quoting and the separators that start a new command,
|
|
164
|
+
* and it knows nothing about control flow, functions or expansion. Everything it gets wrong is bounded by the
|
|
165
|
+
* header's argument. */
|
|
166
|
+
const scanWords = (command: string, skip: readonly CommandSpan[]): ShellWord[] => {
|
|
167
|
+
const words: ShellWord[] = [];
|
|
168
|
+
let index = 0;
|
|
169
|
+
let opensCommand = true;
|
|
170
|
+
while (index < command.length) {
|
|
171
|
+
const skipped = within(skip, index);
|
|
172
|
+
if (skipped !== undefined) {
|
|
173
|
+
index = skipped.end;
|
|
174
|
+
continue;
|
|
175
|
+
}
|
|
176
|
+
const char = command[index] as string;
|
|
177
|
+
if (SEPARATORS.has(char)) {
|
|
178
|
+
opensCommand = true;
|
|
179
|
+
index += 1;
|
|
180
|
+
continue;
|
|
181
|
+
}
|
|
182
|
+
if (BLANKS.has(char)) {
|
|
183
|
+
index += 1;
|
|
184
|
+
continue;
|
|
185
|
+
}
|
|
186
|
+
if (char === "#") {
|
|
187
|
+
// A comment runs to the end of the line. Only reached at a word boundary, so `a#b` is one word.
|
|
188
|
+
const newline = command.indexOf("\n", index);
|
|
189
|
+
index = newline === -1 ? command.length : newline;
|
|
190
|
+
continue;
|
|
191
|
+
}
|
|
192
|
+
const { word, end } = readWord(command, index);
|
|
193
|
+
// A character that is neither a separator nor part of a word (a stray redirect): step over it so the
|
|
194
|
+
// loop always advances.
|
|
195
|
+
index = end === index ? index + 1 : end;
|
|
196
|
+
if (end !== word.start) {
|
|
197
|
+
words.push({ ...word, opensCommand });
|
|
198
|
+
opensCommand = false;
|
|
199
|
+
}
|
|
200
|
+
}
|
|
201
|
+
return words;
|
|
202
|
+
};
|
|
203
|
+
|
|
204
|
+
/* WHERE A COMMAND HOLDS TEXT RATHER THAN A PROGRAM, as spans over the command, unsorted and possibly
|
|
205
|
+
* overlapping — callers ask containment questions of them rather than rendering them.
|
|
206
|
+
*
|
|
207
|
+
* Exported for the classifier, which asks it once per command and hands the answer to every table. */
|
|
208
|
+
export const inertRegions = (command: string): CommandSpan[] => {
|
|
209
|
+
const bodies = heredocBodies(command);
|
|
210
|
+
const regions: CommandSpan[] = [...bodies];
|
|
211
|
+
const words = scanWords(command, bodies);
|
|
212
|
+
// Comments are consumed by the scanner rather than reported, so they are found again here: the scan is
|
|
213
|
+
// where the quoting state lives, and a `#` inside a quoted string is not a comment.
|
|
214
|
+
let quotingVerb = false;
|
|
215
|
+
let previous: ShellWord | undefined;
|
|
216
|
+
for (const word of words) {
|
|
217
|
+
if (word.opensCommand) {
|
|
218
|
+
quotingVerb = QUOTING_VERBS.has(bareVerb(word.text));
|
|
219
|
+
previous = undefined;
|
|
220
|
+
}
|
|
221
|
+
/* An env assignment in front of the verb (`LC_ALL=C grep …`) is not the verb. Re-read the next word as
|
|
222
|
+
* one instead of giving up on the command. */
|
|
223
|
+
if (word.opensCommand && ASSIGNMENT.test(word.text) && word.quoted.length === 0) {
|
|
224
|
+
quotingVerb = false;
|
|
225
|
+
previous = undefined;
|
|
226
|
+
continue;
|
|
227
|
+
}
|
|
228
|
+
const afterMessageFlag = previous !== undefined && MESSAGE_FLAGS.has(previous.text);
|
|
229
|
+
if (quotingVerb || afterMessageFlag) {
|
|
230
|
+
regions.push(...word.quoted.filter((segment) => !segment.expands).map((segment) => segment.span));
|
|
231
|
+
}
|
|
232
|
+
previous = word;
|
|
233
|
+
}
|
|
234
|
+
regions.push(...commentRegions(command, bodies));
|
|
235
|
+
return regions;
|
|
236
|
+
};
|
|
237
|
+
|
|
238
|
+
/* Comments, on the same scan discipline as the words: a `#` only opens one where a word could have started, so
|
|
239
|
+
* `sha#1` is not a comment, and a quoted `"# …"` is not one either.
|
|
240
|
+
*
|
|
241
|
+
* Its own walk rather than a by-product of scanWords, because the two want different things from a quote — the
|
|
242
|
+
* word scan needs what is INSIDE one, this needs only to be past it. Sharing readQuoted keeps them agreeing on
|
|
243
|
+
* where one ends, which is the only fact they both depend on. */
|
|
244
|
+
const commentRegions = (command: string, skip: readonly CommandSpan[]): CommandSpan[] => {
|
|
245
|
+
const comments: CommandSpan[] = [];
|
|
246
|
+
let index = 0;
|
|
247
|
+
while (index < command.length) {
|
|
248
|
+
const skipped = within(skip, index);
|
|
249
|
+
if (skipped !== undefined) {
|
|
250
|
+
index = skipped.end;
|
|
251
|
+
continue;
|
|
252
|
+
}
|
|
253
|
+
const char = command[index] as string;
|
|
254
|
+
if (char === "'" || char === '"') {
|
|
255
|
+
index = readQuoted(command, index).end;
|
|
256
|
+
continue;
|
|
257
|
+
}
|
|
258
|
+
if (char === "\\") {
|
|
259
|
+
index += 2;
|
|
260
|
+
continue;
|
|
261
|
+
}
|
|
262
|
+
// A word boundary in front is what makes it a comment rather than part of a word.
|
|
263
|
+
if (char === "#" && (index === 0 || WORD_END.test(command[index - 1] as string))) {
|
|
264
|
+
const newline = command.indexOf("\n", index);
|
|
265
|
+
const end = newline === -1 ? command.length : newline;
|
|
266
|
+
comments.push({ start: index, end });
|
|
267
|
+
index = end;
|
|
268
|
+
continue;
|
|
269
|
+
}
|
|
270
|
+
index += 1;
|
|
271
|
+
}
|
|
272
|
+
return comments;
|
|
273
|
+
};
|
|
274
|
+
|
|
275
|
+
/* Is this fragment somewhere a shell would RUN it? The question the hard rule asks of every span triage matched.
|
|
276
|
+
*
|
|
277
|
+
* JUDGED ON WHERE THE FRAGMENT STARTS, not on whether a region contains the whole of it, and the difference is
|
|
278
|
+
* not a relaxation — it is the only reading that answers the question asked. Every pattern in the catalog is
|
|
279
|
+
* written to BEGIN at the verb (`rm`, `docker volume rm`, `mkfs`, `git push`), so the span's first character is
|
|
280
|
+
* where the dangerous thing was found. Whether the match then ran on past a closing quote says something about
|
|
281
|
+
* the regex, not about the command: `rm`'s own parser reads to the end of the invocation and `>` is not a
|
|
282
|
+
* terminator, so `echo "rm -rf /" >> notes.md` produces a span covering `rm -rf /" >> notes.md`. Requiring
|
|
283
|
+
* containment made that command live — which is exactly the card this whole change exists to stop raising.
|
|
284
|
+
*
|
|
285
|
+
* The conservative direction is preserved where it matters: a fragment whose verb sits OUTSIDE every region is
|
|
286
|
+
* live no matter what it runs into afterwards, so `echo "tidying" && rm -rf /` and `rm -rf "$(cat list)" # ok`
|
|
287
|
+
* both stay live. */
|
|
288
|
+
export const isLive = (span: CommandSpan, regions: readonly CommandSpan[]): boolean =>
|
|
289
|
+
!regions.some((region) => span.start >= region.start && span.start < region.end);
|
package/src/versions.ts
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
/* COMPARING THE VERSIONS THIS SYSTEM STAMPS ON WHAT IT SHIPS, the daemon, the sandbox image, and the two agents
|
|
2
|
-
* that run on a user's own
|
|
2
|
+
* that run on a user's own device. One release stamps all of them to the SAME version, so "is this one behind
|
|
3
3
|
* that one" is one question with one answer, and it lives here because both ends ask it: the daemon compares its
|
|
4
|
-
* own build against the latest published release, and the browser compares a
|
|
4
|
+
* own build against the latest published release, and the browser compares a device's agent against the same.
|
|
5
5
|
*
|
|
6
6
|
* Shared rather than copied because the two copies would not disagree until the day it mattered: 1.9.0 against
|
|
7
7
|
* 1.10.0 is where a hand-rolled comparator goes wrong, and it goes wrong by reporting "up to date". */
|
package/src/webext-links.ts
CHANGED
|
@@ -35,7 +35,7 @@ export const webextLendUrl = (sandboxUrl: string): string => `${sandboxUrl.repla
|
|
|
35
35
|
|
|
36
36
|
/* ---- the pairing code: the one string that travels from the sandbox's card into the extension ----
|
|
37
37
|
*
|
|
38
|
-
* A connected
|
|
38
|
+
* A connected device is paired by a shell one-liner, which can carry two values in two environment variables
|
|
39
39
|
* because a terminal is a place where long strings are normal. A browser extension's popup is not: what a
|
|
40
40
|
* person will actually do there is paste ONE thing, once, and anything that asks them to copy a URL into one
|
|
41
41
|
* box and a token into another is a flow that fails on the second box.
|
|
@@ -43,7 +43,7 @@ export const webextLendUrl = (sandboxUrl: string): string => `${sandboxUrl.repla
|
|
|
43
43
|
* So both halves ride in one code. It is not encryption and does not pretend to be — base64url of two fields,
|
|
44
44
|
* so that the thing on the clipboard is opaque enough not to be edited by hand, short enough to paste, and
|
|
45
45
|
* carries its own sandbox address, which is the field a person could not possibly be expected to type. The
|
|
46
|
-
* secret in it is the pairing token, which is single-use and expires in ten minutes (
|
|
46
|
+
* secret in it is the pairing token, which is single-use and expires in ten minutes (the daemon's peer store).
|
|
47
47
|
*
|
|
48
48
|
* The prefix is a version marker, and it is here so that a code from an older sandbox meets a clear "this code
|
|
49
49
|
* is from a different version" in the extension rather than a JSON parse error. */
|
package/src/webext-protocol.ts
CHANGED
|
@@ -2,13 +2,13 @@ import { z } from "zod";
|
|
|
2
2
|
|
|
3
3
|
/* The handshake on /system/webext/connect, the ONE message on that socket that is not oRPC.
|
|
4
4
|
*
|
|
5
|
-
* Same two-phase shape as a connected
|
|
5
|
+
* Same two-phase shape as a connected device's (host-protocol.ts) and for the same reason: the daemon has
|
|
6
6
|
* nothing to call until it knows whose socket this is, so the proof cannot itself be an oRPC call. The browser
|
|
7
7
|
* extension's first frame carries its enrollment token, the daemon resolves WHICH capability it belongs to, and
|
|
8
8
|
* from that message on every byte is `webextContract` with the EXTENSION serving.
|
|
9
9
|
*
|
|
10
10
|
* WHY A SEPARATE PROTOCOL FROM host's, when the frame is the same two fields: because the thing on the other
|
|
11
|
-
* end is not a
|
|
11
|
+
* end is not a device. It has no shell, no filesystem and no screen; what it has is tabs, origins the person
|
|
12
12
|
* granted one at a time, and a human watching every click. Sharing the host's schema would have meant a card of
|
|
13
13
|
* switches that mean nothing (`roots`, `sandboxRemove`) and an agent told about a home directory it cannot
|
|
14
14
|
* reach. The two connectors are siblings, not one connector with a flag. */
|
|
@@ -6,6 +6,9 @@ import {
|
|
|
6
6
|
isLockedWorkspacePath,
|
|
7
7
|
isReportedManifest,
|
|
8
8
|
isReviewableLockedPath,
|
|
9
|
+
LOCKED_STATE_ENTRIES,
|
|
10
|
+
lockedWorkspaceEntry,
|
|
11
|
+
PLAN_DOCUMENTS_DIR,
|
|
9
12
|
REPORTED_MANIFEST_PATHS,
|
|
10
13
|
SEARCHABLE_STATE_PATHS,
|
|
11
14
|
SHARED_STATE_PATHS,
|
|
@@ -244,6 +247,47 @@ describe(`isLockedWorkspacePath`, () => {
|
|
|
244
247
|
expect(isLockedWorkspacePath(`.intentic\\secrets\\auth\\codex`)).toBe(true);
|
|
245
248
|
expect(isLockedWorkspacePath(`./.intentic/config/capabilities.json`)).toBe(true);
|
|
246
249
|
});
|
|
250
|
+
|
|
251
|
+
it(`lets the plan documents out of the session store around them`, () => {
|
|
252
|
+
// The plan a card asks the reader to approve, whose full text that card already renders. Refusing the
|
|
253
|
+
// file left the card's one link into the workspace landing on a padlock about the document it was
|
|
254
|
+
// asking about; the transcripts it sits beside stay locked.
|
|
255
|
+
expect(isLockedWorkspacePath(`${PLAN_DOCUMENTS_DIR}/wiggly-spring.md`)).toBe(false);
|
|
256
|
+
expect(isLockedWorkspacePath(PLAN_DOCUMENTS_DIR)).toBe(false);
|
|
257
|
+
expect(isLockedWorkspacePath(`.intentic/records/sessions/claude/projects/x.jsonl`)).toBe(true);
|
|
258
|
+
expect(isLockedWorkspacePath(`.intentic/records/sessions`)).toBe(true);
|
|
259
|
+
// …and it is the directory that is exempt, not the word: a sibling store named for it is not one.
|
|
260
|
+
expect(isLockedWorkspacePath(`.intentic/records/sessions/claude/plans-backup/x.md`)).toBe(true);
|
|
261
|
+
});
|
|
262
|
+
});
|
|
263
|
+
|
|
264
|
+
/* WHICH entry a locked path belongs to, which is what the refusal screen says a file holds and where to manage
|
|
265
|
+
* it. Split out from the boolean so the browser can key its sentences on the daemon's own list instead of a
|
|
266
|
+
* second copy of the rule — the copy it kept drifted through the state regrouping and stranded every locked
|
|
267
|
+
* file on the generic sentence. */
|
|
268
|
+
describe(`lockedWorkspaceEntry`, () => {
|
|
269
|
+
it(`names the entry a path matched, not the leaf it ends at`, () => {
|
|
270
|
+
// A locked FOLDER is one row in the explorer and never descended, so the name worth reporting is the
|
|
271
|
+
// folder's: "Cookies is kept private" is true of something the reader has never heard of.
|
|
272
|
+
expect(lockedWorkspaceEntry(`.intentic/local/browser/Default/Cookies`)).toBe(`local/browser`);
|
|
273
|
+
expect(lockedWorkspaceEntry(`.intentic/secrets/auth/codex/auth.json`)).toBe(`secrets/auth`);
|
|
274
|
+
expect(lockedWorkspaceEntry(`.intentic/config/capabilities.json`)).toBe(`config/capabilities.json`);
|
|
275
|
+
expect(lockedWorkspaceEntry(`.git/config`)).toBe(`.git`);
|
|
276
|
+
});
|
|
277
|
+
|
|
278
|
+
it(`answers undefined for everything the lock does not hold`, () => {
|
|
279
|
+
expect(lockedWorkspaceEntry(`src/app.ts`)).toBeUndefined();
|
|
280
|
+
expect(lockedWorkspaceEntry(`.intentic/config/settings.json`)).toBeUndefined();
|
|
281
|
+
expect(lockedWorkspaceEntry(`${PLAN_DOCUMENTS_DIR}/wiggly-spring.md`)).toBeUndefined();
|
|
282
|
+
});
|
|
283
|
+
|
|
284
|
+
it(`answers for every entry the lock declares`, () => {
|
|
285
|
+
// The set is the daemon's; this is what makes it addressable from the browser. An entry nobody can
|
|
286
|
+
// resolve back out is one the refusal screen could only describe generically.
|
|
287
|
+
for (const entry of LOCKED_STATE_ENTRIES) {
|
|
288
|
+
expect([entry, lockedWorkspaceEntry(`${STATE_DIR}/${entry}`)]).toEqual([entry, entry]);
|
|
289
|
+
}
|
|
290
|
+
});
|
|
247
291
|
});
|
|
248
292
|
|
|
249
293
|
/* The carve-out the diff routes ask for, and the reason it is derived: a locked entry the root repo TRACKS has
|
|
@@ -330,10 +374,13 @@ describe(`VERSIONED_STATE_PATHS`, () => {
|
|
|
330
374
|
/* The connections themselves, and the entry that took the longest to earn its place: it was classed
|
|
331
375
|
* `secret` on the strength of holding each capability's credential, which stopped being true when the
|
|
332
376
|
* vault took the values out and left the shape behind. Connecting a deployment orchestrator, or
|
|
333
|
-
* granting a connected
|
|
377
|
+
* granting a connected device shell and screen control, is the largest change made to what this
|
|
334
378
|
* sandbox can DO, and it used to leave no diff. */
|
|
335
379
|
`${STATE_DIR}/config/capabilities.json`,
|
|
336
380
|
`${STATE_DIR}/config/capability-dismissals.json`,
|
|
381
|
+
// The context shelves: which repositories a conversation opened on one carries. A list of names, and
|
|
382
|
+
// the decision about what a session may see, which is what a review is for.
|
|
383
|
+
`${STATE_DIR}/config/context/`,
|
|
337
384
|
/* The two entries the AGENT authors on its own initiative, and the reason `versioned` is not read as
|
|
338
385
|
* config-only. Both are the sandbox acting outward: a draft publishes words under the owner's name,
|
|
339
386
|
* a workspace extension is code that runs in the app and can serve HTTP with the workspace under
|