cowork-harness 2.3.0 → 2.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude/skills/cowork-harness/SKILL.md +16 -9
- package/.claude/skills/cowork-harness/references/ci-recipe.md +5 -5
- package/.claude/skills/cowork-harness/references/critique.md +1 -1
- package/.claude/skills/cowork-harness/references/fidelity-and-answers.md +1 -1
- package/.claude/skills/cowork-harness/references/scenario-schema.md +1 -1
- package/.claude/skills/cowork-harness/references/task-recipes.md +1 -1
- package/.claude/skills/cowork-harness/scripts/scenario.py +24 -0
- package/CHANGELOG.md +130 -0
- package/README.md +5 -5
- package/dist/assert.js +28 -3
- package/dist/hostloop/workspace-handler.js +12 -2
- package/dist/run/cassette.js +49 -1
- package/dist/run/execute.js +76 -8
- package/dist/runtime/container.js +53 -2
- package/dist/runtime/hostloop.js +52 -8
- package/dist/sync/cowork-sync.js +26 -8
- package/dist/types.js +1 -1
- package/docs/fidelity-gaps.md +128 -5
- package/docs/scenario.md +53 -7
- package/examples/replays/README.md +1 -1
- package/examples/replays/example-pdf-skill.cassette.json +161 -112
- package/examples/scenarios/example-pdf-skill.yaml +8 -1
- package/package.json +1 -1
- package/python/test_scenario_lint.py +39 -0
- package/schema/scenario.schema.json +1 -1
package/dist/run/execute.js
CHANGED
|
@@ -22,7 +22,7 @@ import { loadBaseline } from "../baseline.js";
|
|
|
22
22
|
import { loadSession, resolveSessionPaths, buildLaunchPlan, userVisibleRootsFromPlan, readonlyFolderRootsFromPlan, deleteDeniedRootsFromPlan, pluginSkillRootsFromPlan, isConnectedContent, } from "../session.js";
|
|
23
23
|
import { spawnProtocol } from "../runtime/protocol.js";
|
|
24
24
|
import { spawnContainer } from "../runtime/container.js";
|
|
25
|
-
import { spawnHostLoop, WORKSPACE_TOOL_ALIASES } from "../runtime/hostloop.js";
|
|
25
|
+
import { spawnHostLoop, WORKSPACE_TOOL_ALIASES, VM_LOOP_TOOL_ALIASES } from "../runtime/hostloop.js";
|
|
26
26
|
import { snapshotHostLoopWorkspace } from "../runtime/hostloop-stage.js";
|
|
27
27
|
import { checkHostLoopWriteConsent, logHostWriteNotice } from "../hostloop/safety.js";
|
|
28
28
|
import { warnUnservedHookEvents } from "./hook-events.js";
|
|
@@ -558,6 +558,9 @@ export async function executeScenario(scenario, opts = {}) {
|
|
|
558
558
|
let containerName;
|
|
559
559
|
let deregisterContainerReap; // Ctrl-C cleanup for the agent container
|
|
560
560
|
let hostEgress; // host-routed web_fetch egress
|
|
561
|
+
// Container's web_fetch is host-routed too, so its decisions cannot come from the proxy log and must
|
|
562
|
+
// survive the `egress = eg.entries` teardown assignment. Kept separate for exactly that reason.
|
|
563
|
+
const containerWebFetchEgress = [];
|
|
561
564
|
let hostloopHooks; // hostloop's PreToolUse path-gate bundle
|
|
562
565
|
let hostloopPathGateFired; // tool_use_ids the path gate actually saw
|
|
563
566
|
let hostloopInfraErrors; // spawnHostLoop's live infra sink (sidecar crash + failed execs, tagged by origin) — folded into record.infraErrors below
|
|
@@ -732,6 +735,12 @@ export async function executeScenario(scenario, opts = {}) {
|
|
|
732
735
|
runToken,
|
|
733
736
|
suggestSkillsEnabled,
|
|
734
737
|
proactiveSkillSuggestEnabled,
|
|
738
|
+
// Same gate production reads for its VM-loop web_fetch registration. Read once above and shared
|
|
739
|
+
// with the hostloop branch so the two tiers cannot drift apart on it.
|
|
740
|
+
webFetchViaApi: viaApiOn,
|
|
741
|
+
provenanceRef,
|
|
742
|
+
dedup,
|
|
743
|
+
onEgress: (e) => containerWebFetchEgress.push(e),
|
|
735
744
|
});
|
|
736
745
|
child = ct.child;
|
|
737
746
|
containerName = ct.containerName; // so the Ctrl-C / finally reap removes the agent container by name
|
|
@@ -794,7 +803,11 @@ export async function executeScenario(scenario, opts = {}) {
|
|
|
794
803
|
// fill the provenance bundle (backed by Run's tracker + recorded approval) BEFORE drive().
|
|
795
804
|
// Host-loop only, and only when the web_fetch-via-API gate is on; otherwise the handler stays
|
|
796
805
|
// allowlist-only (ref.current undefined). Run seeds the set from turns + tool_results.
|
|
797
|
-
|
|
806
|
+
// BOTH loops, not just host-loop: production's VM-loop factory calls the provenance path
|
|
807
|
+
// unconditionally and has no allowlist fallback in it at all. Leaving `ref.current` undefined at
|
|
808
|
+
// container drops the handler onto PATH B — the gate-OFF path — even though the tool exists only
|
|
809
|
+
// BECAUSE the gate is on, which is the inverse of production.
|
|
810
|
+
if ((effectiveFidelity === "hostloop" || effectiveFidelity === "container") && viaApiOn) {
|
|
798
811
|
run.enableWebFetchGate();
|
|
799
812
|
provenanceRef.current = {
|
|
800
813
|
isAllowed: (u) => run.provenanceHas(u),
|
|
@@ -811,7 +824,15 @@ export async function executeScenario(scenario, opts = {}) {
|
|
|
811
824
|
subagentAppend: prompts.subagentAppend,
|
|
812
825
|
sdkMcp,
|
|
813
826
|
hooks: hostloopHooks,
|
|
814
|
-
|
|
827
|
+
// Host-loop aliases Bash+WebFetch; the VM loop aliases WebFetch alone, and only when the
|
|
828
|
+
// gate that put the workspace server there is on. Tied to the SAME `viaApiOn` that drives the
|
|
829
|
+
// disallow in spawnContainer — an alias to a server this run never registered would resolve a
|
|
830
|
+
// bare WebFetch onto nothing.
|
|
831
|
+
...(effectiveFidelity === "hostloop"
|
|
832
|
+
? { toolAliases: WORKSPACE_TOOL_ALIASES }
|
|
833
|
+
: effectiveFidelity === "container" && viaApiOn
|
|
834
|
+
? { toolAliases: VM_LOOP_TOOL_ALIASES }
|
|
835
|
+
: {}),
|
|
815
836
|
});
|
|
816
837
|
}
|
|
817
838
|
catch (e) {
|
|
@@ -857,9 +878,14 @@ export async function executeScenario(scenario, opts = {}) {
|
|
|
857
878
|
egressMalformedLines += eg.malformedLines; // applied to record.evidenceErrors after the finally, where `record` is assigned (#39)
|
|
858
879
|
sidecar.teardown();
|
|
859
880
|
}
|
|
860
|
-
// merge host-routed web_fetch decisions
|
|
881
|
+
// merge host-routed web_fetch decisions so they're visible to egress assertions. This MUST come
|
|
882
|
+
// after the `egress = eg.entries` above, which replaces the array wholesale with the proxy log —
|
|
883
|
+
// a log that can never contain a host-side fetch. Both loops route web_fetch off-container, so
|
|
884
|
+
// both need the merge; container's decisions were being discarded by that assignment.
|
|
861
885
|
if (hostEgress?.length)
|
|
862
886
|
egress = [...egress, ...hostEgress];
|
|
887
|
+
if (containerWebFetchEgress.length)
|
|
888
|
+
egress = [...egress, ...containerWebFetchEgress];
|
|
863
889
|
hostProxy?.close();
|
|
864
890
|
}
|
|
865
891
|
// A post-listen egress-sidecar crash (container topology) surfaces as `fatalError` after teardown —
|
|
@@ -1089,9 +1115,21 @@ export async function executeScenario(scenario, opts = {}) {
|
|
|
1089
1115
|
// on-disk content), so a claim about a written artifact is presentation-stable (not a paste-vs-write
|
|
1090
1116
|
// coin-flip). Captured here — BEFORE the semantic pre-pass below — using the pre-run manifest to diff
|
|
1091
1117
|
// added/modified files. (`[]` when there's no manifest, e.g. a --resume run.)
|
|
1092
|
-
// F12: at container/hostloop the agent's cwd is the SESSION
|
|
1093
|
-
//
|
|
1094
|
-
//
|
|
1118
|
+
// F12, CORRECTED 2026-08-27: the old text said "at container/hostloop the agent's cwd is the SESSION
|
|
1119
|
+
// ROOT". True at CONTAINER only. At hostloop the agent process sits at `mnt/outputs` (see
|
|
1120
|
+
// `hostLoopCwds` in src/runtime/hostloop.ts), so a bare `Write` there lands INSIDE `workRoot` and needs
|
|
1121
|
+
// no scratchpad walk to be seen. The branch is still right, but for two different reasons per tier:
|
|
1122
|
+
// container — agent cwd IS the session root, so a relative `Write` lands outside `workRoot`;
|
|
1123
|
+
// hostloop — the agent writes inside `mnt`, but `mcp__workspace__bash` starts at the session root,
|
|
1124
|
+
// so a relative SHELL write lands outside `workRoot`.
|
|
1125
|
+
// Either way `workRoot` ends `/session/mnt` and its parent is the root, so passing it captures what the
|
|
1126
|
+
// run actually authored.
|
|
1127
|
+
//
|
|
1128
|
+
// KNOWN FALSE-GREEN, deliberately not fixed here (see docs/fidelity-gaps.md, "Path resolution"):
|
|
1129
|
+
// production DISCARDS anything written outside `mnt/` — "never reaches the user or your file tools" —
|
|
1130
|
+
// while the harness bind-mounts the whole session dir, so these files persist and can be graded as
|
|
1131
|
+
// authored. That is correct for the semantic judge (the run did write them) and wrong as a model of
|
|
1132
|
+
// delivery. `user_visible_artifact` is unaffected: it checks user-visible ROOTS, not this set.
|
|
1095
1133
|
const scratchpadRoot = workRoot.endsWith(`${sep}mnt`) ? dirname(workRoot) : undefined;
|
|
1096
1134
|
// On a resume the session root is REUSED, so the scratchpad no longer starts empty — a prior turn's files
|
|
1097
1135
|
// would be mis-attributed as this turn's authorship. Skip the scratchpad walk in that case (evidence-
|
|
@@ -1519,10 +1557,37 @@ const isFileRelative = (p) => p !== "(inline)" && !isAbsolute(p) && !p.startsWit
|
|
|
1519
1557
|
* file's directory (not the cwd), so a scenario+session bundle is self-contained and
|
|
1520
1558
|
* relocatable. Use this everywhere a scenario is read from disk (`run`, `record`).
|
|
1521
1559
|
*/
|
|
1560
|
+
/** True when the YAML did not name a tier, so `fidelity` came from the schema default.
|
|
1561
|
+
*
|
|
1562
|
+
* Must be read from the RAW document: Zod's `.default("container")` makes the parsed object
|
|
1563
|
+
* indistinguishable from one that said `fidelity: container` on purpose, and those two cases deserve
|
|
1564
|
+
* different treatment — an author who chose the tier has made the choice, one who omitted it has not.
|
|
1565
|
+
*
|
|
1566
|
+
* Why anyone cares: the default models the VM-LOOP lane, and production runs HOST-LOOP (gate 1143815894
|
|
1567
|
+
* is force-ON in every shipped baseline). So a scenario that omits the key is measured against the lane
|
|
1568
|
+
* real users are not on — silently. Measured 2026-08-27; see docs/fidelity-gaps.md, "Path resolution". */
|
|
1569
|
+
export function fidelityWasDefaulted(raw) {
|
|
1570
|
+
return typeof raw === "object" && raw !== null && !("fidelity" in raw);
|
|
1571
|
+
}
|
|
1572
|
+
/** The deprecation notice for a defaulted tier. `fidelity` becomes REQUIRED in the next major; until
|
|
1573
|
+
* then this warns rather than failing, so consumers get told before they get an error. */
|
|
1574
|
+
export function defaultedFidelityNotice(name) {
|
|
1575
|
+
return (`::warning:: [scenario] ${name}: no \`fidelity:\` — defaulting to \`container\`, which models the ` +
|
|
1576
|
+
`VM-LOOP lane. Production runs HOST-LOOP by default (gate 1143815894), so this scenario is likely ` +
|
|
1577
|
+
`measured against a lane your users are not on: the file tools resolve a bare relative path ` +
|
|
1578
|
+
`differently, the shell starts somewhere else, and the offered tool set differs. Name a tier ` +
|
|
1579
|
+
`explicitly — \`fidelity: hostloop\` to match production, \`fidelity: cowork\` to auto-pick the way ` +
|
|
1580
|
+
`Cowork does, or \`fidelity: container\` to keep today's behaviour deliberately. Switching tiers can ` +
|
|
1581
|
+
`COST you assertions: \`no_scratchpad_leak\` is container-only (a lint error elsewhere) and ` +
|
|
1582
|
+
`\`transcript_no_host_path\` fails by design at hostloop/protocol. ` +
|
|
1583
|
+
`DEPRECATION: the default is being removed — \`fidelity:\` becomes REQUIRED in the next major.`);
|
|
1584
|
+
}
|
|
1522
1585
|
export function parseScenarioFile(path) {
|
|
1523
1586
|
let scenario;
|
|
1587
|
+
let rawDoc;
|
|
1524
1588
|
try {
|
|
1525
|
-
|
|
1589
|
+
rawDoc = parseYaml(readFileSync(path, "utf8"));
|
|
1590
|
+
scenario = Scenario.parse(rawDoc);
|
|
1526
1591
|
}
|
|
1527
1592
|
catch (e) {
|
|
1528
1593
|
// A schema violation is a USER mistake (a typo'd/retired key like `profile:`, a bad enum value),
|
|
@@ -1535,6 +1600,9 @@ export function parseScenarioFile(path) {
|
|
|
1535
1600
|
// `name` defaults to the filename (sans extension) — the file is the identity.
|
|
1536
1601
|
if (!scenario.name)
|
|
1537
1602
|
scenario.name = basename(path).replace(/\.ya?ml$/i, "");
|
|
1603
|
+
// Warn, do not fail: this is the deprecation window before `fidelity` becomes required.
|
|
1604
|
+
if (fidelityWasDefaulted(rawDoc))
|
|
1605
|
+
process.stderr.write(defaultedFidelityNotice(scenario.name) + "\n");
|
|
1538
1606
|
if (isFileRelative(scenario.session))
|
|
1539
1607
|
scenario.session = resolve(dirname(path), scenario.session);
|
|
1540
1608
|
// Load-time regex validation: fail fast with a clear message rather than letting a malformed pattern
|
|
@@ -9,9 +9,28 @@ import { capturePreRunManifest } from "../run/pre-run-manifest.js";
|
|
|
9
9
|
import { makeCoworkHandler } from "../hostloop/cowork-handler.js";
|
|
10
10
|
import { makeSkillsHandler, SKILLS_PLUGINS_TOOL_NAMES } from "../hostloop/skills-handler.js";
|
|
11
11
|
import { makePluginsHandler } from "../hostloop/plugins-handler.js";
|
|
12
|
+
import { makeWorkspaceHandler } from "../hostloop/workspace-handler.js";
|
|
12
13
|
import { combineSdkMcp } from "../agent/session.js";
|
|
13
14
|
import { listMountedSkills } from "../run/skill-metadata.js";
|
|
14
15
|
import { resolveAgentImage, resolveContainerRuntime } from "./agent-image.js";
|
|
16
|
+
/** The VM loop's web_fetch surface, as three coupled answers derived from ONE gate reading.
|
|
17
|
+
*
|
|
18
|
+
* Exported and pure because the bug this prevents lives at the CALL SITE, not in any one value: the
|
|
19
|
+
* three parts (advertise, do-not-pre-approve, disallow the built-in) are only correct together, and
|
|
20
|
+
* each can be dropped independently without any other test noticing. An adversarial review deleted the
|
|
21
|
+
* disallow and the alias and the entire suite still passed.
|
|
22
|
+
*
|
|
23
|
+
* - `advertised` — the workspace tool the model can see.
|
|
24
|
+
* - `preApproved` — deliberately EMPTY. Production gates web_fetch at can_use_tool (its VM-loop
|
|
25
|
+
* registration passes the same approval hook the host loop does), so pre-approving
|
|
26
|
+
* would make a scripted `webfetch:<domain>` answer and `decide: deny` silently inert.
|
|
27
|
+
* - `disallowed` — the built-in name production removes. Ships with the alias in execute.ts; without
|
|
28
|
+
* that alias a bare `WebFetch` resolves onto nothing instead of the workspace tool. */
|
|
29
|
+
export function vmLoopWebFetchSurface(webFetchViaApi) {
|
|
30
|
+
if (!webFetchViaApi)
|
|
31
|
+
return { advertised: [], preApproved: [], disallowed: [] };
|
|
32
|
+
return { advertised: ["mcp__workspace__web_fetch"], preApproved: [], disallowed: ["WebFetch"] };
|
|
33
|
+
}
|
|
15
34
|
/**
|
|
16
35
|
* L1 — container parity runtime. Runs the staged in-VM agent in a sandboxed arm64
|
|
17
36
|
* Linux container that reproduces the Desktop→agent spawn contract (asar 1.12603.1):
|
|
@@ -71,6 +90,12 @@ export function spawnContainer(_scenario, baseline, plan, outDir, sessionId, opt
|
|
|
71
90
|
// `lane: remote` serves no cowork server, so the tool must not be advertised or pre-approved either:
|
|
72
91
|
// a registered tool with no backing server is a phantom capability the model can try and fail to use.
|
|
73
92
|
const coworkTools = plan.lane === "remote" ? [] : ["mcp__cowork__present_files"];
|
|
93
|
+
// Mirror production's VM-loop web_fetch swap: the built-in name goes away and the workspace tool takes
|
|
94
|
+
// its place. ADVERTISED but deliberately NOT pre-approved — production's VM-loop registration passes
|
|
95
|
+
// the same `requestWebFetchApproval` hook the host loop does, so the call is gated at can_use_tool.
|
|
96
|
+
// `spawnHostLoop` splits extraTools/extraAllowedTools for exactly this reason; pre-approving here
|
|
97
|
+
// would make a scripted `webfetch:<domain>` answer, and a `decide: deny` on it, silently inert.
|
|
98
|
+
const { advertised: webFetchTools, disallowed: webFetchDisallowed } = vmLoopWebFetchSurface(!!opts.webFetchViaApi);
|
|
74
99
|
const claudeArgs = agentArgs(baseline, plan, {
|
|
75
100
|
mntRoot,
|
|
76
101
|
systemPromptAppend: opts.systemPromptAppend,
|
|
@@ -83,7 +108,10 @@ export function spawnContainer(_scenario, baseline, plan, outDir, sessionId, opt
|
|
|
83
108
|
// The 5 skills/plugins discovery tools are declared on the SAME cowork lane as present_files (spec
|
|
84
109
|
// §3: `isEnabled` = `sessionType==="cowork"`, which container satisfies) — pre-approved for the same
|
|
85
110
|
// off-registry-auto-allow reason present_files is.
|
|
86
|
-
|
|
111
|
+
disallowed: webFetchDisallowed,
|
|
112
|
+
extraTools: [...coworkTools, ...webFetchTools, ...SKILLS_PLUGINS_TOOL_NAMES],
|
|
113
|
+
// web_fetch is absent here ON PURPOSE — see webFetchTools above. It is the one registered tool this
|
|
114
|
+
// tier does not pre-approve, because production gates it at can_use_tool.
|
|
87
115
|
extraAllowedTools: [...coworkTools, ...SKILLS_PLUGINS_TOOL_NAMES],
|
|
88
116
|
});
|
|
89
117
|
const dockerArgs = dockerRunArgv({
|
|
@@ -133,7 +161,30 @@ export function spawnContainer(_scenario, baseline, plan, outDir, sessionId, opt
|
|
|
133
161
|
servers: ["plugins"],
|
|
134
162
|
handle: makePluginsHandler({ mountedPlugins }),
|
|
135
163
|
};
|
|
136
|
-
|
|
164
|
+
// VM-LOOP web_fetch. Production's non-host-loop site, gated on `coworkWebFetchViaApi`, registers a
|
|
165
|
+
// workspace server exposing **web_fetch only**, disallows the built-in `WebFetch`, and aliases the name.
|
|
166
|
+
// Bash is deliberately untouched here — that replacement is host-loop-only, which is why this tier keeps
|
|
167
|
+
// the built-in shell. `containerName` is supplied because the handler's type wants it; the web_fetch
|
|
168
|
+
// path never execs into the container.
|
|
169
|
+
const workspaceBundle = opts.webFetchViaApi
|
|
170
|
+
? {
|
|
171
|
+
servers: ["workspace"],
|
|
172
|
+
handle: makeWorkspaceHandler({
|
|
173
|
+
containerName,
|
|
174
|
+
vmMnt: `${sessionRoot}/mnt`,
|
|
175
|
+
provenanceRef: opts.provenanceRef,
|
|
176
|
+
dedup: opts.dedup,
|
|
177
|
+
onEgress: opts.onEgress,
|
|
178
|
+
// MUST be passed. The handler defaults this to ["*"], and compile(["*"]) is `() => true` — so
|
|
179
|
+
// omitting it hands the tier an UNRESTRICTED fetcher that runs in the harness's own Node
|
|
180
|
+
// process, outside the container network namespace and therefore invisible to the sidecar
|
|
181
|
+
// proxy. That silently voids this tier's default-deny egress promise for this one tool.
|
|
182
|
+
webFetchAllow: plan.egressAllow,
|
|
183
|
+
tools: ["web_fetch"],
|
|
184
|
+
}),
|
|
185
|
+
}
|
|
186
|
+
: undefined;
|
|
187
|
+
const sdkMcp = combineSdkMcp(...(workspaceBundle ? [workspaceBundle] : []), ...(coworkBundle ? [coworkBundle] : []), skillsBundle, pluginsBundle);
|
|
137
188
|
// `sessionRoot` is the VM path the agent sees (`-w` above, and the cowork handler's own
|
|
138
189
|
// `sessionRootVm`). Returned so the caller classifies present_files against the root THIS spawn used,
|
|
139
190
|
// instead of re-deriving one — the two lived in different path spaces (host vs VM) once, which made
|
package/dist/runtime/hostloop.js
CHANGED
|
@@ -33,9 +33,19 @@ export const HOSTLOOP_PATH_GATE_ID = "hostloop-path-gate";
|
|
|
33
33
|
/** Production's host-loop tool aliases (asar `et()`; single-hop, deny rules do NOT expand across the
|
|
34
34
|
* alias). An alias never GRANTS a tool — it resolves only when the target is already in the caller's
|
|
35
35
|
* bound set, so a bare Bash/WebFetch from a sub-agent without the workspace tool bound still fails.
|
|
36
|
-
*
|
|
37
|
-
*
|
|
36
|
+
* BOTH names are host-loop-only. The VM loop replaces web_fetch alone and never touches Bash — use
|
|
37
|
+
* `VM_LOOP_TOOL_ALIASES` there, not this map. */
|
|
38
38
|
export const WORKSPACE_TOOL_ALIASES = { Bash: "mcp__workspace__bash", WebFetch: "mcp__workspace__web_fetch" };
|
|
39
|
+
/** The VM loop's alias set: web_fetch ONLY, applied when `coworkWebFetchViaApi` is on.
|
|
40
|
+
*
|
|
41
|
+
* "Bash is the only tool that truly diverges between loops" — the VM loop keeps the built-in shell and
|
|
42
|
+
* aliases only WebFetch, so aliasing Bash here would invent a tool production does not replace.
|
|
43
|
+
*
|
|
44
|
+
* Without this, disallowing `WebFetch` at container is a REGRESSION rather than a fidelity fix: the
|
|
45
|
+
* built-in stops resolving and nothing catches the bare name, so a model that emits `WebFetch` hard-fails
|
|
46
|
+
* where production silently resolves it to the workspace tool. The disallow and the alias are two halves
|
|
47
|
+
* of one behaviour and must ship together. */
|
|
48
|
+
export const VM_LOOP_TOOL_ALIASES = { WebFetch: "mcp__workspace__web_fetch" };
|
|
39
49
|
/**
|
|
40
50
|
* Pure builder for the hostloop native process's env: `hostNativeSpawnEnv`'s contract-layer output
|
|
41
51
|
* layered over this REAL macOS process's own `...process.env` base (unlike container/microvm, which
|
|
@@ -69,6 +79,23 @@ export function buildHostLoopNativeEnv(baseline, opts) {
|
|
|
69
79
|
* (/tmp→/private/tmp, /var→/private/var on macOS) makes the agent's realpath'd wire cwd differ from the
|
|
70
80
|
* un-canonicalized spawner cwd even for the SAME directory — a false alarm this collapses. The gate
|
|
71
81
|
* decision is unaffected (it realpaths candidate and roots itself); this only governs the diagnostic. */
|
|
82
|
+
/** The host-loop cwd SPLIT, in one place because the two halves are only correct TOGETHER.
|
|
83
|
+
*
|
|
84
|
+
* Production keeps them deliberately different, measured on desktop-local Cowork 2026-08-27:
|
|
85
|
+
* - the agent process sits at the OUTPUTS dir, so its file tools resolve a bare `Write` there;
|
|
86
|
+
* - every `mcp__workspace__bash` call starts at the bare SESSION ROOT, and bash resets its cwd
|
|
87
|
+
* between calls ("no cwd/env carryover"), so a relative shell path can never be `cd`-ed elsewhere.
|
|
88
|
+
*
|
|
89
|
+
* Cowork's own sub-agent prompt states the second half: "Each command starts in `<vmCwd>`; anything
|
|
90
|
+
* written outside `<vmCwd>/mnt/` (including /tmp) stays in that environment and never reaches the user
|
|
91
|
+
* or your file tools."
|
|
92
|
+
*
|
|
93
|
+
* Collapsing them — which this harness did until 2026-08-27, running bash at the outputs dir — makes a
|
|
94
|
+
* skill that writes relative paths from a script look correct here and deliver nothing in production.
|
|
95
|
+
* Keep them as one function so a future edit cannot move one and leave the other. */
|
|
96
|
+
export function hostLoopCwds(sessionRoot, hostOutputsDir) {
|
|
97
|
+
return { agentProcessCwd: hostOutputsDir, workspaceBashCwd: sessionRoot };
|
|
98
|
+
}
|
|
72
99
|
export function pathGateCwdMismatch(wireCwd, spawnerCwd) {
|
|
73
100
|
const canon = (p) => {
|
|
74
101
|
try {
|
|
@@ -274,7 +301,11 @@ export function spawnHostLoop(_scenario, baseline, plan, outDir, sessionId, opts
|
|
|
274
301
|
return checkHostLoopPathGate(input?.tool_name, input?.tool_input ?? {}, gateCfg);
|
|
275
302
|
},
|
|
276
303
|
};
|
|
277
|
-
const child = spawn(agentNativeHost, nativeArgs, {
|
|
304
|
+
const child = spawn(agentNativeHost, nativeArgs, {
|
|
305
|
+
cwd: hostLoopCwds(sessionRoot, hostOutputsDir).agentProcessCwd,
|
|
306
|
+
env: nativeEnv,
|
|
307
|
+
stdio: ["pipe", "pipe", "pipe"],
|
|
308
|
+
});
|
|
278
309
|
// The VM sidecar container: bash/web_fetch's `docker exec` target. No agent inside it (the agent is
|
|
279
310
|
// the native `child` above) — it runs a keep-alive command (dockerRunArgv's default when `agentArgv` is
|
|
280
311
|
// omitted). Folders are bind-mounted here as REAL host paths (never copied); `.claude/skills`+
|
|
@@ -310,11 +341,24 @@ export function spawnHostLoop(_scenario, baseline, plan, outDir, sessionId, opts
|
|
|
310
341
|
const infraErrors = [];
|
|
311
342
|
const { logSidecarInfra, logExecInfra } = makeInfraEmitters(outDir, infraErrors);
|
|
312
343
|
const { markTearingDown } = watchHostLoopSidecar(sidecarChild, logSidecarInfra);
|
|
313
|
-
//
|
|
314
|
-
// outputs
|
|
315
|
-
//
|
|
316
|
-
|
|
317
|
-
|
|
344
|
+
// Every `mcp__workspace__bash` call starts at the bare SESSION ROOT — not a connected folder, not
|
|
345
|
+
// outputs. MEASURED on desktop-local Cowork 2026-08-27, twice: `pwd` returned `/sessions/<id>` with no
|
|
346
|
+
// folder connected AND with one connected. Cowork's own sub-agent prompt says the same thing: "Each
|
|
347
|
+
// command starts in `<vmCwd>`; anything written outside `<vmCwd>/mnt/` (including /tmp) stays in that
|
|
348
|
+
// environment and never reaches the user or your file tools."
|
|
349
|
+
//
|
|
350
|
+
// This REPLACES a `${sessionRoot}/mnt/${firstFolder ?? "outputs"}` derivation whose comment claimed
|
|
351
|
+
// production's vmCwd was the first connected folder "never the bare session root". That claim came from
|
|
352
|
+
// the asar's `cwd: c.vmCwd` spawn argument, which is NOT load-bearing on the cowork path — only the
|
|
353
|
+
// `chat` branch prepends an explicit `cd ${vmCwd}`, which would be redundant if the argument worked.
|
|
354
|
+
// The old value was never faithful: it reproduced a prompt claim rather than an observed behaviour.
|
|
355
|
+
//
|
|
356
|
+
// It is deliberately NOT the agent-process cwd. The agent spawns at `hostOutputsDir` (see the
|
|
357
|
+
// `spawn(agentNativeHost, …, { cwd: hostOutputsDir })` above) so its FILE TOOLS resolve relative paths
|
|
358
|
+
// against outputs, while the SHELL sits at the session root. Production keeps those two values
|
|
359
|
+
// different on purpose; collapsing them is the bug this replaces. Both are pinned together in
|
|
360
|
+
// test/baseline.test.ts — a single-value assertion cannot express the split.
|
|
361
|
+
const execCwd = hostLoopCwds(sessionRoot, hostOutputsDir).workspaceBashCwd;
|
|
318
362
|
// Host-routed web_fetch bypasses the sidecar proxy, so collect its egress decisions here and
|
|
319
363
|
// surface them to execute.ts → result.egress, making host-loop web_fetch visible to egress assertions.
|
|
320
364
|
const hostEgress = [];
|
package/dist/sync/cowork-sync.js
CHANGED
|
@@ -46,8 +46,11 @@ export const PINNED_GATES = {
|
|
|
46
46
|
// replaces the hardcoded subagent_env_hl / subagent_env_vm fallback texts (resolveSection). OFF live
|
|
47
47
|
// (source defaultValue) -> the hardcoded texts are the wire text the committed paraphrase assets
|
|
48
48
|
// model. An ON flip is invisible to the text sentinel (the hardcoded template is unchanged), so
|
|
49
|
-
// checkSubagentOverrideGate
|
|
50
|
-
//
|
|
49
|
+
// checkSubagentOverrideGate emits a non-blocking WARNING on ON (downgraded from a hard delta
|
|
50
|
+
// 2026-08-27): gate-ON only enables the lookup, and the payload that would actually override is
|
|
51
|
+
// delivered per-session by the server — invisible to every input sync reads. A guard that can never
|
|
52
|
+
// clear itself from its own inputs blocks forever rather than tripping. Settled by a live sub-agent
|
|
53
|
+
// probe instead; see the message body.
|
|
51
54
|
"124685897": "subagentPromptServerOverride",
|
|
52
55
|
// Spawn-env conditional gates: each controls a key in the Desktop→agent spawn env
|
|
53
56
|
// (SPAWN_GATES). Pinned so a production flip surfaces BOTH as a provenance.gates diff AND as the
|
|
@@ -351,10 +354,23 @@ export function checkSubagentOverrideGate(gates) {
|
|
|
351
354
|
if (!gates?.["124685897"]?.on)
|
|
352
355
|
return [];
|
|
353
356
|
return [
|
|
354
|
-
"gate subagentPromptServerOverride:124685897 reads ON —
|
|
355
|
-
"
|
|
356
|
-
"
|
|
357
|
-
"
|
|
357
|
+
"gate subagentPromptServerOverride:124685897 reads ON — the sub-agent append MAY be server-overridden " +
|
|
358
|
+
"and this sync CANNOT TELL from its own inputs. Gate-ON only enables the lookup: the asar reads the " +
|
|
359
|
+
"section entry and, when it is missing or empty, logs `using hardcoded fallback` and returns the " +
|
|
360
|
+
"built-in text anyway. The entry is delivered PER SESSION by the server — it is in neither the asar, " +
|
|
361
|
+
"the fcache nor config.json — so gate state alone cannot separate 'override active' from 'gate on, " +
|
|
362
|
+
"no payload, fallback still correct'. " +
|
|
363
|
+
"This gate is SERVER-SIDE and Desktop-version-INDEPENDENT (it flipped off->on via " +
|
|
364
|
+
'`source:"defaultValue"` with the asar byte-identical, 1.37937.1 -> .3) — do NOT go looking for a ' +
|
|
365
|
+
"Desktop change. " +
|
|
366
|
+
"WARNING, not a refusal: probed live 2026-08-27 on desktop-local Cowork (agent 2.1.246, hl branch, " +
|
|
367
|
+
"no folder connected). A real sub-agent's environment section matched the committed asset on all " +
|
|
368
|
+
"four load-bearing claims — host cwd, mcp__workspace__bash in an isolated Linux env, folders under " +
|
|
369
|
+
"<vmCwd>/mnt/, and shell starting in <vmCwd> with non-mnt writes reaching neither the user nor the " +
|
|
370
|
+
"file tools — so NO override was reaching that account and the committed paraphrase is faithful. " +
|
|
371
|
+
"That is EVIDENCE, NOT PROOF: one account, one session, and a server rule can be segment-targeted. " +
|
|
372
|
+
"If the sub-agent append matters to what you are about to ship, re-probe (dispatch a sub-agent, ask " +
|
|
373
|
+
"for its environment section verbatim, diff the four claims) rather than trusting this note.",
|
|
358
374
|
];
|
|
359
375
|
}
|
|
360
376
|
/** Read `network.allowDomains` from the NEWEST committed baseline — the pinned, hand-curated egress
|
|
@@ -445,8 +461,10 @@ export function sync() {
|
|
|
445
461
|
// only count gates that actually matched a live fcache feature.
|
|
446
462
|
flag(unknown, "gates: fcache decoded but NONE of the pinned gate IDs matched — gate IDs may have been re-keyed; update PINNED_GATES in cowork-sync.ts");
|
|
447
463
|
}
|
|
448
|
-
|
|
449
|
-
|
|
464
|
+
// WARNING, not a hard delta — downgraded 2026-08-27 on MEASURED evidence, see checkSubagentOverrideGate.
|
|
465
|
+
// It blocked the write while being unable to distinguish the two states it names; a guard that can
|
|
466
|
+
// never clear itself from its own inputs is a permanent block, not a tripwire.
|
|
467
|
+
notes.push(...checkSubagentOverrideGate(gates));
|
|
450
468
|
return {
|
|
451
469
|
appVersion,
|
|
452
470
|
agentVersion,
|
package/dist/types.js
CHANGED
|
@@ -719,7 +719,7 @@ export const ScenarioObject = z.strictObject({
|
|
|
719
719
|
fidelity: z
|
|
720
720
|
.enum(FIDELITY_TIERS)
|
|
721
721
|
.default("container")
|
|
722
|
-
.describe("isolation tier: protocol (L0, no sandbox) | container/microvm (force a VM-loop tier) | hostloop (force host-loop) | cowork (auto-pick host-loop vs. container via Cowork's own gate logic)"),
|
|
722
|
+
.describe("isolation tier: protocol (L0, no sandbox) | container/microvm (force a VM-loop tier) | hostloop (force host-loop) | cowork (auto-pick host-loop vs. container via Cowork's own gate logic). DEPRECATION: omitting this key is deprecated and the field becomes REQUIRED in the next major. The `container` default models the VM loop, while production runs the host loop by default (gate 1143815894), so an omitted key likely measures the scenario against a lane your users are not on — a bare relative path lands elsewhere, the shell starts elsewhere, and the offered tool set differs. Name a tier: hostloop to match production, cowork to auto-pick the way Cowork does, or container to keep the current behaviour deliberately."),
|
|
723
723
|
// execution LOCATION, orthogonal to `fidelity` (a local privilege tier) — do NOT collapse the two.
|
|
724
724
|
// `cloud-describe` is RESERVED: no runner exists yet, so authoring it is a load-time error (see
|
|
725
725
|
// execute.ts's validateScenarioRegexes, which mirrors the `replay_protocol_fidelity` rejection).
|
package/docs/fidelity-gaps.md
CHANGED
|
@@ -181,6 +181,46 @@ The `--help` text notes this at runtime. If you need to test egress policy, use
|
|
|
181
181
|
|
|
182
182
|
---
|
|
183
183
|
|
|
184
|
+
## Browser tools are not served — and egress assertions say nothing about that path
|
|
185
|
+
|
|
186
|
+
**Real Cowork behaviour:** Desktop `1.37937.1` serves an in-app browser to Cowork agent sessions as
|
|
187
|
+
`mcp__Claude_Browser__*` — 16 tools (`browser_batch`, `computer`, `find`, `form_input`, `get_page_text`,
|
|
188
|
+
`javascript_tool`, `navigate`, `preview_start`, `read_console_messages`, `read_network_requests`,
|
|
189
|
+
`read_page`, `resize_window`, `tabs_close`, `tabs_context`, `tabs_create`, `tabs_select`). Two gates
|
|
190
|
+
carry it, `17519066` and `3990395613`; both read `on:true source:"force"` in the live fcache
|
|
191
|
+
(2026-08-27). Chat mode does not receive them. The tool list comes from real sessions' `system/init`
|
|
192
|
+
arrays; the browser's *behaviour* is unprobed, so treat the limits Anthropic's prompt text describes
|
|
193
|
+
(no `file://`, no Claude-started `localhost`) as documentation rather than observation.
|
|
194
|
+
|
|
195
|
+
**Harness behaviour:** none of the 16 are served, and neither gate is in `PINNED_GATES`. A skill that
|
|
196
|
+
calls one gets an unknown-tool error — a false-RED, and the cheaper half of this gap.
|
|
197
|
+
|
|
198
|
+
**The half that matters is what a GREEN does not cover.** Egress entries have exactly two sources: the
|
|
199
|
+
sidecar proxy's `onDecision` (container traffic, which is what `bash` uses) and the workspace
|
|
200
|
+
`web_fetch`'s `onEgress`. In production the browser pane runs Desktop-side — it traverses neither the
|
|
201
|
+
container nor the harness's proxy. So this tier of the contract has a hole that no assertion reports:
|
|
202
|
+
|
|
203
|
+
> **`egress_denied` and the other `egress_*` assertions do not cover browser-tool traffic.** They are
|
|
204
|
+
> evidence about `bash` and `web_fetch` only. A run where they pass is silent about whether the skill
|
|
205
|
+
> could reach the network through the browser in production — not proof that it could not.
|
|
206
|
+
|
|
207
|
+
That silence is easy to over-read, because "default-deny egress held" is exactly the sentence a reader
|
|
208
|
+
wants from a green egress assertion. It is true of the paths the harness models and says nothing about
|
|
209
|
+
the one it does not.
|
|
210
|
+
|
|
211
|
+
**Why the tools are not modelled:** driving a real browser pane is disproportionate to the value.
|
|
212
|
+
Declaring the 16 names so calls resolve is a much smaller step than serving behaviour, and the
|
|
213
|
+
declaration-vs-rendered distinction elsewhere in this document covers the difference — worth doing if a
|
|
214
|
+
real skill needs it, not before. Anything added must be gate-conditional: this is per-account
|
|
215
|
+
(force-ON here, not necessarily elsewhere), so it needs both gate ids pinned first.
|
|
216
|
+
|
|
217
|
+
**One sync trap, latent today.** `preview_start` is served in Cowork with a **different input schema**
|
|
218
|
+
than its base definition — `{url}` (open a tab; explicitly no dev server on that surface) versus
|
|
219
|
+
`{name}` (start a dev server from `.claude/launch.json`). Same tool name, no shared field, chosen by
|
|
220
|
+
session kind at tool-list construction. Nothing in the harness reasons about that tool, so nothing is
|
|
221
|
+
wrong today; if a sentinel or `deriveSpawnEnv` ever does, the surface qualifier is load-bearing and a
|
|
222
|
+
name-only match would silently pick the wrong schema.
|
|
223
|
+
|
|
184
224
|
## No session resume in `chat`
|
|
185
225
|
|
|
186
226
|
**Real Cowork behaviour:** Sessions persist and can be resumed across launches.
|
|
@@ -464,11 +504,39 @@ read-only categories (uploads hardlink write-block, spool, plugin) ARE modeled.
|
|
|
464
504
|
`web_fetch` (gate `coworkWebFetchViaApi`, live true) — "Bash is the only tool that truly diverges
|
|
465
505
|
between loops."
|
|
466
506
|
|
|
467
|
-
**Harness behaviour:**
|
|
468
|
-
|
|
469
|
-
|
|
470
|
-
|
|
471
|
-
|
|
507
|
+
**Harness behaviour:** CLOSED at `container`, still OPEN at `microvm`.
|
|
508
|
+
|
|
509
|
+
Host-loop sets both aliases (`WORKSPACE_TOOL_ALIASES`, hostloop.ts) on the `initialize`
|
|
510
|
+
control_request. `container` now models the VM-loop registration: when `coworkWebFetchViaApi` resolves
|
|
511
|
+
on (true in every shipped baseline), it registers a workspace server exposing **web_fetch only**,
|
|
512
|
+
disallows the built-in `WebFetch`, and aliases the name via `VM_LOOP_TOOL_ALIASES`. Bash is deliberately
|
|
513
|
+
untouched there — "Bash is the only tool that truly diverges between loops" — which is why `container`
|
|
514
|
+
keeps the built-in shell while host-loop replaces it.
|
|
515
|
+
|
|
516
|
+
The three parts ship together on purpose. Disallowing the built-in **without** the alias would turn a
|
|
517
|
+
fidelity fix into a regression: the bare name would stop resolving instead of landing on the workspace
|
|
518
|
+
tool. Registering the tool without pre-approving it would trip the off-registry auto-allow guard.
|
|
519
|
+
|
|
520
|
+
**`chat` does not do the swap.** `chat --fidelity container` spawns without the gate, so it still offers
|
|
521
|
+
the built-in `WebFetch` and no `mcp__workspace__web_fetch`. The swap is a `run`/`record` property today.
|
|
522
|
+
Debugging a skill in `chat` and then running it as a scenario therefore presents two different tool sets
|
|
523
|
+
at the same declared tier — check the tier's inventory in the run's `system/init` rather than assuming
|
|
524
|
+
`chat` matches.
|
|
525
|
+
|
|
526
|
+
**`microvm` is still unaliased** — `spawnMicroVm` does not receive the gate, so that tier continues to
|
|
527
|
+
offer the built-in `WebFetch` and a bare emission there does not reach a workspace server. Use
|
|
528
|
+
`container` or `hostloop` for alias-sensitive scenarios.
|
|
529
|
+
|
|
530
|
+
**Consequence for scenarios:** at `container` the offered set contains `mcp__workspace__web_fetch`, and
|
|
531
|
+
not `WebFetch`. An assertion naming the built-in (`tool_called: WebFetch`, `tool_not_called: WebFetch`)
|
|
532
|
+
therefore names a tool that does not exist at this tier — and `tool_not_called` passes **vacuously**
|
|
533
|
+
rather than failing loudly, which is the harder direction to notice. Name `mcp__workspace__web_fetch`
|
|
534
|
+
instead — **but that name is itself tier-dependent**: `microvm` and `protocol` never offer it, so the same
|
|
535
|
+
assertion is permanently vacuous THERE. There is no single spelling that is meaningful at every tier,
|
|
536
|
+
because the tiers genuinely differ; switching a scenario's `fidelity:` can silently turn a web-fetch
|
|
537
|
+
assertion into a tautology in either direction. Pair it with a positive assertion that fails if the tool
|
|
538
|
+
set is not what you assumed, or keep the scenario on the tier it was written for. `verify-cassettes` emits a `replaced-builtin` note when a recorded init inventory names a
|
|
539
|
+
built-in absent from the tier's current set.
|
|
472
540
|
|
|
473
541
|
## Skill/plugin discovery SDK-MCP servers — modeled on container/hostloop; microvm/protocol pending
|
|
474
542
|
|
|
@@ -892,6 +960,61 @@ branch never promotes, so there is no scratch→outputs copy to leak there.
|
|
|
892
960
|
path model to work against; `microvm` stages into a different tree than the artifact scan walks. Both
|
|
893
961
|
report can't-verify rather than passing vacuously.
|
|
894
962
|
|
|
963
|
+
### Path resolution: the shell and the file tools use DIFFERENT roots (measured 2026-08-27)
|
|
964
|
+
|
|
965
|
+
**The two roots are modeled; what remains divergent is whether the write SURVIVES.** On the
|
|
966
|
+
desktop-local lane, production resolves a relative path differently depending on which tool writes it:
|
|
967
|
+
|
|
968
|
+
| | production (host-loop) | `container`/`microvm` (VM-loop) | `hostloop` |
|
|
969
|
+
|---|---|---|---|
|
|
970
|
+
| file tools (`Write`/`Read`) base | `mnt/outputs` | session root | `mnt/outputs` ✓ |
|
|
971
|
+
| `mcp__workspace__bash` cwd | session root | session root ✓ | session root ✓ |
|
|
972
|
+
|
|
973
|
+
Both cwds come from a single function (`hostLoopCwds`, `src/runtime/hostloop.ts`) and are pinned together
|
|
974
|
+
in one test, because a single-value assertion cannot express "the shell and the file tools disagree, on
|
|
975
|
+
purpose" — and collapsing them into one value is the mistake this arrangement exists to prevent.
|
|
976
|
+
|
|
977
|
+
Confirmed by Cowork's own sub-agent prompt: *"Each command starts in `<vmCwd>`; anything written outside
|
|
978
|
+
`<vmCwd>/mnt/` (including `/tmp`) stays in that environment and never reaches the user or your file
|
|
979
|
+
tools."* Bash also **resets its cwd between every call** (*"no cwd/env carryover"*), so a relative path
|
|
980
|
+
from a script always resolves against the session root and cannot be `cd`-ed away from.
|
|
981
|
+
|
|
982
|
+
Three consequences — the first for anyone reading a green run, the other two for anyone writing a skill
|
|
983
|
+
or an assertion against these tools:
|
|
984
|
+
|
|
985
|
+
1. **A relative bash write lands in the right place but PERSISTS, where production discards it.**
|
|
986
|
+
Production throws away anything outside `mnt/`; the harness bind-mounts the whole session dir, so the
|
|
987
|
+
same write survives into the run dir. Measured 2026-08-27: `printf > rel.txt` from a shell tool landed
|
|
988
|
+
at `/sessions/<id>/rel.txt` on `container` and **persisted** as `session/rel.txt`.
|
|
989
|
+
|
|
990
|
+
**Scope, so this is not over-read:** `file_exists`, `user_visible_artifact` and
|
|
991
|
+
`computer_links_resolve` are all bounded by `workRoot` (`…/session/mnt`) and **cannot** reach such a
|
|
992
|
+
file. The exposure is `semantic_matches` and `no_lost_write_back`, which grade the *authored* set. That
|
|
993
|
+
set still contains the file — capturing it is correct, since the run did write it — but every entry
|
|
994
|
+
from the scratchpad walk is labelled `— SCRATCH, NOT delivered to the user` in the judged document,
|
|
995
|
+
with a note saying such files are evidence of what the run DID and not that anything was delivered.
|
|
996
|
+
That label is what keeps a rubric like "the report was written" from grading TRUE on a file the user
|
|
997
|
+
would never receive.
|
|
998
|
+
|
|
999
|
+
**What is still divergent:** production would have discarded the file entirely, so a skill whose
|
|
1000
|
+
bundled scripts take relative output paths still gets further here than it would there. The location
|
|
1001
|
+
half is now faithful on both tiers; the survival half is not, and removing it would mean unpicking a
|
|
1002
|
+
bind-mount every tier's artifact capture, `--resume` and the microvm snapshot depend on.
|
|
1003
|
+
2. **`Write`'s tool result echoes the RAW path it was given — it never absolutizes.** Structural, read
|
|
1004
|
+
from the agent binary. A bare `Write foo.md` comes back as `foo.md`, not
|
|
1005
|
+
`/sessions/<id>/mnt/outputs/foo.md`. Nothing in the harness parses a path out of a `Write` result
|
|
1006
|
+
today, and this note exists so nothing starts: an assertion or doc that reads an absolute path out of
|
|
1007
|
+
one is reading something production does not emit. Cowork's own chat-surface prompt asserts the
|
|
1008
|
+
opposite ("Write's result shows the file's full path"), so the product's documentation of its own tool
|
|
1009
|
+
is wrong here — do not take it as a spec.
|
|
1010
|
+
3. **The literal prefix `outputs/` DOUBLES on the desktop-local lane** (`outputs/x` → `outputs/outputs/x`,
|
|
1011
|
+
invisible), and **`<folder>/x` builds a same-named decoy inside `outputs`** rather than reaching the
|
|
1012
|
+
connected folder — silently, with a success result. No relative path from the file tools reaches a
|
|
1013
|
+
connected folder. See [scenario.md](./scenario.md), "Where a relative path actually lands".
|
|
1014
|
+
|
|
1015
|
+
The **cloud** lane shares none of this: cwd is `/home/claude`, there is no `mnt/` tree, and the shell and
|
|
1016
|
+
file tools share one root.
|
|
1017
|
+
|
|
895
1018
|
### Remote device bridge — `internal__remote-devices__*`, deliberately unmodeled
|
|
896
1019
|
|
|
897
1020
|
The remote lane also has an internal MCP server the harness does not model at all, wire-named
|