humanish 0.96.0 → 0.97.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (81) hide show
  1. package/README.md +4 -1
  2. package/dist/automatic-analysis-config.d.ts +15 -5
  3. package/dist/automatic-analysis-config.js +25 -4
  4. package/dist/automatic-analysis-config.js.map +1 -1
  5. package/dist/automatic-study-analysis.js +4 -2
  6. package/dist/automatic-study-analysis.js.map +1 -1
  7. package/dist/cua-actor-lab.d.ts +18 -286
  8. package/dist/cua-actor-lab.js +151 -1955
  9. package/dist/cua-actor-lab.js.map +1 -1
  10. package/dist/cua-desktop-lane.d.ts +35 -0
  11. package/dist/cua-desktop-lane.js +13 -0
  12. package/dist/cua-desktop-lane.js.map +1 -0
  13. package/dist/desktop-session.d.ts +41 -0
  14. package/dist/desktop-session.js +44 -0
  15. package/dist/desktop-session.js.map +1 -0
  16. package/dist/doctor-lab.d.ts +6 -1
  17. package/dist/doctor-lab.js +17 -3
  18. package/dist/doctor-lab.js.map +1 -1
  19. package/dist/e2b-cua-desktop.d.ts +3 -0
  20. package/dist/e2b-cua-desktop.js +675 -0
  21. package/dist/e2b-cua-desktop.js.map +1 -0
  22. package/dist/e2b-cua-provisioning.d.ts +311 -0
  23. package/dist/e2b-cua-provisioning.js +1213 -0
  24. package/dist/e2b-cua-provisioning.js.map +1 -0
  25. package/dist/e2b-desktop-session.d.ts +7 -0
  26. package/dist/e2b-desktop-session.js +29 -0
  27. package/dist/e2b-desktop-session.js.map +1 -0
  28. package/dist/index.d.ts +1 -1
  29. package/dist/lab-summary.d.ts +3 -1
  30. package/dist/local-agent-cli.js +1 -1
  31. package/dist/local-agent-cli.js.map +1 -1
  32. package/dist/observer-app.html +2 -2
  33. package/dist/program.js +24 -9
  34. package/dist/program.js.map +1 -1
  35. package/dist/restricted-codex-analysis.d.ts +15 -0
  36. package/dist/restricted-codex-analysis.js +13 -0
  37. package/dist/restricted-codex-analysis.js.map +1 -0
  38. package/dist/restricted-codex-policy.d.ts +56 -0
  39. package/dist/restricted-codex-policy.js +151 -0
  40. package/dist/restricted-codex-policy.js.map +1 -0
  41. package/dist/restricted-codex-session.d.ts +19 -0
  42. package/dist/restricted-codex-session.js +419 -0
  43. package/dist/restricted-codex-session.js.map +1 -0
  44. package/dist/restricted-codex-transport.d.ts +58 -0
  45. package/dist/restricted-codex-transport.js +233 -0
  46. package/dist/restricted-codex-transport.js.map +1 -0
  47. package/dist/run.d.ts +1 -1
  48. package/dist/run.js +1 -1
  49. package/dist/run.js.map +1 -1
  50. package/dist/study-analysis-codex-config.d.ts +11 -0
  51. package/dist/study-analysis-codex-config.js +34 -0
  52. package/dist/study-analysis-codex-config.js.map +1 -0
  53. package/dist/study-analysis-engine.d.ts +6 -3
  54. package/dist/study-analysis-engine.js +19 -10
  55. package/dist/study-analysis-engine.js.map +1 -1
  56. package/dist/study-analysis-job.d.ts +1 -0
  57. package/dist/study-analysis-job.js +1 -1
  58. package/dist/study-analysis-job.js.map +1 -1
  59. package/dist/study-analysis-provider.d.ts +4 -2
  60. package/dist/study-analysis-provider.js +1 -1
  61. package/dist/study-analysis-provider.js.map +1 -1
  62. package/dist/study-analysis-service.d.ts +3 -0
  63. package/dist/study-analysis-service.js +24 -6
  64. package/dist/study-analysis-service.js.map +1 -1
  65. package/dist/study-analysis-validation.d.ts +29 -5
  66. package/dist/study-analysis-validation.js +23 -10
  67. package/dist/study-analysis-validation.js.map +1 -1
  68. package/dist/study-analysis.d.ts +27 -2
  69. package/dist/tui-app.js +102 -102
  70. package/docs/architecture/desktop-sessions.md +80 -0
  71. package/docs/architecture/restricted-codex-analysis.md +89 -0
  72. package/docs/contracts/run-bundle.md +7 -4
  73. package/docs/contracts/schemas.md +1 -1
  74. package/docs/contracts/study-analysis.md +44 -2
  75. package/docs/goals/current.md +4 -4
  76. package/docs/product/automatic-analysis.md +24 -3
  77. package/docs/ramp/README.md +11 -1
  78. package/docs/release/0.96.1-browser-navigation.md +22 -0
  79. package/docs/release/0.97.0-codex-account-analysis.md +45 -0
  80. package/package.json +1 -1
  81. package/skills/humanish/SKILL.md +19 -0
@@ -1,7 +1,11 @@
1
- import { withTransientCommsSecrets } from "./run-narration-secrets.js";
1
+ export { inboxRecipientFor, laneHasInboxRecipient } from "./cua-desktop-lane.js";
2
+ export { CUA_ACTOR_LAB_PROVIDER_METADATA, DEFAULT_MOBILE_USER_AGENT, SANDBOX_CAMERA_PATH, SANDBOX_MEDIA_DIR, SUBJECT_DIR, SYNTHETIC_CAMERA_COMMAND, applyMobileEmulation, buildFillDesktopWindowCommand, captureDesktopBrowserGeometry, commandDigestOf, declaredScreenForRender, desktopBrowserFamily, inspectDesktopScreenGeometry, makeChromeBrowserStateObserver, makeChromeDesktopGeometryObserver, parseXwininfoGeometry, prepareDesktopMedia, provisionCloneSubject, provisionLocalTreeSubject } from "./e2b-cua-provisioning.js";
2
3
  import { prepareReceivingRun, receivingPublication } from "./comms-receiving-runtime.js";
3
- import { deployReceivingInbox } from "./comms-receiving-inbox.js";
4
+ import { laneHasInboxRecipient } from "./cua-desktop-lane.js";
5
+ import { createE2BCuaDesktopLane } from "./e2b-cua-desktop.js";
6
+ import { DEFAULT_STATE_STEP_TIMEOUT_MS, commandDigestOf, declaredScreenForRender } from "./e2b-cua-provisioning.js";
4
7
  import { receivingEmailValidationReason } from "./lab-config.js";
8
+ import { withTransientCommsSecrets } from "./run-narration-secrets.js";
5
9
  // The computer-use lab backend: a subject (an app-url the caller provisioned, or a repo the
6
10
  // lab clones AND serves in-sandbox) driven by a REGISTRY-RESOLVED computer-use actor inside a
7
11
  // hosted E2B desktop. This is the path that makes `actors[].type` load-bearing — the
@@ -24,52 +28,47 @@ import { receivingEmailValidationReason } from "./lab-config.js";
24
28
  // harness errors are redacted at THIS boundary; the bundle's `stream.actor` carries the
25
29
  // conformant humanish.actor-trace.v1 projection, whose `redaction.screenshots` records the
26
30
  // run's actual mode ("raw" | "blurred" | "n/a") — every label downstream derives from it.
27
- import { resolveAutomaticAnalysis } from "./automatic-analysis-config.js";
28
- import { completeAutomaticAnalysis, markFinalizedStudyResult } from "./automatic-analysis-completion.js";
29
- import { desktopMediaValidationReason, taskProtocolValidationReason } from "./lab-config.js";
30
31
  import { randomBytes } from "node:crypto";
31
- import { describeMissingKeys } from "./key-resolution.js";
32
32
  import { readFile, realpath, rm } from "node:fs/promises";
33
33
  import path from "node:path";
34
+ import { completeAutomaticAnalysis, markFinalizedStudyResult } from "./automatic-analysis-completion.js";
35
+ import { resolveAutomaticAnalysis } from "./automatic-analysis-config.js";
36
+ import { describeMissingKeys } from "./key-resolution.js";
37
+ import { desktopMediaValidationReason, taskProtocolValidationReason } from "./lab-config.js";
38
+ import { pathToFileURL } from "node:url";
39
+ import { toErrorMessage } from "./command-failure.js";
34
40
  import { cuaLaneDiagnostics, summarizeCuaDiagnostics } from "./cua-diagnostics.js";
35
41
  import { feedbackProofCommands } from "./feedback-proof.js";
36
- import { runDesktopCommandOrThrow, toErrorMessage } from "./command-failure.js";
37
- import { pathToFileURL } from "node:url";
38
- import { beginRunStatus, withRunStatusScope } from "./run-status.js";
39
- import { adapterScoreFailureMessage, applyBrowserAdapterHooks } from "./adapter-extension.js";
40
42
  import { actorRegistry, isCuaActorDescriptor } from "./actor-registry.js";
41
- import { CHROMIUM_EVIDENCE_HYGIENE_FLAGS, chromiumEvidenceProfilePreferencesJson } from "./browser-evidence-hygiene.js";
42
- import { DEFAULT_OPENAI_CU_MODEL } from "./openai-responses-cu.js";
43
- import { createLocalAgentProvider, detectLocalAgents } from "./local-agent-cli.js";
44
- import { startAppServerSession } from "./local-agent-appserver.js";
45
- import { startClaudeSession } from "./local-agent-claude-session.js";
46
- import { createDesktopSandbox, withOneRetryOnTransientE2BError, loadE2BDesktopModule } from "./e2b-desktop-launch.js";
47
- import { probeUrl, readDetachedLog, runDetachedStep, startDetachedProcess } from "./e2b-detached.js";
48
- import { DEFAULT_SANDBOX_CATCH_PORT, collectCommsThread, collectExternalCommsThread, deployCommsCatch, externalCatchHealthy, externalInboxUrl, refreshInboxSurface, writeInboxSurface } from "./comms-sandbox-catch.js";
43
+ import { actorEnding } from "./actor-stop-cause.js";
44
+ import { adapterScoreFailureMessage, applyBrowserAdapterHooks } from "./adapter-extension.js";
49
45
  import { FakeInbox } from "./comms-fake-inbox.js";
50
- import { buildOriginMap, recipientInboxUrl } from "./comms-inbox.js";
51
- import { DEFAULT_DEVICE_PRESET, isDevicePresetName, resolveDevicePreset } from "./device-presets.js";
52
- import { cuaLaneValidationReason, outputTokenLimitValidationReason, isHttpUrl, isLoopbackUrl, MAX_CUA_LANES, subjectStateInvalidReason } from "./lab-config.js";
46
+ import { recipientInboxUrl } from "./comms-inbox.js";
47
+ import { collectExternalCommsThread, externalCatchHealthy, externalInboxUrl } from "./comms-sandbox-catch.js";
53
48
  import { mapWithConcurrency } from "./concurrency.js";
54
- import { appendSandboxReceipt } from "./sandbox-receipts.js";
49
+ import { DEFAULT_DEVICE_PRESET, isDevicePresetName, resolveDevicePreset } from "./device-presets.js";
50
+ import {} from "./e2b-desktop-launch.js";
51
+ import {} from "./e2b-desktop-resources.js";
52
+ import {} from "./e2b-detached.js";
55
53
  import { assertScreenshotEvidence } from "./image-evidence.js";
54
+ import { MAX_CUA_LANES, cuaLaneValidationReason, isHttpUrl, isLoopbackUrl, outputTokenLimitValidationReason, subjectStateInvalidReason } from "./lab-config.js";
55
+ import { startAppServerSession } from "./local-agent-appserver.js";
56
+ import { startClaudeSession } from "./local-agent-claude-session.js";
57
+ import { createLocalAgentProvider, detectLocalAgents } from "./local-agent-cli.js";
56
58
  import { buildObserverData } from "./observer-data.js";
57
- import { corepackCommandFor, needsNodeRuntime, nodeBootstrapCommand } from "./subject-runtime.js";
58
- import { TERMINAL_NODE_BOOTSTRAP_COMMAND } from "./terminal-node-bootstrap.js";
59
- import { chromeCdpProbeCommand, parseChromeCdpProbeOutput } from "./chrome-cdp-probe.js";
60
- import { personaToDirectives, renderPersonaPromptSection } from "./persona.js";
61
- import { labPersonaIds, resolveCommittedPersonas } from "./persona-resolve.js";
62
- import { renderTaskPrompt } from "./tasks.js";
63
59
  import { attachObserverRuntimeStreamUrls, renderObserver } from "./observer.js";
64
- import { containsSensitive, digestText, redactedTail, redactText } from "./redaction.js";
60
+ import { DEFAULT_OPENAI_CU_MODEL } from "./openai-responses-cu.js";
65
61
  import { participantAssignment } from "./participant-assignment.js";
66
- import { actorEnding } from "./actor-stop-cause.js";
67
- import { assertPreparedSelectedOutputDirectory, assertSafeOutputPathSegment, prepareContainedOutputDirectory, prepareSelectedOutputDirectory, writeContainedOutputFile, writePreparedRunLatestPointer } from "./selected-output-paths.js";
62
+ import { labPersonaIds, resolveCommittedPersonas } from "./persona-resolve.js";
63
+ import { personaToDirectives, renderPersonaPromptSection } from "./persona.js";
64
+ import { MODEL_RATES, estimateActorCost, estimateAllocatedDesktopCost, estimateDesktopCost, round6 } from "./pricing.js";
65
+ import { containsSensitive, digestText, redactText } from "./redaction.js";
68
66
  import { prepareRunArtifactPaths, validatePreparedRunArtifactPaths } from "./run-paths.js";
67
+ import { beginRunStatus, withRunStatusScope } from "./run-status.js";
68
+ import { PUBLIC_TARGET_CWD, REVIEW_SCHEMA, RUN_BUNDLE_SCHEMA, aggregateTaskFunnels, buildRunSource, formatParticipantOutcomes, formatStudyTaskFunnel, loadRunBundle, tallyParticipantOutcomes, withCuaReviewProvenance } from "./run.js";
69
+ import { assertPreparedSelectedOutputDirectory, assertSafeOutputPathSegment, prepareContainedOutputDirectory, prepareSelectedOutputDirectory, writeContainedOutputFile, writePreparedRunLatestPointer } from "./selected-output-paths.js";
69
70
  import { createLocalTreeArchive } from "./source-archive.js";
70
- import { buildRunSource, loadRunBundle, PUBLIC_TARGET_CWD, REVIEW_SCHEMA, RUN_BUNDLE_SCHEMA, aggregateTaskFunnels, formatParticipantOutcomes, formatStudyTaskFunnel, tallyParticipantOutcomes, withCuaReviewProvenance } from "./run.js";
71
- import { estimateActorCost, estimateDesktopCost, estimateAllocatedDesktopCost, MODEL_RATES, round6 } from "./pricing.js";
72
- import { observeDesktopResources } from "./e2b-desktop-resources.js";
71
+ import { renderTaskPrompt } from "./tasks.js";
73
72
  export const CUA_ACTOR_LAB_SCHEMA = "humanish.cua-lab-result.v2";
74
73
  // The only fan-out topology this slice ships: N lanes = N independent E2B desktop sandboxes,
75
74
  // each its own world (clone/serve + subject.state per lane). Shared-world is layer 7 (#164).
@@ -77,10 +76,6 @@ export const CUA_FANOUT_STRATEGY = "per-lane-worlds";
77
76
  // Env override that may only LOWER the effective concurrency (never raise concurrent paid
78
77
  // desktops — invariant 3). Read names-only into a local; the value never persists.
79
78
  const CUA_MAX_CONCURRENCY_ENV = "HUMANISH_CUA_MAX_CONCURRENCY";
80
- export const CUA_ACTOR_LAB_PROVIDER_METADATA = {
81
- mode: "cua-actor-lab",
82
- tool: "humanish"
83
- };
84
79
  // The DEFAULT session budget, sized so a study can FINISH (docs/principles/three-roles.md: a
85
80
  // session ends because the participant is done, not because a timer fired — the time-box is a
86
81
  // session-level cap a researcher sets generously; spend protection is the dollar caps' job).
@@ -103,60 +98,6 @@ function defaultSessionTimeoutMs(config) {
103
98
  const room = MAX_SANDBOX_MS - SUBJECT_PROVISION_BUDGET_MS - stateBudgetMs - SANDBOX_TIMEOUT_BUFFER_MS;
104
99
  return Math.max(MIN_DERIVED_SESSION_TIMEOUT_MS, Math.min(DEFAULT_APP_URL_SESSION_TIMEOUT_MS, room));
105
100
  }
106
- // Settle after opening the browser, before the first screenshot — long enough for a cold
107
- // browser + page load to paint (2s captured a blank desktop; the render empirically needs ~6-9s).
108
- const BROWSER_SETTLE_MS = 8_000;
109
- /** Where a lane's synthetic camera feed lives inside the sandbox: a tmpfs the sandbox user can
110
- * write, and a path that contains neither /tmp/ nor /home/, which the public-safety scan reads
111
- * as an operator's local path (this one is the harness's own and belongs in the bundle). */
112
- export const SANDBOX_MEDIA_DIR = "/dev/shm/humanish-media";
113
- export const SANDBOX_CAMERA_PATH = `${SANDBOX_MEDIA_DIR}/camera.y4m`;
114
- /** The synthetic feed: ffmpeg's test pattern, 640x480 at 10 fps, six seconds (about 28 MB of
115
- * raw Y4M on the tmpfs), looped by Chrome's fake capture device. */
116
- export const SYNTHETIC_CAMERA_COMMAND = `mkdir -p ${SANDBOX_MEDIA_DIR} && ffmpeg -y -loglevel error -f lavfi -i testsrc=size=640x480:rate=10 -t 6 -pix_fmt yuv420p ${SANDBOX_CAMERA_PATH}`;
117
- /**
118
- * Put the declared camera feed in the sandbox and return the Chromium flags that present it as a
119
- * capture device (#509). Fails CLOSED: a feed that cannot be produced (no ffmpeg on the image, an
120
- * unreadable host file) is named before the browser launches, because a participant told it has
121
- * a camera and finds none reports the instrument's gap as the product's.
122
- */
123
- export async function prepareDesktopMedia(desktop, media, permission, cwd, requestTimeoutMs, readHostFile = (absolutePath) => readFile(absolutePath)) {
124
- if (media.microphone !== undefined) {
125
- throw new Error("execution.desktop.media.microphone.source injection is unsupported; the declared microphone file cannot be delivered, including on custom templates.");
126
- }
127
- const flags = [];
128
- let camera;
129
- if (media.camera !== undefined) {
130
- if (media.camera.source === "synthetic") {
131
- const made = await desktop.commands.run(SYNTHETIC_CAMERA_COMMAND, { requestTimeoutMs, timeoutMs: 60_000 });
132
- if (made.exitCode !== undefined && made.exitCode !== 0) {
133
- throw new Error(`the synthetic camera feed could not be generated on this desktop image (ffmpeg exited ${made.exitCode}: ${tailOf(made.stderr ?? made.stdout ?? "")}); give execution.desktop.media.camera.source a .y4m file instead`);
134
- }
135
- camera = { source: "synthetic", file: SANDBOX_CAMERA_PATH };
136
- }
137
- else {
138
- const absolutePath = path.resolve(cwd, media.camera.source);
139
- let bytes;
140
- try {
141
- bytes = await readHostFile(absolutePath);
142
- }
143
- catch (error) {
144
- throw new Error(`execution.desktop.media.camera.source could not be read (${toErrorMessage(error)})`);
145
- }
146
- if (bytes.length > 64 * 1024 * 1024) {
147
- throw new Error(`execution.desktop.media.camera.source is ${bytes.length} bytes; the camera feed is capped at 64 MiB`);
148
- }
149
- await desktop.commands.run(`mkdir -p ${SANDBOX_MEDIA_DIR}`, { requestTimeoutMs, timeoutMs: 15_000 });
150
- const payload = bytes.buffer.slice(bytes.byteOffset, bytes.byteOffset + bytes.byteLength);
151
- await desktop.files.write(SANDBOX_CAMERA_PATH, payload, { requestTimeoutMs, useOctetStream: true });
152
- camera = { source: "file", file: SANDBOX_CAMERA_PATH };
153
- }
154
- flags.push("--use-fake-device-for-media-stream", `--use-file-for-fake-video-capture=${SANDBOX_CAMERA_PATH}`);
155
- }
156
- if (permission === "granted")
157
- flags.push("--use-fake-ui-for-media-stream");
158
- return { ...(camera === undefined ? {} : { camera }), permission, flags };
159
- }
160
101
  // Device/screen size comes from the named-preset registry (device-presets.ts), selectable per run
161
102
  // via execution.desktop.device (default `desktop`=1440x950). NOTE: this is run-wide for now; a
162
103
  // per-PERSONA device dimension (N personas × devices, as the bespoke sims author) lands with
@@ -172,19 +113,6 @@ const SUBJECT_PROVISION_BUDGET_MS = 30 * 60_000;
172
113
  * The derived per-lane deadline has to stay under it, and saying so at plan time beats discovering
173
114
  * it from a raw provider 400 after a plan has already printed. */
174
115
  const MAX_SANDBOX_MS = 60 * 60_000;
175
- export const SUBJECT_DIR = "/home/user/subject";
176
- // Remote path for the once-per-run packed local-tree archive; removed by the extract step
177
- // after it unpacks into SUBJECT_DIR.
178
- const LOCAL_TREE_REMOTE_ARCHIVE_PATH = "/home/user/.humanish-source.tar.gz";
179
- const CLONE_TIMEOUT_MS = 5 * 60_000;
180
- const INSTALL_TIMEOUT_MS = 10 * 60_000;
181
- const BUILD_TIMEOUT_MS = 10 * 60_000;
182
- const DEFAULT_READY_TIMEOUT_MS = 180_000;
183
- // Per-step budget for subject.state seed steps; each step's declared (or default) budget is
184
- // also summed into the default sandbox deadline so seeding never eats the session's room.
185
- const DEFAULT_STATE_STEP_TIMEOUT_MS = 5 * 60_000;
186
- // How much of a failing step's log tail rides the (redacted) error message.
187
- const ERROR_TAIL_CHARS = 2000;
188
116
  const DEFAULT_MISSION = "You are testing a web application. The browser is already open at the subject URL. Explore it, accomplish what the scenario asks, and stop when done.";
189
117
  /**
190
118
  * The participant's outcome as ONE fixed first line of its last message (#570, second half). The
@@ -262,21 +190,6 @@ export function withInboxMission(spec, inboxUrl, address, receiving = false) {
262
190
  instructions: `${spec.instructions}\n\nEmail inbox:${identity} When the app tells you it has emailed you (a verification link, confirmation code, or magic link), open ${recipientInboxUrl(inboxUrl, address)} in the browser to read that email and follow its link or enter its code. All email the app sends you arrives there. Waiting for an email is normal, not a blocker — do not end your session while waiting; open the inbox and refresh it until the email appears.`
263
191
  };
264
192
  }
265
- /** The lane's addressed comms recipient, when one exists — the gate AND the address source for the
266
- * inbox instruction (#351). A lane told to check an inbox it can never receive into would stall,
267
- * so no addressed recipient means no instruction. */
268
- export function inboxRecipientFor(commsEmail, laneId) {
269
- return (commsEmail.recipients ?? []).find((recipient) => recipient.lane === laneId && recipient.address !== undefined);
270
- }
271
- /** True when a lane has a declared comms recipient WITH an address, so the drain can actually match the
272
- * mail the persona will be told to read. Gates the inbox instruction to lanes that can receive mail —
273
- * a lane told to check an inbox it can never receive into would just stall. */
274
- export function laneHasInboxRecipient(commsEmail, laneId) {
275
- return inboxRecipientFor(commsEmail, laneId) !== undefined;
276
- }
277
- /** Mid-run inbox-surface render cadence (ms). Coarse enough that the per-tick `cat` + file writes stay
278
- * cheap; fine enough that a verification email is visible seconds after the app sends it. */
279
- const INBOX_SURFACE_CADENCE_MS = 2500;
280
193
  /**
281
194
  * The narrowest browser WINDOW Chrome/Chromium will render on the E2B desktop. Chrome refuses to
282
195
  * make its window narrower than this (~500 CSS px observed: a 414-wide X screen produced a 500-wide
@@ -292,20 +205,6 @@ export const MIN_DESKTOP_RENDER_WIDTH = 500;
292
205
  export function floorRenderResolution(resolution) {
293
206
  return [Math.max(resolution[0], MIN_DESKTOP_RENDER_WIDTH), resolution[1]];
294
207
  }
295
- /**
296
- * The DECLARED preset to record alongside the rendered screen, or undefined when the preset
297
- * rendered faithfully.
298
- *
299
- * `desktopGeometry.screen.verified` compares the FLOORED number with itself, so on its own a
300
- * floored run is indistinguishable from a faithful one: a reader sees requested 500 / verified 500
301
- * and concludes a 500-wide screen was asked for. Recording the declared preset is what makes
302
- * "the preset width did not render" legible in the bundle.
303
- */
304
- export function declaredScreenForRender(preset, presetName, rendered) {
305
- if (preset.width === rendered[0] && preset.height === rendered[1])
306
- return undefined;
307
- return { width: preset.width, height: preset.height, preset: presetName };
308
- }
309
208
  /**
310
209
  * Resolve a lane's device + rendered resolution (most-specific wins, exactly as the single-lane
311
210
  * path always has): a raw execution.desktop.resolution escape hatch (only legal when no lane
@@ -582,65 +481,6 @@ function formatLanePlanEntry(lane) {
582
481
  ].filter((part) => part !== undefined);
583
482
  return `${lane.id}: persona=${lane.persona}${taxonomy.length > 0 ? ` ${taxonomy.join(" ")}` : ""} device=${lane.device} ${lane.resolution[0]}x${lane.resolution[1]} prompt#${lane.instructionDigest}${lane.targetDigest ? ` target#${lane.targetDigest}` : ""}`;
584
483
  }
585
- /** ISO timestamp from an injectable clock (tests freeze `now` for deterministic durationMs). */
586
- function isoNow(now) {
587
- return new Date(now()).toISOString();
588
- }
589
- /** Emit a phase-started event (no ok/durationMs: those belong to the matching completed event). */
590
- /**
591
- * Run a provisioning step and, when it fails with an EXIT CODE, run it once more (#602). A cold
592
- * install of 0.74.0 lost its whole first live study to one transient TLS error inside the
593
- * sandbox's `npm install`; the parallel install twenty seconds later passed, as had the ten
594
- * before it. One retry clears that class. A TIMEOUT is not retried: its budget is already spent,
595
- * and a second wait would double it. The retry runs under its own step name so both logs stay.
596
- */
597
- async function runProvisioningStepWithOneRetry(desktop, args) {
598
- const first = await runDetachedStep(desktop, {
599
- name: args.name,
600
- command: args.command,
601
- cwd: args.cwd,
602
- timeoutMs: args.timeoutMs,
603
- requestTimeoutMs: args.requestTimeoutMs,
604
- ...args.timers
605
- });
606
- if (first.ok || first.timedOut)
607
- return { ...first, attempts: 1 };
608
- const retryStartedAt = args.now();
609
- emitPhaseStarted(args.onPhase, args.now, args.retryPhase, `${args.retryMessage} (first attempt exited ${first.exitCode ?? "null"}; retrying once)`);
610
- const second = await runDetachedStep(desktop, {
611
- name: `${args.name}-retry`,
612
- command: args.command,
613
- cwd: args.cwd,
614
- timeoutMs: args.timeoutMs,
615
- requestTimeoutMs: args.requestTimeoutMs,
616
- ...args.timers
617
- });
618
- emitPhaseCompleted(args.onPhase, args.now, retryStartedAt, args.retryPhase, second.ok, second.ok ? `${args.retryMessage}: succeeded on the second attempt` : `${args.retryMessage}: failed twice`);
619
- return { ...second, attempts: 2, ...(first.exitCode === undefined ? {} : { firstExitCode: first.exitCode }) };
620
- }
621
- function emitPhaseStarted(onPhase, now, phase, message) {
622
- onPhase?.({ at: isoNow(now), type: `cua-lab.subject.${phase}.started`, message });
623
- }
624
- /** Emit the matching phase-completed event: always carries ok and durationMs (>= 0). */
625
- function emitPhaseCompleted(onPhase, now, startedAt, phase, ok, message) {
626
- onPhase?.({
627
- at: isoNow(now),
628
- type: `cua-lab.subject.${phase}.completed`,
629
- ok,
630
- durationMs: Math.max(0, now() - startedAt),
631
- message
632
- });
633
- }
634
- /** Default phase-boundary sink (stderr): one line per event, prefixed with the lane id ONLY
635
- * when laneCount > 1. Single-lane emission is unconditional: total single-lane silence for the
636
- * whole clone/install/build/ready boot is the bug this event stream exists to close.
637
- * Overridable via CuaActorLabHooks.onPhase so deterministic tests capture instead of writing to
638
- * the real stderr. */
639
- function defaultSubjectPhaseSink(event, ctx) {
640
- const durationSuffix = event.durationMs === undefined ? "" : ` (${event.durationMs}ms)`;
641
- const prefix = ctx.laneCount > 1 ? `humanish cua [${ctx.laneId}]` : "humanish cua";
642
- process.stderr.write(`${prefix}: ${event.message}${durationSuffix}\n`);
643
- }
644
484
  /** Short id-safe suffix for a subject-phase RunEvent: drops the shared prefix/suffix so each
645
485
  * phase gets a distinct bundle event id (e.g. "clone", "state-before-build"). */
646
486
  function phaseEventIdSuffix(type) {
@@ -681,36 +521,6 @@ export function makeLaneWriteScreenshot(artifactRoot, spec, screenshots) {
681
521
  return rel;
682
522
  };
683
523
  }
684
- /**
685
- * Verify the desktop screen geometry IN-SANDBOX (the per-lane device claim is checked, never
686
- * assumed). A parseable mismatch fails closed. Unavailable/unparseable evidence is returned as
687
- * an explicit warning: the lane may still run, but its bundle records only the requested screen
688
- * and never upgrades that request into a verified measurement.
689
- */
690
- export async function inspectDesktopScreenGeometry(args) {
691
- let out = "";
692
- try {
693
- const result = await args.desktop.commands.run("xdpyinfo 2>/dev/null | grep -i dimensions || true", { requestTimeoutMs: args.requestTimeoutMs });
694
- out = (result.stdout ?? "").trim();
695
- }
696
- catch {
697
- return { warning: `Desktop screen geometry could not be measured for lane ${args.laneId}; requested geometry remains unverified.` };
698
- }
699
- const match = out.match(/(\d+)\s*x\s*(\d+)\s*pixels/i);
700
- if (!match) {
701
- return { warning: `Desktop screen geometry could not be parsed for lane ${args.laneId}; requested geometry remains unverified.` };
702
- }
703
- const width = Number(match[1]);
704
- const height = Number(match[2]);
705
- const [expectedWidth, expectedHeight] = args.requestedScreen;
706
- if (width === expectedWidth && height === expectedHeight) {
707
- return { verified: { width, height, source: "xdpyinfo" } };
708
- }
709
- return {
710
- verified: { width, height, source: "xdpyinfo" },
711
- error: `HUMANISH_CUA_LAB_DEVICE_GEOMETRY: lane ${args.laneId} requested a ${expectedWidth}x${expectedHeight} desktop but xdpyinfo reports ${width}x${height} in-sandbox; the per-lane device geometry is unverified (fail-closed).`
712
- };
713
- }
714
524
  /** A blocked lane outcome (pipeline gate / fail-fast skipped it before it ran). */
715
525
  function blockedLaneOutcome(spec, reason) {
716
526
  return {
@@ -728,709 +538,6 @@ function blockedLaneOutcome(spec, reason) {
728
538
  harnessError: false
729
539
  };
730
540
  }
731
- async function findVisibleBrowserWindowId(desktop, requestTimeoutMs, browserFamily, launchIdentity) {
732
- if (browserFamily === "unknown")
733
- return undefined;
734
- // The candidate loop keeps the LAST identity match: with a launch identity the match is
735
- // unique anyway, and without one every family candidate matches, so the newest visible
736
- // window of the launched family wins (the window this lane just opened).
737
- const finder = browserFamily === "firefox"
738
- ? [
739
- "find_firefox_window() {",
740
- " timeout 2s xdotool search --onlyvisible --class 'firefox|Firefox' 2>/dev/null || true",
741
- "}",
742
- "window_id=",
743
- "for _ in $(seq 1 10); do",
744
- " for candidate in $(find_firefox_window); do",
745
- " window_pid=\"$(xdotool getwindowpid \"$candidate\" 2>/dev/null || true)\"",
746
- " if matches_launch_identity \"$window_pid\"; then window_id=\"$candidate\"; fi",
747
- " done",
748
- " if [ -n \"$window_id\" ]; then break; fi",
749
- " sleep 0.5",
750
- "done"
751
- ]
752
- : [
753
- "find_chrome_window() {",
754
- " timeout 2s xdotool search --onlyvisible --class 'google-chrome|Google-chrome|chromium|Chromium|chrome|Chrome' 2>/dev/null || true",
755
- "}",
756
- "window_id=",
757
- "for _ in $(seq 1 10); do",
758
- " for candidate in $(find_chrome_window); do",
759
- " window_pid=\"$(xdotool getwindowpid \"$candidate\" 2>/dev/null || true)\"",
760
- " if matches_launch_identity \"$window_pid\"; then window_id=\"$candidate\"; fi",
761
- " done",
762
- " if [ -n \"$window_id\" ]; then break; fi",
763
- " sleep 0.5",
764
- "done"
765
- ];
766
- const result = await desktop.commands.run([
767
- "set -euo pipefail",
768
- "export DISPLAY=\"${DISPLAY:-:0}\"",
769
- `launch_pid=${shellSingleQuote(launchIdentity?.processId ?? "")}`,
770
- `profile_dir=${shellSingleQuote(launchIdentity?.profileDir ?? "")}`,
771
- "matches_launch_identity() {",
772
- " if [ -z \"$launch_pid\" ] && [ -z \"$profile_dir\" ]; then return 0; fi",
773
- " local current=\"${1:-}\"",
774
- " while [[ \"$current\" =~ ^[0-9]+$ ]] && [ \"$current\" -gt 1 ]; do",
775
- " cmdline=\"$(tr '\\0' ' ' < \"/proc/$current/cmdline\" 2>/dev/null || true)\"",
776
- " if [ -n \"$profile_dir\" ] && [[ \"$cmdline\" == *\"$profile_dir\"* ]]; then return 0; fi",
777
- " if [ \"$current\" = \"$launch_pid\" ]; then return 0; fi",
778
- " current=\"$(ps -o ppid= -p \"$current\" 2>/dev/null | tr -d ' ' || true)\"",
779
- " done",
780
- " return 1",
781
- "}",
782
- ...finder,
783
- "if [ -n \"$window_id\" ]; then printf 'WINDOW_ID=%s\\n' \"$window_id\"; fi"
784
- ].join("\n"), {
785
- requestTimeoutMs,
786
- timeoutMs: 15_000
787
- });
788
- return (result.stdout ?? "").match(/^WINDOW_ID=(\S+)$/m)?.[1];
789
- }
790
- /**
791
- * Build the xdotool command that makes a browser window fill the desktop.
792
- * Exported (pure) for contract tests. A window manager can ignore Chrome's
793
- * --window-size, so xdotool is the robust path: move the window to the origin,
794
- * then size it to the exact desktop resolution so Observer screenshots carry no
795
- * dead margin around the browser.
796
- */
797
- export function buildFillDesktopWindowCommand(windowId, width, height) {
798
- return [
799
- "set -euo pipefail",
800
- `win=${shellSingleQuote(windowId)}`,
801
- `xdotool windowactivate "$win" >/dev/null 2>&1 || true`,
802
- `xdotool windowmove "$win" 0 0 >/dev/null 2>&1 || true`,
803
- `xdotool windowsize "$win" ${width} ${height} >/dev/null 2>&1 || true`,
804
- ].join("\n");
805
- }
806
- /**
807
- * Best-effort initial fill. A contained smaller window remains usable; the capture
808
- * below checks for clipping and refuses an uncorrectable window before the actor runs.
809
- */
810
- async function fillDesktopBrowserWindow(desktop, windowId, resolution, requestTimeoutMs) {
811
- const [width, height] = resolution;
812
- await desktop.commands
813
- .run(buildFillDesktopWindowCommand(windowId, width, height), {
814
- requestTimeoutMs,
815
- timeoutMs: 10_000,
816
- })
817
- .catch(() => undefined);
818
- }
819
- async function openDesktopBrowserTarget(desktop, targetUrl, requestTimeoutMs, browserPreference,
820
- /** Launch-time flags that make mobile fidelity (#221) hold across every tab: the user agent and
821
- * touch events are browser-wide here, where the CDP holder covers only the launch page. */
822
- extraChromiumFlags = []) {
823
- const requestedBrowser = browserPreference ?? "default";
824
- if (isHttpUrl(targetUrl)) {
825
- const chromiumFlags = [...CHROMIUM_EVIDENCE_HYGIENE_FLAGS, ...extraChromiumFlags].map(shellSingleQuote).join(" ");
826
- const browserLaunchCommand = [
827
- "set -euo pipefail",
828
- `target_url=${shellSingleQuote(targetUrl)}`,
829
- `browser_preference=${shellSingleQuote(requestedBrowser)}`,
830
- "chrome_profile_dir=",
831
- `chrome_preferences_json=${shellSingleQuote(chromiumEvidenceProfilePreferencesJson())}`,
832
- "prepare_chrome_profile() {",
833
- " chrome_profile_dir=\"$(mktemp -d /tmp/humanish-chrome-profile.XXXXXX)\"",
834
- " mkdir -p \"$chrome_profile_dir/Default\"",
835
- " printf '%s\\n' \"$chrome_preferences_json\" > \"$chrome_profile_dir/Default/Preferences\"",
836
- "}",
837
- "launch_browser() {",
838
- " local label=\"$1\"",
839
- " local binary=\"$2\"",
840
- " shift 2",
841
- " if command -v \"$binary\" >/dev/null 2>&1; then",
842
- " nohup \"$binary\" \"$@\" \"$target_url\" >/tmp/humanish-browser-open.log 2>&1 &",
843
- " local launch_pid=$!",
844
- " printf 'HUMANISH_BROWSER_RESOLVED=%s\\n' \"$label\"",
845
- " printf 'HUMANISH_BROWSER_PID=%s\\n' \"$launch_pid\"",
846
- " printf 'HUMANISH_BROWSER_PROFILE_DIR=%s\\n' \"$chrome_profile_dir\"",
847
- " if [[ \"$label\" =~ ^(google-chrome|google-chrome-stable|chromium|chromium-browser)$ ]]; then",
848
- " for _ in $(seq 1 30); do",
849
- " if [ -s \"$chrome_profile_dir/DevToolsActivePort\" ]; then",
850
- " head -n 1 \"$chrome_profile_dir/DevToolsActivePort\" | sed 's/^/HUMANISH_BROWSER_CDP_PORT=/'",
851
- " break",
852
- " fi",
853
- " sleep 0.1",
854
- " done",
855
- " fi",
856
- " return 0",
857
- " fi",
858
- " return 1",
859
- "}",
860
- // Fixed CDP port (not :0/random): each seat has its OWN desktop sandbox, so a known port
861
- // cannot conflict, and it makes the observer's port resolution deterministic. With :0 the
862
- // real port lives only in DevToolsActivePort; when the launch-time capture misses on a cold
863
- // start the observer falls back to 9222 and — being wrong — every CDP read fails for the
864
- // whole run (the lobby-code handoff then never sees the host's /lobby URL). 9222 is already
865
- // the fallback, so making it the actual port aligns launch, capture, and fallback.
866
- `chrome_debug_flags=(--remote-debugging-address=127.0.0.1 --remote-debugging-port=9222 ${chromiumFlags})`,
867
- "open_target() {",
868
- " case \"$browser_preference\" in",
869
- " chrome)",
870
- " prepare_chrome_profile",
871
- " launch_browser google-chrome google-chrome --new-window \"--user-data-dir=$chrome_profile_dir\" \"${chrome_debug_flags[@]}\" && return 0",
872
- " launch_browser google-chrome-stable google-chrome-stable --new-window \"--user-data-dir=$chrome_profile_dir\" \"${chrome_debug_flags[@]}\" && return 0",
873
- " echo 'requested browser chrome was not found' >&2",
874
- " return 127",
875
- " ;;",
876
- " chromium)",
877
- " prepare_chrome_profile",
878
- " launch_browser chromium chromium --new-window \"--user-data-dir=$chrome_profile_dir\" \"${chrome_debug_flags[@]}\" && return 0",
879
- " launch_browser chromium-browser chromium-browser --new-window \"--user-data-dir=$chrome_profile_dir\" \"${chrome_debug_flags[@]}\" && return 0",
880
- " echo 'requested browser chromium was not found' >&2",
881
- " return 127",
882
- " ;;",
883
- " firefox)",
884
- " prepare_chrome_profile",
885
- " launch_browser firefox firefox --new-instance --no-remote --new-window --profile \"$chrome_profile_dir\" && return 0",
886
- " echo 'requested browser firefox was not found' >&2",
887
- " return 127",
888
- " ;;",
889
- " default)",
890
- " prepare_chrome_profile",
891
- " launch_browser google-chrome google-chrome --new-window \"--user-data-dir=$chrome_profile_dir\" \"${chrome_debug_flags[@]}\" && return 0",
892
- " launch_browser google-chrome-stable google-chrome-stable --new-window \"--user-data-dir=$chrome_profile_dir\" \"${chrome_debug_flags[@]}\" && return 0",
893
- " launch_browser chromium chromium --new-window \"--user-data-dir=$chrome_profile_dir\" \"${chrome_debug_flags[@]}\" && return 0",
894
- " launch_browser chromium-browser chromium-browser --new-window \"--user-data-dir=$chrome_profile_dir\" \"${chrome_debug_flags[@]}\" && return 0",
895
- " launch_browser firefox firefox --new-instance --no-remote --new-window --profile \"$chrome_profile_dir\" && return 0",
896
- " launch_browser xdg-open xdg-open && return 0",
897
- " echo 'no browser opener found' >&2",
898
- " return 127",
899
- " ;;",
900
- " esac",
901
- "}",
902
- "open_target"
903
- ].join("\n");
904
- const result = await runDesktopCommandOrThrow(() => desktop.commands.run(browserLaunchCommand, {
905
- requestTimeoutMs,
906
- timeoutMs: 15_000,
907
- }), ({ exitCode, stderrTail }) => new Error(`browser launch failed${exitCode === undefined ? "" : ` with exit ${exitCode}`}: ${stderrTail}`));
908
- if (result.exitCode !== undefined && result.exitCode !== 0) {
909
- throw new Error(`browser launch failed with exit ${result.exitCode}: ${tailOf(result.stderr ?? result.stdout ?? "")}`);
910
- }
911
- const resolved = (result.stdout ?? "").match(/^HUMANISH_BROWSER_RESOLVED=(\S+)$/m)?.[1];
912
- const processId = (result.stdout ?? "").match(/^HUMANISH_BROWSER_PID=(\d+)$/m)?.[1];
913
- const profileDir = (result.stdout ?? "").match(/^HUMANISH_BROWSER_PROFILE_DIR=(\S+)$/m)?.[1];
914
- const cdpPortRaw = (result.stdout ?? "").match(/^HUMANISH_BROWSER_CDP_PORT=(\d+)$/m)?.[1];
915
- const cdpPort = cdpPortRaw === undefined ? undefined : Number(cdpPortRaw);
916
- return {
917
- family: desktopBrowserFamily(resolved ?? requestedBrowser),
918
- ...(processId === undefined || profileDir === undefined
919
- ? {}
920
- : { identity: { processId, profileDir, targetUrl, ...(cdpPort === undefined ? {} : { cdpPort }) } }),
921
- ...(browserPreference === undefined
922
- ? {}
923
- : { evidence: { requested: requestedBrowser, ...(resolved === undefined ? {} : { resolved }) } })
924
- };
925
- }
926
- if (browserPreference === undefined || browserPreference === "default") {
927
- if (desktop.open) {
928
- await desktop.open(targetUrl);
929
- }
930
- else {
931
- await desktop.launch("google-chrome", targetUrl);
932
- }
933
- return {
934
- family: desktop.open ? "unknown" : "chromium",
935
- ...(browserPreference === undefined ? {} : { evidence: { requested: requestedBrowser } })
936
- };
937
- }
938
- const launchTarget = requestedBrowser === "chrome" ? "google-chrome"
939
- : requestedBrowser === "chromium" ? "chromium"
940
- : requestedBrowser === "firefox" ? "firefox"
941
- : "google-chrome";
942
- await desktop.launch(launchTarget, targetUrl);
943
- return {
944
- family: desktopBrowserFamily(launchTarget),
945
- evidence: { requested: requestedBrowser, resolved: launchTarget }
946
- };
947
- }
948
- export function desktopBrowserFamily(value) {
949
- if (value === "firefox")
950
- return "firefox";
951
- if (value === "chrome" || value === "chromium" || value === "google-chrome" || value === "google-chrome-stable" || value === "chromium-browser") {
952
- return "chromium";
953
- }
954
- return "unknown";
955
- }
956
- /**
957
- * The URL / title / page-text / scroll observer behind stopWhen and task criteria. One probe per
958
- * observation, run on the sandbox's python3 (see chrome-cdp-probe.ts for why not node: #514).
959
- *
960
- * "active": follow the participant to whatever tab they are driving now — never pin the state
961
- * observer to the launch tab (a verification link that opened in a NEW tab left a pinned observer
962
- * reading the old tab forever).
963
- *
964
- * `onUnavailable` fires ONCE, on the first probe that could not read the page, with the reason.
965
- * The observer still degrades to `{}` for the loop; the callback is how a lane says out loud that
966
- * url/text criteria are not being measured, instead of letting the funnel report 0/N (#514).
967
- */
968
- export function makeChromeBrowserStateObserver(desktop, requestTimeoutMs, endpoint, targetId, onUnavailable,
969
- /**
970
- * Mobile emulation on later tabs (#623): the holder attaches to every page target Chrome opens
971
- * after the launch page, so a tab the participant opens later should lay out at the phone width
972
- * too. The first observation on each new target reads that page's OWN report; a target that
973
- * reports the requested width is recorded through `onCovered`, and one that does not (or cannot
974
- * be read) fires `onDrift` once, so a phone-labelled lane that spent part of its session at
975
- * desktop layout says so with the number the page gave.
976
- */
977
- drift) {
978
- let reported = false;
979
- let drifted = false;
980
- const checkedTargets = new Set(drift === undefined ? [] : [drift.emulatedTargetId]);
981
- const unavailable = (reason) => {
982
- if (!reported) {
983
- reported = true;
984
- onUnavailable?.(reason);
985
- }
986
- return {};
987
- };
988
- const checkLaterTarget = async (newTargetId) => {
989
- if (drift === undefined || checkedTargets.has(newTargetId))
990
- return;
991
- checkedTargets.add(newTargetId);
992
- const read = await desktop.commands.run(chromeCdpProbeCommand({ ...endpoint, targetId: newTargetId, prefer: "pinned", mode: "fidelity" }), { requestTimeoutMs, timeoutMs: 5_000 });
993
- const fidelity = read.exitCode !== undefined && read.exitCode !== 0 ? undefined : parseChromeCdpProbeOutput(read.stdout).fidelity;
994
- if (fidelity !== undefined && fidelity.innerWidth === drift.expectedWidth) {
995
- drift.onCovered?.(newTargetId, { innerWidth: fidelity.innerWidth, devicePixelRatio: fidelity.devicePixelRatio, maxTouchPoints: fidelity.maxTouchPoints });
996
- if (drift.expectTouch === true && fidelity.maxTouchPoints === 0 && !drifted) {
997
- // The viewport followed; touch did not (yet): the holder reloads a later tab once after its
998
- // first navigation commits, and this observation may have landed before that reload.
999
- drifted = true;
1000
- drift.onDrift(`a later page target reports the ${fidelity.innerWidth} px viewport but navigator.maxTouchPoints 0 on its first observation; touch reaches a document only when it loads under the override`);
1001
- }
1002
- return;
1003
- }
1004
- if (drifted)
1005
- return;
1006
- drifted = true;
1007
- drift.onDrift(fidelity === undefined
1008
- ? "the participant drove a page target other than the emulated launch tab and that page's own read-back could not be taken; whether it laid out at the phone width is not known"
1009
- : `the participant drove a page target other than the emulated launch tab and that page reports a ${fidelity.innerWidth} px viewport where ${drift.expectedWidth} px was requested (DPR ${fidelity.devicePixelRatio}); the mobile user agent and touch events are browser-wide, the viewport override was not re-applied to it`);
1010
- };
1011
- return async () => {
1012
- const result = await desktop.commands.run(chromeCdpProbeCommand({ ...endpoint, ...(targetId === undefined ? {} : { targetId }), prefer: "active", mode: "state" }), { requestTimeoutMs, timeoutMs: 5_000 });
1013
- if (result.exitCode !== undefined && result.exitCode !== 0) {
1014
- return unavailable(`probe exited ${result.exitCode}: ${tailOf(result.stderr ?? result.stdout ?? "")}`);
1015
- }
1016
- const parsed = parseChromeCdpProbeOutput(result.stdout);
1017
- if (parsed.unavailable !== undefined)
1018
- return unavailable(parsed.unavailable);
1019
- if (parsed.targetId !== undefined)
1020
- await checkLaterTarget(parsed.targetId);
1021
- return {
1022
- ...(parsed.url === undefined ? {} : { url: parsed.url }),
1023
- ...(parsed.title === undefined ? {} : { title: parsed.title }),
1024
- ...(parsed.text === undefined ? {} : { text: parsed.text }),
1025
- ...(parsed.scrollY === undefined ? {} : { scrollY: parsed.scrollY })
1026
- };
1027
- };
1028
- }
1029
- /**
1030
- * Read the running browser's actual outer-window bounds and CSS layout viewport through the
1031
- * already-enabled local Chrome DevTools endpoint. The returned values come from `window.*` in
1032
- * the target page; requested E2B resolution is deliberately not an input to this function.
1033
- * Missing channels report their reason via `onUnavailable`, so the geometry warning can name
1034
- * the cause (a dead CDP endpoint, no python3) instead of only the symptom. Returns `undefined`
1035
- * only when neither channel could be measured.
1036
- * Outer bounds and CSS dimensions are independent channels: a background page can report zero
1037
- * outer dimensions while still reporting a CSS viewport. Final captures follow the active tab;
1038
- * launch captures and emulation attribution keep the pinned target.
1039
- */
1040
- export function makeChromeDesktopGeometryObserver(desktop, requestTimeoutMs, endpoint, targetId, onUnavailable, prefer = "pinned") {
1041
- return async () => {
1042
- const result = await desktop.commands.run(chromeCdpProbeCommand({ ...endpoint, ...(targetId === undefined ? {} : { targetId }), prefer, mode: "geometry" }), { requestTimeoutMs, timeoutMs: 5_000 });
1043
- if (result.exitCode !== undefined && result.exitCode !== 0) {
1044
- onUnavailable?.(`probe exited ${result.exitCode}: ${tailOf(result.stderr ?? result.stdout ?? "")}`);
1045
- return undefined;
1046
- }
1047
- const parsed = parseChromeCdpProbeOutput(result.stdout);
1048
- if (parsed.unavailable !== undefined) {
1049
- onUnavailable?.(parsed.unavailable);
1050
- return undefined;
1051
- }
1052
- const browserWindow = isMeasuredRect(parsed.browserWindow) ? { ...parsed.browserWindow, source: "cdp" } : undefined;
1053
- const viewport = isMeasuredViewport(parsed.viewport) ? { ...parsed.viewport, source: "cdp" } : undefined;
1054
- if (browserWindow === undefined && viewport === undefined) {
1055
- onUnavailable?.("the page reported no usable window or viewport dimensions");
1056
- return undefined;
1057
- }
1058
- if (browserWindow === undefined)
1059
- onUnavailable?.("the page reported no usable outer-window dimensions");
1060
- if (viewport === undefined)
1061
- onUnavailable?.("the page reported no usable CSS viewport dimensions");
1062
- return {
1063
- ...(browserWindow === undefined ? {} : { browserWindow }),
1064
- ...(viewport === undefined ? {} : { viewport }),
1065
- ...(parsed.targetId === undefined ? {} : { targetId: parsed.targetId })
1066
- };
1067
- };
1068
- }
1069
- /** The user agent a mobile-emulated lane presents unless the lab sets its own. */
1070
- export const DEFAULT_MOBILE_USER_AGENT = "Mozilla/5.0 (iPhone; CPU iPhone OS 17_0 like Mac OS X) AppleWebKit/605.1.15 (KHTML, like Gecko) Version/17.0 Mobile/15E148 Safari/604.1";
1071
- /**
1072
- * Apply mobile emulation (#221) to the lane's launch page and read back what the page reports.
1073
- * Fails CLOSED: a request that cannot be applied throws, because a desktop run labelled mobile is
1074
- * the over-trust this feature exists to prevent. A read-back that cannot be taken is a warning
1075
- * (the emulation was applied; only the proof is missing).
1076
- */
1077
- export async function applyMobileEmulation(desktop, requestTimeoutMs, endpoint, targetId, request) {
1078
- const command = (mode) => chromeCdpProbeCommand({ ...endpoint, ...(targetId === undefined ? {} : { targetId }), prefer: "pinned", mode, emulation: request });
1079
- const read = async () => {
1080
- const result = await desktop.commands.run(command("fidelity"), { requestTimeoutMs, timeoutMs: 15_000 });
1081
- if (result.exitCode !== undefined && result.exitCode !== 0) {
1082
- return { unavailable: `probe exited ${result.exitCode}: ${tailOf(result.stderr ?? result.stdout ?? "")}` };
1083
- }
1084
- return parseChromeCdpProbeOutput(result.stdout);
1085
- };
1086
- // The UA / touch / DPR overrides are bound to the DevTools session that set them and lapse the
1087
- // moment its socket closes (measured: only the viewport width survived a one-shot apply). So the
1088
- // applier stays attached for the lane's whole life as a detached process; the sandbox teardown
1089
- // ends it. Its first stdout line says what was applied.
1090
- const holderName = `mobile-emulation-${Date.now().toString(36)}`;
1091
- await startDetachedProcess(desktop, { name: holderName, command: command("hold"), requestTimeoutMs });
1092
- let announced;
1093
- for (let attempt = 0; attempt < 30 && announced === undefined; attempt += 1) {
1094
- await new Promise((resolve) => setTimeout(resolve, 500));
1095
- const log = await readDetachedLog(desktop, holderName, requestTimeoutMs).catch(() => "");
1096
- const line = log.split("\n").find((candidate) => candidate.trim().startsWith("{"));
1097
- if (line !== undefined)
1098
- announced = parseChromeCdpProbeOutput(line);
1099
- }
1100
- if (announced === undefined) {
1101
- throw new Error("mobile emulation could not be applied: the in-sandbox applier printed nothing within 15 s");
1102
- }
1103
- if (announced.unavailable !== undefined) {
1104
- throw new Error(`mobile emulation could not be applied (${announced.unavailable}); applied before failing: ${(announced.applied ?? []).join(", ") || "nothing"}`);
1105
- }
1106
- const applied = announced;
1107
- // Viewport/touch read-back proves context settings, not gesture equivalence. Two hosted
1108
- // replicas and a native-X conversion-toggle control reproduced reset click counts (#676).
1109
- const warnings = request.touch
1110
- ? ["Mobile emulation uses desktop pointer-to-touch conversion, which can differ for repeated taps. Confirm gesture failures with direct or native touch input before attributing them to the app."]
1111
- : [];
1112
- // The reload inside the applier takes a moment; the read-back is retried until the page reports
1113
- // the requested viewport and user agent, so a slow page does not read as "no proof".
1114
- let readBack = await read();
1115
- for (let attempt = 0; attempt < 20 && (readBack.fidelity === undefined || readBack.fidelity.innerWidth !== request.width || !readBack.fidelity.userAgent.includes(request.userAgent.slice(0, 24))); attempt += 1) {
1116
- await new Promise((resolve) => setTimeout(resolve, 500));
1117
- readBack = await read();
1118
- }
1119
- const fidelityRead = readBack;
1120
- const requested = {
1121
- width: request.width,
1122
- height: request.height,
1123
- deviceScaleFactor: request.deviceScaleFactor,
1124
- touch: request.touch,
1125
- userAgent: request.userAgent
1126
- };
1127
- const emulatedTargetId = applied.targetId ?? fidelityRead.targetId;
1128
- if (fidelityRead.fidelity === undefined) {
1129
- warnings.push(`Mobile emulation was applied but the page's own report could not be read (${fidelityRead.unavailable ?? "no fidelity read"}); desktopGeometry.fidelity carries the request without a resolved block.`);
1130
- return { fidelity: { tier: "mobile-emulated", requested, applied: applied.applied ?? [] }, warnings, holderName, ...(emulatedTargetId === undefined ? {} : { targetId: emulatedTargetId }) };
1131
- }
1132
- const resolved = { ...fidelityRead.fidelity, source: "cdp" };
1133
- if (resolved.innerWidth !== request.width) {
1134
- warnings.push(`Mobile emulation requested a ${request.width} px viewport; the page reports ${resolved.innerWidth} px.`);
1135
- }
1136
- if (resolved.devicePixelRatio !== request.deviceScaleFactor) {
1137
- warnings.push(`Mobile emulation requested devicePixelRatio ${request.deviceScaleFactor}; the page reports ${resolved.devicePixelRatio}.`);
1138
- }
1139
- if (request.touch && resolved.maxTouchPoints === 0) {
1140
- warnings.push("Mobile emulation requested touch; the page reports navigator.maxTouchPoints 0.");
1141
- }
1142
- if (!resolved.userAgent.includes("Mobile") && !resolved.userAgent.includes("Android") && !resolved.userAgent.includes("iPhone")) {
1143
- warnings.push("Mobile emulation requested a mobile user agent; the page reports a desktop one.");
1144
- }
1145
- return { fidelity: { tier: "mobile-emulated", requested, applied: applied.applied ?? [], resolved }, warnings, holderName, ...(emulatedTargetId === undefined ? {} : { targetId: emulatedTargetId }) };
1146
- }
1147
- function isMeasuredRect(value) {
1148
- if (!value || typeof value !== "object")
1149
- return false;
1150
- const record = value;
1151
- return Number.isFinite(record.x)
1152
- && Number.isFinite(record.y)
1153
- && isPositiveMeasurement(record.width)
1154
- && isPositiveMeasurement(record.height);
1155
- }
1156
- function isMeasuredViewport(value) {
1157
- if (!value || typeof value !== "object")
1158
- return false;
1159
- const record = value;
1160
- return isPositiveMeasurement(record.width)
1161
- && isPositiveMeasurement(record.height)
1162
- && isPositiveMeasurement(record.deviceScaleFactor);
1163
- }
1164
- function isPositiveMeasurement(value) {
1165
- return typeof value === "number" && Number.isFinite(value) && value > 0;
1166
- }
1167
- async function measureBrowserWindowWithXdotool(desktop, windowId, requestTimeoutMs) {
1168
- const result = await desktop.commands.run([
1169
- "set -euo pipefail",
1170
- `win=${shellSingleQuote(windowId)}`,
1171
- "xdotool getwindowgeometry --shell \"$win\" 2>/dev/null || true"
1172
- ].join("\n"), { requestTimeoutMs, timeoutMs: 5_000 });
1173
- const output = result.stdout ?? "";
1174
- const read = (name) => {
1175
- const raw = output.match(new RegExp(`^${name}=(-?\\d+)$`, "m"))?.[1];
1176
- if (raw === undefined)
1177
- return undefined;
1178
- const value = Number(raw);
1179
- return Number.isFinite(value) ? value : undefined;
1180
- };
1181
- const x = read("X");
1182
- const y = read("Y");
1183
- const width = read("WIDTH");
1184
- const height = read("HEIGHT");
1185
- if (x === undefined || y === undefined || width === undefined || height === undefined || width <= 0 || height <= 0) {
1186
- return undefined;
1187
- }
1188
- return { x, y, width, height, source: "xdotool" };
1189
- }
1190
- /** Physical X client bounds, never the page's emulated window.outerWidth/Height. */
1191
- function isBrowserWindowContained(bounds, [width, height]) {
1192
- return bounds.x >= 0 && bounds.y >= 0
1193
- && bounds.x + bounds.width <= width && bounds.y + bounds.height <= height;
1194
- }
1195
- /** One bounded repair. Window-manager decorations may keep the client origin below (0, 0),
1196
- * so a full-screen client height can clip the bottom even after windowmove succeeds. */
1197
- async function fitBrowserWindowWithinDesktop(desktop, windowId, resolution, requestTimeoutMs) {
1198
- const run = (command) => desktop.commands.run([
1199
- "set -euo pipefail",
1200
- `win=${shellSingleQuote(windowId)}`,
1201
- command
1202
- ].join("\n"), { requestTimeoutMs, timeoutMs: 5_000 }).catch(() => undefined);
1203
- await run('xdotool windowmove "$win" 0 0');
1204
- await desktop.wait(250).catch(() => undefined);
1205
- const moved = await measureBrowserWindowWithXdotool(desktop, windowId, requestTimeoutMs).catch(() => undefined);
1206
- if (moved === undefined)
1207
- return moved;
1208
- let resized = moved;
1209
- const width = resolution[0] - moved.x;
1210
- const height = resolution[1] - moved.y;
1211
- // Resizing alone cannot fix an offscreen client origin. The window manager
1212
- // can also center a minimum-width client at a negative x on a narrow screen.
1213
- if (moved.x >= 0 && moved.y >= 0 && width > 0 && height > 0) {
1214
- await run(`xdotool windowsize "$win" ${width} ${height}`);
1215
- await desktop.wait(250).catch(() => undefined);
1216
- const measured = await measureBrowserWindowWithXdotool(desktop, windowId, requestTimeoutMs).catch(() => undefined);
1217
- if (measured === undefined)
1218
- return measured;
1219
- resized = measured;
1220
- }
1221
- if (resized === undefined || isBrowserWindowContained(resized, resolution))
1222
- return resized;
1223
- // Chrome's minimum client width can equal the whole desktop. Window-manager
1224
- // borders then make a decorated window impossible to contain, even after a
1225
- // successful move/resize. Request fullscreen once and prove the physical result.
1226
- // xprop/xdotool ship with the desktop template; wmctrl is not required.
1227
- // Check state first so the fullscreen shortcut cannot toggle an existing state off.
1228
- await run([
1229
- 'state=$(xprop -id "$win" _NET_WM_STATE)',
1230
- 'case "$state" in',
1231
- ' *_NET_WM_STATE_FULLSCREEN*) ;;',
1232
- ' *) xdotool windowactivate --sync "$win"; xdotool key --clearmodifiers F11 ;;',
1233
- 'esac'
1234
- ].join("\n"));
1235
- // The fullscreen animation may report its new origin before its final width.
1236
- // Give the window manager a bounded settling window, keeping missing reads unverified.
1237
- for (let attempt = 0; attempt < 4; attempt += 1) {
1238
- await desktop.wait(250).catch(() => undefined);
1239
- const measured = await measureBrowserWindowWithXdotool(desktop, windowId, requestTimeoutMs).catch(() => undefined);
1240
- if (measured === undefined || isBrowserWindowContained(measured, resolution))
1241
- return measured;
1242
- resized = measured;
1243
- }
1244
- return resized;
1245
- }
1246
- /** Shared hosted-browser geometry capture used by per-lane and sequential shared-world routes. */
1247
- export async function captureDesktopBrowserGeometry(args) {
1248
- const warnings = [];
1249
- let browserWindowId = args.browserWindowId;
1250
- if (browserWindowId === undefined && args.browserFamily !== "unknown") {
1251
- browserWindowId = await findVisibleBrowserWindowId(args.desktop, args.requestTimeoutMs, args.browserFamily, args.launchIdentity).catch((error) => {
1252
- warnings.push(`Browser window lookup failed for lane ${args.laneId}: ${redactText(toErrorMessage(error))}`);
1253
- return undefined;
1254
- });
1255
- }
1256
- let xdotoolWindow;
1257
- if (browserWindowId !== undefined) {
1258
- if (args.resize !== false) {
1259
- await fillDesktopBrowserWindow(args.desktop, browserWindowId, args.requestedScreen, args.requestTimeoutMs);
1260
- // Let the window manager apply the resize before querying both X and page layout geometry.
1261
- await args.desktop.wait(250).catch(() => undefined);
1262
- }
1263
- xdotoolWindow = await measureBrowserWindowWithXdotool(args.desktop, browserWindowId, args.requestTimeoutMs)
1264
- .catch(() => undefined);
1265
- }
1266
- else {
1267
- warnings.push(`Browser window bounds could not be measured for lane ${args.laneId}; the live stream will use the full desktop.`);
1268
- }
1269
- let unusable;
1270
- if (xdotoolWindow !== undefined && !isBrowserWindowContained(xdotoolWindow, args.requestedScreen)) {
1271
- const before = xdotoolWindow;
1272
- if (args.resize !== false && browserWindowId !== undefined) {
1273
- xdotoolWindow = await fitBrowserWindowWithinDesktop(args.desktop, browserWindowId, args.requestedScreen, args.requestTimeoutMs);
1274
- if (xdotoolWindow === undefined) {
1275
- // Keep the last measured bad state; a missing observation cannot prove a successful fix.
1276
- xdotoolWindow = before;
1277
- unusable = `Physical browser containment could not be verified after correction for lane ${args.laneId}; the last measured window was clipped.`;
1278
- }
1279
- else if (isBrowserWindowContained(xdotoolWindow, args.requestedScreen)) {
1280
- warnings.push(`Browser window clipping corrected for lane ${args.laneId}; physical bounds are ${xdotoolWindow.width}x${xdotoolWindow.height} at (${xdotoolWindow.x}, ${xdotoolWindow.y}).`);
1281
- }
1282
- }
1283
- if (unusable === undefined && !isBrowserWindowContained(xdotoolWindow, args.requestedScreen)) {
1284
- unusable = `Browser window is outside the captured ${args.requestedScreen[0]}x${args.requestedScreen[1]} desktop for lane ${args.laneId}: physical bounds ${xdotoolWindow.width}x${xdotoolWindow.height} at (${xdotoolWindow.x}, ${xdotoolWindow.y}), right=${xdotoolWindow.x + xdotoolWindow.width}, bottom=${xdotoolWindow.y + xdotoolWindow.height}.`;
1285
- }
1286
- if (unusable !== undefined)
1287
- warnings.push(unusable);
1288
- }
1289
- if (xdotoolWindow === undefined) {
1290
- warnings.push(`Physical browser containment is unverified for lane ${args.laneId}; X window bounds could not be measured. Page-reported outer dimensions can be emulated and do not prove physical visibility.`);
1291
- }
1292
- let cdpUnavailable;
1293
- const chromeGeometry = args.browserFamily === "chromium"
1294
- ? await makeChromeDesktopGeometryObserver(args.desktop, args.requestTimeoutMs, {
1295
- ...(args.launchIdentity?.cdpPort === undefined ? {} : { cdpPort: args.launchIdentity.cdpPort }),
1296
- ...(args.launchIdentity?.profileDir === undefined ? {} : { profileDir: args.launchIdentity.profileDir }),
1297
- targetUrl: args.targetUrl
1298
- }, args.browserTargetId, (reason) => {
1299
- cdpUnavailable = reason;
1300
- }, args.pagePreference ?? "pinned")().catch((error) => {
1301
- cdpUnavailable = toErrorMessage(error);
1302
- return undefined;
1303
- })
1304
- : undefined;
1305
- const browserWindow = xdotoolWindow ?? chromeGeometry?.browserWindow;
1306
- const viewport = chromeGeometry?.viewport;
1307
- // The fill check reads the X window when it was measured: under mobile emulation (#221) the
1308
- // page's window.outerWidth reports the EMULATED screen (414), which is not a fill failure.
1309
- const fillBounds = xdotoolWindow;
1310
- if (!browserWindow) {
1311
- warnings.push(`Browser outer bounds could not be measured for lane ${args.laneId}.`);
1312
- }
1313
- else if (unusable === undefined && fillBounds !== undefined && (fillBounds.x !== 0 || fillBounds.y !== 0 || fillBounds.width !== args.requestedScreen[0] || fillBounds.height !== args.requestedScreen[1])) {
1314
- warnings.push(`Browser window fill did not reach the requested ${args.requestedScreen[0]}x${args.requestedScreen[1]} screen for lane ${args.laneId}; measured physical bounds are ${fillBounds.width}x${fillBounds.height} at (${fillBounds.x}, ${fillBounds.y}).`);
1315
- }
1316
- if (!viewport) {
1317
- // Name the cause, not only the symptom: the same dead DevTools channel that loses the viewport
1318
- // loses every url/text observation, and a reader of the bundle should learn that here (#514).
1319
- const cause = cdpUnavailable === undefined ? "" : ` DevTools probe: ${redactText(cdpUnavailable)}.`;
1320
- warnings.push(args.browserFamily === "firefox"
1321
- ? `Browser CSS viewport measurement is unavailable for Firefox on lane ${args.laneId}; stream.viewport is omitted instead of reading a different browser's CDP endpoint.`
1322
- : `Browser CSS viewport could not be measured for lane ${args.laneId}; stream.viewport is omitted instead of copying the requested screen resolution.${cause}`);
1323
- }
1324
- return {
1325
- ...(unusable === undefined ? {} : { unusable }),
1326
- ...(browserWindowId === undefined ? {} : { browserWindowId }),
1327
- ...(chromeGeometry?.targetId === undefined ? {} : { browserTargetId: chromeGeometry.targetId }),
1328
- ...(browserWindow === undefined ? {} : { browserWindow }),
1329
- ...(viewport === undefined ? {} : { viewport }),
1330
- warnings
1331
- };
1332
- }
1333
- function shellSingleQuote(value) {
1334
- return `'${value.replace(/'/g, "'\\''")}'`;
1335
- }
1336
- /**
1337
- * Prepare a CLI study's runtime and, only when declared, its product (#495, #515).
1338
- *
1339
- * The install runs UNKEYED and before the session starts, for the same reason the clone route
1340
- * provisions its subject first: what is being studied begins when the participant looks at the
1341
- * screen. Omitting install deliberately studies product installation; Node/npm remain a
1342
- * harness prerequisite so the participant can follow the product's public npm instructions.
1343
- */
1344
- async function provisionDesktopCli(desktop, args) {
1345
- const install = args.install;
1346
- const now = () => Date.now();
1347
- if (install === undefined || needsNodeRuntime([install])) {
1348
- const startedAt = now();
1349
- emitPhaseStarted(args.onPhase, now, "runtime", "providing Node/npm for the desktop CLI study");
1350
- const bootstrap = await runDetachedStep(desktop, {
1351
- name: "desktop-cli-runtime-node",
1352
- command: TERMINAL_NODE_BOOTSTRAP_COMMAND,
1353
- cwd: "/home/user",
1354
- timeoutMs: INSTALL_TIMEOUT_MS,
1355
- requestTimeoutMs: args.requestTimeoutMs
1356
- });
1357
- emitPhaseCompleted(args.onPhase, now, startedAt, "runtime", bootstrap.ok, bootstrap.ok
1358
- ? "Node runtime ready"
1359
- : "Node runtime bootstrap failed");
1360
- if (!bootstrap.ok) {
1361
- throw new Error(`desktop-cli runtime bootstrap failed for "${args.product}"`);
1362
- }
1363
- }
1364
- if (install === undefined)
1365
- return;
1366
- const startedAt = now();
1367
- emitPhaseStarted(args.onPhase, now, "install", `installing ${args.product} on the desktop`);
1368
- const result = await runDetachedStep(desktop, {
1369
- name: "desktop-cli-install",
1370
- command: install,
1371
- cwd: "/home/user",
1372
- timeoutMs: INSTALL_TIMEOUT_MS,
1373
- requestTimeoutMs: args.requestTimeoutMs
1374
- });
1375
- emitPhaseCompleted(args.onPhase, now, startedAt, "install", result.ok, result.ok
1376
- ? `${args.product} installed`
1377
- : `installing ${args.product} failed`);
1378
- if (!result.ok) {
1379
- // Fail closed: a participant handed a desktop where the product is not installed would produce
1380
- // a transcript about a missing command, and that finding belongs to the harness, not the tool.
1381
- // The tail rides along, scrubbed before truncation like every other provisioning failure — a
1382
- // bare "install failed" is unactionable to whoever wrote the command.
1383
- throw new Error(args.scrub(`desktop-cli install failed for "${args.product}" (${result.timedOut ? "timed out" : `exit ${result.exitCode ?? "?"}`}): ${tailOf(args.scrub(result.logTail))}`));
1384
- }
1385
- }
1386
- /**
1387
- * Open a terminal window on the desktop.
1388
- *
1389
- * The stock template is XFCE and ships xfce4-terminal (also aliased x-terminal-emulator), verified
1390
- * live before this route was built. `x-terminal-emulator` is tried first so a template that swaps
1391
- * the emulator still works; a desktop with neither is a template problem and fails closed rather
1392
- * than handing a participant an empty screen and calling it a study.
1393
- */
1394
- async function openDesktopTerminal(desktop, requestTimeoutMs, workdir) {
1395
- const dir = workdir ?? "/home/user";
1396
- const result = await runDetachedStep(desktop, {
1397
- name: "desktop-cli-terminal",
1398
- command: [
1399
- "for candidate in x-terminal-emulator xfce4-terminal gnome-terminal konsole xterm; do",
1400
- ' if command -v "$candidate" >/dev/null 2>&1; then',
1401
- // LANG is set on the terminal we open, not globally: the stock image declares no locale, and
1402
- // a study that measures our own mojibake against an unconfigured template would be measuring
1403
- // the template. The PRODUCT-side fix (an ASCII fallback when the locale is not UTF-8) is in
1404
- // src/terminal-encoding.ts, and it is the one that matters for real users.
1405
- ` (cd ${shellSingleQuote(dir)} 2>/dev/null || cd /home/user; DISPLAY=:0 LANG=C.UTF-8 LC_ALL=C.UTF-8 HUMANISH_STUDY_PARTICIPANT=1 nohup "$candidate" >/dev/null 2>&1 &)`,
1406
- " sleep 3",
1407
- ' echo "humanish: opened $candidate"',
1408
- " exit 0",
1409
- " fi",
1410
- "done",
1411
- "echo 'humanish: no terminal emulator on this desktop template' >&2",
1412
- "exit 1"
1413
- ].join("\n"),
1414
- cwd: "/home/user",
1415
- timeoutMs: 60_000,
1416
- requestTimeoutMs
1417
- });
1418
- if (!result.ok) {
1419
- throw new Error("desktop-cli lane could not open a terminal on this desktop template");
1420
- }
1421
- }
1422
- async function startDesktopStream(desktop, browserWindowId) {
1423
- if (!browserWindowId) {
1424
- await desktop.stream.start({ requireAuth: true });
1425
- return;
1426
- }
1427
- try {
1428
- await desktop.stream.start({ requireAuth: true, windowId: browserWindowId });
1429
- }
1430
- catch {
1431
- await desktop.stream.start({ requireAuth: true });
1432
- }
1433
- }
1434
541
  // "can't" followed by a PERCEPTION verb describes what the screen showed, not an inability to
1435
542
  // proceed: "the canvas truncates it so you can't even read the whole thing", "I can't tell from
1436
543
  // the screen whether the rename is persisted", "so I could not read its full description". Five
@@ -1640,118 +747,18 @@ export function resolveSelfReportedFriction(session) {
1640
747
  return reports.join("\n\n");
1641
748
  return undefined;
1642
749
  }
1643
- /**
1644
- * Run ONE E2B desktop lane end-to-end: create the sandbox (per-lane metadata + the lane's device
1645
- * resolution), prepareDesktop, verify geometry, (clone+serve+seed the subject per lane), open the
1646
- * browser, run the session, and ALWAYS tear down THIS lane's sandbox BY ID in a finally. Never
1647
- * enumerates sandboxes. Extracted from the former single-lane block; at N=1 it writes the exact
1648
- * same artifacts (actor.json, screenshots/<name>) the bundle has always referenced.
1649
- */
750
+ /** Run one participant against a prepared desktop. The adapter owns provisioning, final
751
+ * evidence and cleanup; this runner owns the model, trace and participant outcome. */
1650
752
  export async function runCuaLane(spec, deps) {
1651
- const { config, appUrl, cloneRoute, localTreeRoute, serve, subjectRepo, subjectEnvNames } = deps;
1652
- const desktopCliRoute = deps.desktopCliRoute === true;
1653
- // The local brain, when there is one. `appServer` / `claudeSession` own a process, so the lane
1654
- // closes it.
753
+ const { config, env } = deps;
1655
754
  let appServer;
1656
755
  let claudeSession;
1657
756
  let localAgentProvider;
1658
- const subjectEnvValues = config.subject.envValues ?? {};
1659
- const targetUrl = spec.targetUrl ?? appUrl;
1660
- const env = deps.env;
1661
- // Off-app comms (#297): on an in-sandbox subject route, redirect the app's email-API sends into an
1662
- // in-sandbox catch (loopback) so its verification mail is CAPTURED, not sent to the internet. Gated
1663
- // ENTIRELY on config.comms — no comms declared → zero change. The base-URL env is injected at
1664
- // sandbox-create (below, so the app reads it at boot); the catch is started right after create.
1665
- const commsEmail = (cloneRoute || localTreeRoute) && config.comms?.email?.kind === "fake" ? config.comms.email : undefined;
1666
- const commsPort = commsEmail ? (commsEmail.port ?? DEFAULT_SANDBOX_CATCH_PORT) : undefined;
1667
- // Hoisted so the finally can drain the catch before teardown; `commsArtifactPath` is the written
1668
- // evidence path folded into the lane outcome.
1669
- let deployedComms;
1670
- let commsArtifactPath;
1671
- let receivingInboxUrl;
1672
- // injectEnv is absent on an adopter-hosted plane (#328): there is no subject env to inject
1673
- // because the operator points their own app at their own catch.
1674
- const commsEnv = commsEmail?.injectEnv !== undefined && commsPort !== undefined
1675
- ? { [commsEmail.injectEnv]: `http://127.0.0.1:${commsPort}` }
1676
- : {};
1677
- // SMTP transport: the same idea as injectEnv, but an app that speaks SMTP needs a host and a port
1678
- // rather than a base URL. The catch accepts any credentials (loopback only), yet many apps refuse
1679
- // to boot unless the user/password vars exist at all, so those are injected when declared.
1680
- const commsSmtpPort = commsEmail?.smtp?.port;
1681
- if (commsEmail?.smtp && commsSmtpPort !== undefined) {
1682
- commsEnv[commsEmail.smtp.hostEnv] = "127.0.0.1";
1683
- commsEnv[commsEmail.smtp.portEnv] = String(commsSmtpPort);
1684
- if (commsEmail.smtp.userEnv)
1685
- commsEnv[commsEmail.smtp.userEnv] = commsEmail.smtp.user ?? "humanish";
1686
- if (commsEmail.smtp.passwordEnv)
1687
- commsEnv[commsEmail.smtp.passwordEnv] = commsEmail.smtp.password ?? "humanish";
1688
- }
1689
- // Persona inbox SURFACE (#297 slice B): the loopback URL the persona opens to read captured mail; the
1690
- // origin-rewrite map (identity on this same-sandbox route, but covers localhost/0.0.0.0 alias skew + an
1691
- // operator-declared linkOrigin); and a disposable background loop that renders the surface DURING the
1692
- // session so the inbox is live when the persona checks. The surface uses its OWN FakeInbox + cursor,
1693
- // independent of the teardown evidence drain (two readers of the append-only NDJSON — no double-count).
1694
- const commsInboxUrl = commsEmail && commsPort !== undefined ? `http://127.0.0.1:${commsPort}/inbox` : undefined;
1695
- const commsOriginMap = commsEmail
1696
- ? buildOriginMap({
1697
- ...(config.subject.serve?.url === undefined ? {} : { internalServeUrl: config.subject.serve.url }),
1698
- reachableBaseUrl: targetUrl,
1699
- ...(commsEmail.linkOrigin === undefined ? {} : { linkOrigin: commsEmail.linkOrigin })
1700
- })
1701
- : [];
1702
- const surfaceRecipients = (commsEmail?.recipients ?? [])
1703
- .filter((recipient) => recipient.address !== undefined)
1704
- .map((recipient) => ({ lane: recipient.lane, address: recipient.address }));
1705
- let surfaceRenderedCount = 0;
1706
- let surfaceDisposed = false;
1707
- let releaseSurface = () => { };
1708
- const surfaceDispose = new Promise((resolve) => { releaseSurface = resolve; });
1709
- let surfaceLoop;
1710
757
  const warnings = [];
1711
758
  const screenshots = [];
1712
759
  const writeScreenshot = makeLaneWriteScreenshot(deps.artifactRoot, spec, screenshots);
1713
- const stateStepRecords = [];
1714
- // Completed-only trail (durationMs/ok are set on completed events, never on started ones):
1715
- // this is what survives into bundle.events. The default/injected sink below sees EVERY event,
1716
- // started and completed alike, so an operator watching stderr sees both halves of each phase.
1717
- const phaseRecords = [];
1718
- const onSubjectPhase = (event) => {
1719
- if (event.ok !== undefined) {
1720
- phaseRecords.push(event);
1721
- }
1722
- (deps.hooks.onPhase ?? defaultSubjectPhaseSink)(event, { laneId: spec.laneId, laneCount: deps.laneCount });
1723
- };
1724
760
  let session;
1725
761
  let sessionError;
1726
- let failureCode;
1727
- let sandboxId;
1728
- // Host-side E2B desktop billed-span endpoints, measured via the injected clock. Captured right
1729
- // after create() succeeds and again in the finally after teardown resolves (both the killed and
1730
- // kept-for-debug paths). This measured span excludes allocation before the acquired handle;
1731
- // a kept/unconfirmed allocation gets an extra unknown lifetime cost line.
1732
- let sandboxCreatedAtMs;
1733
- let sandboxTornDownAtMs;
1734
- let desktopResources;
1735
- let killed = false;
1736
- let streamUrl;
1737
- let subjectCommit;
1738
- let desktopBrowser;
1739
- let launchedBrowserFamily = "unknown";
1740
- let browserLaunchIdentity;
1741
- let browserLaunched = false;
1742
- let initialBrowserGeometry;
1743
- let appliedFidelity;
1744
- let emulatedTargetId;
1745
- let emulationHolderName;
1746
- let browserWindowId;
1747
- let browserTargetId;
1748
- const declaredScreen = declaredScreenForRender(spec.devicePreset, spec.deviceName, spec.resolution);
1749
- let desktopGeometry = {
1750
- screen: {
1751
- requested: { width: spec.resolution[0], height: spec.resolution[1] },
1752
- ...(declaredScreen ? { declared: declaredScreen } : {})
1753
- }
1754
- };
1755
762
  let provisioned = false;
1756
763
  let signaled = false;
1757
764
  const signal = (ok) => {
@@ -1760,635 +767,137 @@ export async function runCuaLane(spec, deps) {
1760
767
  deps.signalProvisioned(ok);
1761
768
  }
1762
769
  };
1763
- let desktopModule;
1764
- let desktop;
770
+ const desktopLane = deps.createDesktopLane?.(spec, warnings) ?? createE2BCuaDesktopLane(spec, deps, warnings);
1765
771
  try {
1766
- desktopModule = await (deps.hooks.loadDesktopModule ?? loadE2BDesktopModule)();
1767
- // Optional custom desktop template (image): present → Sandbox.create(template, opts); absent →
1768
- // the byte-stable Sandbox.create(opts) default (stock `desktop` template).
1769
- desktop = await createDesktopSandbox(desktopModule, {
1770
- apiKey: deps.e2bApiKey,
1771
- requestTimeoutMs: deps.requestTimeoutMs,
1772
- timeoutMs: deps.perLaneSandboxMs,
1773
- metadata: {
1774
- ...CUA_ACTOR_LAB_PROVIDER_METADATA,
1775
- labId: config.id,
1776
- simId: spec.simId,
1777
- laneId: spec.laneId,
1778
- laneIndex: String(spec.laneIndex),
1779
- laneCount: String(deps.laneCount)
1780
- },
1781
- // Env placement per the doctrine: the ACTOR's key never enters the sandbox (the model drives
1782
- // from outside). The SUBJECT's declared env NAMES are provisioned here on the clone route.
1783
- // Three sources, in precedence order: committed non-secret config (subject.envValues), then
1784
- // secret values forwarded from the caller's environment (subject.env), then the harness's own
1785
- // comms wiring, which must win because only it knows the catch's address.
1786
- ...(subjectEnvNames.length > 0 || Object.keys(subjectEnvValues).length > 0 || Object.keys(commsEnv).length > 0
1787
- ? {
1788
- envs: {
1789
- ...subjectEnvValues,
1790
- ...Object.fromEntries(subjectEnvNames.map((name) => [name, env[name]])),
1791
- ...commsEnv
1792
- }
1793
- }
1794
- : {}),
1795
- resolution: spec.resolution,
1796
- dpi: 96,
1797
- lifecycle: { onTimeout: "kill" }
1798
- }, config.execution?.desktop?.template, {
1799
- // The default loader reclaims an acquired handle before retrying failed desktop startup.
1800
- // Its error names the cleanup outcome; pre-construction allocation failures remain unowned.
1801
- onRetry: (reason) => {
1802
- const named = redactText(deps.scrubKnownValues(reason));
1803
- warnings.push(`Sandbox create for lane ${spec.laneId} retried once after a transient provider error (${named}).`);
1804
- onSubjectPhase({ at: new Date(deps.now()).toISOString(), type: "cua-lab.sandbox.create.retry", message: `sandbox create retried once (${named})` });
1805
- }
1806
- });
1807
- sandboxId = desktop.sandboxId;
1808
- // #358 salvage: journal the id to disk before any work — an interrupted run reclaims by
1809
- // exact recorded id (`humanish reclaim`), never by enumerating the account.
1810
- await appendSandboxReceipt(deps.artifactRoot, { at: new Date(deps.now()).toISOString(), laneId: spec.laneId, sandboxId, timeoutMs: deps.perLaneSandboxMs });
1811
- // The billed span starts the instant the sandbox exists.
1812
- sandboxCreatedAtMs = deps.now();
1813
- desktopResources = await observeDesktopResources(desktop);
1814
- if ("reason" in desktopResources) {
1815
- warnings.push(`Desktop resource size unavailable (${desktopResources.reason}); compute cost remains unpriced.`);
1816
- }
1817
- if (deps.hooks.prepareDesktop) {
1818
- await deps.hooks.prepareDesktop(desktop, { laneId: spec.laneId, laneIndex: spec.laneIndex, laneCount: deps.laneCount });
1819
- }
1820
- // Start the in-sandbox email catch BEFORE the subject serve, so the app's send-API base URL (injected
1821
- // into its env at create) resolves the moment it boots. A comms-declared lab that can't stand the
1822
- // catch up is a setup failure (fail closed) rather than silently sending real mail.
1823
- if (deps.receiving) {
1824
- const surface = await deployReceivingInbox(desktop, { leaseId: spec.streamId, requestTimeoutMs: Math.min(deps.requestTimeoutMs, 30_000) });
1825
- receivingInboxUrl = surface.url;
1826
- const email = config.comms?.email;
1827
- try {
1828
- await deps.receiving.attach(spec.laneId, {
1829
- surface,
1830
- allowedOrigins: [...new Set([new URL(targetUrl).origin, ...(email?.allowedOrigins ?? [])])],
1831
- originMap: buildOriginMap({
1832
- ...(config.subject.serve?.url === undefined ? {} : { internalServeUrl: config.subject.serve.url }),
1833
- reachableBaseUrl: targetUrl,
1834
- ...(email?.linkOrigin === undefined ? {} : { linkOrigin: email.linkOrigin })
1835
- })
1836
- });
1837
- commsArtifactPath = "comms/receiving.json";
1838
- }
1839
- catch (error) {
1840
- await surface.stop().catch(() => { });
1841
- throw error;
1842
- }
1843
- }
1844
- if (commsEmail && commsPort !== undefined) {
1845
- deployedComms = await deployCommsCatch(desktop, {
1846
- port: commsPort,
1847
- ...(commsSmtpPort === undefined ? {} : { smtpPort: commsSmtpPort }),
1848
- requestTimeoutMs: deps.requestTimeoutMs
772
+ await desktopLane.prepare();
773
+ // Start the brain BEFORE the first screenshot: the app-server handshake is ~500ms, and it
774
+ // is paid here, while the sandbox is still settling, rather than inside turn one.
775
+ if (deps.localAgent === "codex") {
776
+ appServer = await startAppServerSession({
777
+ ...(spec.reasoningEffort === undefined ? {} : { reasoningEffort: spec.reasoningEffort }),
778
+ ...(config.actors[0]?.model === undefined ? {} : { model: config.actors[0].model }),
779
+ // The persona lives on the THREAD, so it is stated once instead of re-sent every turn.
780
+ baseInstructions: spec.instructions
1849
781
  });
1850
- if (!deployedComms.ready) {
1851
- throw new Error(`comms email catch did not become ready on 127.0.0.1:${commsPort} in the subject sandbox`);
1852
- }
1853
- // Write the EMPTY inbox once up front so the persona's /inbox always resolves to the "No messages
1854
- // yet." page — never a bare 404 — the instant it navigates there, even before any mail arrives OR if
1855
- // the app sends to an address no declared recipient matches (the loop only re-renders on new mail).
1856
- await writeInboxSurface(desktop, deployedComms.surfaceDir, [], { originMap: commsOriginMap, requestTimeoutMs: deps.requestTimeoutMs });
1857
- const deployedRef = deployedComms;
1858
- surfaceLoop = (async () => {
1859
- // Render-first (so even a short session gets a populated inbox), then refresh on a cadence. The
1860
- // cadence uses a REAL timer, NOT the injected instant clock: this loop is unbounded, so an instant
1861
- // sleep would busy-spin and starve the session's own timers. The wait is interruptible by
1862
- // surfaceDispose (and the timer cleared) so teardown never blocks for a full cadence. Each refresh
1863
- // is a full, idempotent rebuild; `surfaceRenderedCount` only advances on a SUCCESSFUL render so a
1864
- // transient failure retries cleanly (no duplicate emails).
1865
- for (;;) {
1866
- try {
1867
- const refreshed = await refreshInboxSurface({
1868
- desktop,
1869
- deployed: deployedRef,
1870
- recipients: surfaceRecipients,
1871
- sinceCount: surfaceRenderedCount,
1872
- originMap: commsOriginMap,
1873
- requestTimeoutMs: deps.requestTimeoutMs
1874
- });
1875
- if (refreshed.rendered)
1876
- surfaceRenderedCount = refreshed.count;
1877
- }
1878
- catch {
1879
- // Never throw into the render loop; the teardown drain + by-id teardown must still run.
1880
- }
1881
- if (surfaceDisposed)
1882
- break;
1883
- await new Promise((resolve) => {
1884
- const timer = setTimeout(resolve, INBOX_SURFACE_CADENCE_MS);
1885
- void surfaceDispose.then(() => { clearTimeout(timer); resolve(); });
1886
- });
1887
- if (surfaceDisposed)
1888
- break;
1889
- }
1890
- })();
1891
- }
1892
- // Per-lane geometry assertion (fail-closed) — the device claim is verified in-sandbox.
1893
- const screenGeometry = await inspectDesktopScreenGeometry({
1894
- desktop,
1895
- laneId: spec.laneId,
1896
- requestedScreen: spec.resolution,
1897
- requestTimeoutMs: deps.requestTimeoutMs
1898
- });
1899
- if (screenGeometry.verified) {
1900
- desktopGeometry = {
1901
- ...desktopGeometry,
1902
- screen: { ...desktopGeometry.screen, verified: screenGeometry.verified }
1903
- };
1904
- }
1905
- if (screenGeometry.warning) {
1906
- warnings.push(screenGeometry.warning);
1907
- desktopGeometry = { ...desktopGeometry, warnings: [screenGeometry.warning] };
1908
- }
1909
- if (screenGeometry.error && deps.screenMismatchPolicy !== "record-evidence") {
1910
- sessionError = screenGeometry.error;
1911
- failureCode = "HUMANISH_CUA_LAB_DEVICE_GEOMETRY";
1912
- }
1913
- else {
1914
- if (screenGeometry.error && screenGeometry.verified) {
1915
- // record-evidence policy: the bundle keeps requested vs verified as separate facts and
1916
- // discloses the divergence instead of failing this lane's world mid-flight.
1917
- const mismatchWarning = deps.scrubKnownValues(`Lane ${spec.laneId} requested a ${spec.resolution[0]}x${spec.resolution[1]} screen but xdpyinfo reports ${screenGeometry.verified.width}x${screenGeometry.verified.height}; recording requested vs verified separately instead of failing the lane closed.`);
1918
- warnings.push(mismatchWarning);
1919
- desktopGeometry = {
1920
- ...desktopGeometry,
1921
- warnings: [...(desktopGeometry.warnings ?? []), mismatchWarning]
1922
- };
1923
- }
1924
- if (desktopCliRoute) {
1925
- // Prepare the runtime and any declared product install, UNKEYED. With install omitted,
1926
- // the participant discovers and installs the product from its public surfaces.
1927
- await provisionDesktopCli(desktop, {
1928
- product: config.subject.product?.name ?? "",
1929
- ...(config.subject.product?.install === undefined ? {} : { install: config.subject.product.install }),
1930
- requestTimeoutMs: deps.requestTimeoutMs,
1931
- scrub: deps.scrubKnownValues,
1932
- onPhase: onSubjectPhase
1933
- });
1934
- }
1935
- if (cloneRoute && serve && subjectRepo) {
1936
- subjectCommit = await provisionCloneSubject(desktop, {
1937
- repo: subjectRepo,
1938
- depth: config.subject.clone?.depth ?? 1,
1939
- serve,
1940
- ...(config.subject.state === undefined ? {} : { state: config.subject.state }),
1941
- hasGithubToken: deps.hasGithubToken,
1942
- requestTimeoutMs: deps.requestTimeoutMs,
1943
- scrub: deps.scrubKnownValues,
1944
- onCommit: (commit) => {
1945
- subjectCommit = commit;
1946
- },
1947
- onStateStep: (record) => {
1948
- stateStepRecords.push(record);
1949
- },
1950
- onPhase: onSubjectPhase,
1951
- ...(deps.hooks.detachedTimers ?? {})
1952
- });
1953
- }
1954
- else if (localTreeRoute && serve && deps.localTreeArchiveBuffer) {
1955
- await provisionLocalTreeSubject(desktop, {
1956
- archiveBuffer: deps.localTreeArchiveBuffer,
1957
- serve,
1958
- ...(config.subject.state === undefined ? {} : { state: config.subject.state }),
1959
- requestTimeoutMs: deps.requestTimeoutMs,
1960
- scrub: deps.scrubKnownValues,
1961
- onStateStep: (record) => {
1962
- stateStepRecords.push(record);
1963
- },
1964
- onPhase: onSubjectPhase,
1965
- ...(deps.hooks.detachedTimers ?? {})
782
+ localAgentProvider = appServer.provider;
783
+ }
784
+ else if (deps.localAgent === "claude") {
785
+ // One session for the whole run, like the codex thread above (#520). The one-shot
786
+ // provider (createLocalAgentProvider) spawned `claude -p` per turn, and every turn
787
+ // started with no memory of the last. HUMANISH_LOCAL_AGENT_ONE_SHOT=1 keeps that path
788
+ // reachable as a MEASUREMENT switch: MemTrapBench (2026-08) reports memory frameworks
789
+ // degrading agent performance by 10-40% on some tasks, so "remembers" has to be measured
790
+ // against "does not" on the same lab, not assumed. The trace records which one ran.
791
+ const oneShot = env.HUMANISH_LOCAL_AGENT_ONE_SHOT !== undefined
792
+ && env.HUMANISH_LOCAL_AGENT_ONE_SHOT !== ""
793
+ && env.HUMANISH_LOCAL_AGENT_ONE_SHOT !== "0";
794
+ if (oneShot) {
795
+ localAgentProvider = createLocalAgentProvider({
796
+ agent: "claude",
797
+ ...(spec.reasoningEffort === undefined ? {} : { reasoningEffort: spec.reasoningEffort }),
798
+ ...(config.actors[0]?.model === undefined ? {} : { model: config.actors[0].model })
1966
799
  });
1967
800
  }
1968
- if (!desktopCliRoute) {
1969
- const requestedFidelity = config.execution?.desktop?.fidelity;
1970
- // A declared camera (#509) is in place before the browser starts: the feed is generated or
1971
- // uploaded first, and a feed that cannot be produced fails the lane closed here.
1972
- const requestedMedia = config.execution?.desktop?.media;
1973
- const mediaEvidence = requestedMedia === undefined
1974
- ? undefined
1975
- : await prepareDesktopMedia(desktop, requestedMedia, config.policies?.mediaPermission ?? "prompt", deps.labCwd, deps.requestTimeoutMs);
1976
- const browserLaunch = await openDesktopBrowserTarget(desktop, targetUrl, deps.requestTimeoutMs, config.execution?.desktop?.browser, [
1977
- ...(requestedFidelity?.mobileEmulation && spec.devicePreset.isMobile
1978
- ? [
1979
- `--user-agent=${requestedFidelity.userAgent ?? DEFAULT_MOBILE_USER_AGENT}`,
1980
- ...(requestedFidelity.touch === false ? [] : ["--touch-events=enabled"])
1981
- ]
1982
- : []),
1983
- ...(mediaEvidence?.flags ?? [])
1984
- ]);
1985
- desktopBrowser = mediaEvidence === undefined
1986
- ? browserLaunch.evidence
1987
- : { requested: config.execution?.desktop?.browser ?? "default", ...(browserLaunch.evidence ?? {}), media: mediaEvidence };
1988
- if (mediaEvidence !== undefined && browserLaunch.family !== "chromium") {
1989
- throw new Error(`execution.desktop.media needs Chrome or Chromium on lane ${spec.laneId} (the fake-device flags are Chromium's); the launched browser family is ${browserLaunch.family}. Set execution.desktop.browser: chrome.`);
1990
- }
1991
- launchedBrowserFamily = browserLaunch.family;
1992
- browserLaunchIdentity = browserLaunch.identity;
1993
- browserLaunched = true;
1994
- await desktop.wait(BROWSER_SETTLE_MS).catch(() => undefined);
1995
- // Mobile fidelity beyond viewport size (#221): applied to the launch page before the
1996
- // geometry capture and the participant's first observation, OUTSIDE the stream/geometry
1997
- // try below (whose catch degrades to a warning): a request that cannot be applied fails
1998
- // the lane closed with the reason.
1999
- // Only lanes on a mobile preset are emulated: a run-wide flag must not hand a desktop or
2000
- // tablet lane an iPhone user agent (the first live proof did exactly that to the desktop
2001
- // newcomer beside the phone lane). Those lanes carry no fidelity block, which is honest.
2002
- const fidelityRequest = config.execution?.desktop?.fidelity;
2003
- if (fidelityRequest?.mobileEmulation && spec.devicePreset.isMobile) {
2004
- if (launchedBrowserFamily !== "chromium") {
2005
- throw new Error(`execution.desktop.fidelity.mobileEmulation needs Chrome or Chromium on lane ${spec.laneId}; the launched browser family is ${launchedBrowserFamily}. Set execution.desktop.browser: chrome.`);
2006
- }
2007
- const applied = await applyMobileEmulation(desktop, deps.requestTimeoutMs, {
2008
- ...(browserLaunchIdentity?.cdpPort === undefined ? {} : { cdpPort: browserLaunchIdentity.cdpPort }),
2009
- ...(browserLaunchIdentity?.profileDir === undefined ? {} : { profileDir: browserLaunchIdentity.profileDir }),
2010
- targetUrl
2011
- }, browserTargetId, {
2012
- width: spec.devicePreset.width,
2013
- height: spec.devicePreset.height,
2014
- deviceScaleFactor: fidelityRequest.deviceScaleFactor ?? spec.devicePreset.deviceScaleFactor,
2015
- touch: fidelityRequest.touch ?? true,
2016
- userAgent: fidelityRequest.userAgent ?? DEFAULT_MOBILE_USER_AGENT
2017
- });
2018
- appliedFidelity = applied.fidelity;
2019
- emulatedTargetId = applied.targetId;
2020
- emulationHolderName = applied.holderName;
2021
- warnings.push(...applied.warnings);
2022
- }
2023
- }
2024
801
  else {
2025
- // A terminal window, opened the way the browser is opened on every other route: the
2026
- // participant arrives at a desktop with the thing they were asked to use already in front
2027
- // of them. They can still open another from the dock — that is the point of a desktop.
2028
- await openDesktopTerminal(desktop, deps.requestTimeoutMs, config.subject.product?.workdir);
2029
- await desktop.wait(BROWSER_SETTLE_MS).catch(() => undefined);
2030
- }
2031
- // Start the brain BEFORE the first screenshot: the app-server handshake is ~500ms, and it
2032
- // is paid here, while the sandbox is still settling, rather than inside turn one.
2033
- if (deps.localAgent === "codex") {
2034
- appServer = await startAppServerSession({
802
+ claudeSession = await startClaudeSession({
2035
803
  ...(spec.reasoningEffort === undefined ? {} : { reasoningEffort: spec.reasoningEffort }),
2036
- ...(config.actors[0]?.model === undefined ? {} : { model: config.actors[0].model }),
2037
- // The persona lives on the THREAD, so it is stated once instead of re-sent every turn.
2038
- baseInstructions: spec.instructions
804
+ ...(config.actors[0]?.model === undefined ? {} : { model: config.actors[0].model })
2039
805
  });
2040
- localAgentProvider = appServer.provider;
2041
- }
2042
- else if (deps.localAgent === "claude") {
2043
- // One session for the whole run, like the codex thread above (#520). The one-shot
2044
- // provider (createLocalAgentProvider) spawned `claude -p` per turn, and every turn
2045
- // started with no memory of the last. HUMANISH_LOCAL_AGENT_ONE_SHOT=1 keeps that path
2046
- // reachable as a MEASUREMENT switch: MemTrapBench (2026-08) reports memory frameworks
2047
- // degrading agent performance by 10-40% on some tasks, so "remembers" has to be measured
2048
- // against "does not" on the same lab, not assumed. The trace records which one ran.
2049
- const oneShot = env.HUMANISH_LOCAL_AGENT_ONE_SHOT !== undefined
2050
- && env.HUMANISH_LOCAL_AGENT_ONE_SHOT !== ""
2051
- && env.HUMANISH_LOCAL_AGENT_ONE_SHOT !== "0";
2052
- if (oneShot) {
2053
- localAgentProvider = createLocalAgentProvider({
2054
- agent: "claude",
2055
- ...(spec.reasoningEffort === undefined ? {} : { reasoningEffort: spec.reasoningEffort }),
2056
- ...(config.actors[0]?.model === undefined ? {} : { model: config.actors[0].model })
2057
- });
2058
- }
2059
- else {
2060
- claudeSession = await startClaudeSession({
2061
- ...(spec.reasoningEffort === undefined ? {} : { reasoningEffort: spec.reasoningEffort }),
2062
- ...(config.actors[0]?.model === undefined ? {} : { model: config.actors[0].model })
2063
- });
2064
- localAgentProvider = claudeSession.provider;
2065
- }
2066
- }
2067
- // World is ready: release the pipeline gate so the remaining lanes may start.
2068
- provisioned = true;
2069
- signal(true);
2070
- try {
2071
- // No browser means no browser geometry, and none is invented: the CSS-viewport facts a
2072
- // browser reports have no counterpart in a terminal window, and an empty record shaped like
2073
- // a measurement would read as one. The screen geometry above is still verified.
2074
- if (!desktopCliRoute) {
2075
- const browserGeometry = await captureDesktopBrowserGeometry({
2076
- desktop,
2077
- browserFamily: launchedBrowserFamily,
2078
- ...(browserLaunchIdentity === undefined ? {} : { launchIdentity: browserLaunchIdentity }),
2079
- laneId: spec.laneId,
2080
- targetUrl,
2081
- requestedScreen: spec.resolution,
2082
- requestTimeoutMs: deps.requestTimeoutMs
2083
- });
2084
- initialBrowserGeometry = browserGeometry;
2085
- browserWindowId = browserGeometry.browserWindowId;
2086
- browserTargetId = browserGeometry.browserTargetId;
2087
- }
2088
- // The WHOLE desktop, not one window: a person studying a terminal app opens other windows,
2089
- // and a stream bound to the first one would quietly stop being evidence.
2090
- await startDesktopStream(desktop, browserWindowId);
2091
- const candidateStreamUrl = desktop.stream.getUrl({
2092
- authKey: desktop.stream.getAuthKey(),
2093
- autoConnect: true,
2094
- viewOnly: true,
2095
- resize: "scale"
2096
- });
2097
- if (typeof candidateStreamUrl === "string" && candidateStreamUrl.trim().length > 0) {
2098
- streamUrl = candidateStreamUrl;
2099
- await deps.hooks.onRuntimeStreamReady?.({
2100
- laneId: spec.laneId,
2101
- sandboxId: desktop.sandboxId,
2102
- simId: spec.simId,
2103
- streamId: spec.streamId,
2104
- url: streamUrl
2105
- });
2106
- }
2107
- else {
2108
- warnings.push("Live desktop stream started but did not return a usable watch URL; Observer will fall back to screenshots.");
2109
- }
806
+ localAgentProvider = claudeSession.provider;
2110
807
  }
2111
- catch (error) {
2112
- warnings.push(`Live desktop stream unavailable (run continues; evidence still captured): ${redactText(deps.scrubKnownValues(toErrorMessage(error)))}`);
2113
- }
2114
- // This is outside the stream's best-effort catch: unusable geometry is a harness failure,
2115
- // never a participant finding about missing controls. Both per-lane and concurrent seats
2116
- // use this route; sequential seats enforce the same capture result in shared-world-lab.
2117
- if (initialBrowserGeometry?.unusable !== undefined) {
2118
- failureCode = "HUMANISH_CUA_LAB_DEVICE_GEOMETRY";
2119
- throw new Error(`${failureCode}: ${initialBrowserGeometry.unusable} Participant actions were not started.`);
2120
- }
2121
- // The FAIL-CLOSED spend cap (execution.caps.maxUsd) is wired into the loop as maxUsd + an
2122
- // injected pure per-turn estimator keyed on the resolved model. Preflight already refused a
2123
- // cap on an unpriced model, so the estimate is measurable whenever a cap is in force. The
2124
- // model id here matches provider.version (openai-responses-cu resolves the default when unset).
2125
- const capModelId = config.actors[0]?.model ?? DEFAULT_OPENAI_CU_MODEL;
2126
- const maxUsd = config.execution?.caps?.maxUsd;
2127
- const sessionOptions = {
2128
- // Tell the persona where its inbox is — but only when comms is live AND this lane has a declared
2129
- // recipient it can actually receive mail into (else it would stall on an inbox that stays
2130
- // empty). Two comms planes, mutually exclusive by parse: the in-sandbox catch humanish
2131
- // deployed, or the adopter-hosted one (#380).
2132
- instructions: deps.receiving && receivingInboxUrl
2133
- ? withInboxMission(spec, receivingInboxUrl, deps.receiving.address(spec.laneId), true).instructions
2134
- : commsEmail && commsInboxUrl && deployedComms?.ready && laneHasInboxRecipient(commsEmail, spec.laneId)
2135
- ? withInboxMission(spec, commsInboxUrl, inboxRecipientFor(commsEmail, spec.laneId)?.address).instructions
2136
- : deps.externalComms && laneHasInboxRecipient(deps.externalComms.email, spec.laneId)
2137
- ? withInboxMission(spec, deps.externalComms.inboxUrl, inboxRecipientFor(deps.externalComms.email, spec.laneId)?.address).instructions
2138
- : spec.instructions,
2139
- persona: spec.persona,
2140
- timeoutMs: deps.timeoutMs,
2141
- // The brain is either a keyed API client or a CLI the operator is already signed in to.
2142
- // Everything below this line — loop, executor, trace, affordances — is identical either
2143
- // way, which is what makes a local-agent run comparable to an API one.
2144
- ...(localAgentProvider === undefined ? {} : { provider: localAgentProvider }),
2145
- openai: {
2146
- apiKey: deps.openaiApiKey,
2147
- ...(config.actors[0]?.model ? { model: config.actors[0].model } : {}),
2148
- // Per-LANE, not per-actor: two lanes at different efforts is the control this exists for.
2149
- ...(spec.reasoningEffort === undefined ? {} : { reasoningEffort: spec.reasoningEffort }),
2150
- ...(spec.maxOutputTokens === undefined ? {} : { maxOutputTokens: spec.maxOutputTokens })
2151
- },
2152
- ...(maxUsd === undefined
2153
- ? {}
2154
- : {
2155
- maxUsd,
2156
- estimateTurnCostUsd: (usage) => estimateActorCost(usage, capModelId).estimatedCostUsd
2157
- }),
2158
- desktop: desktop,
2159
- ...(launchedBrowserFamily === "chromium"
2160
- ? {
2161
- executorOptions: {
2162
- observeBrowserState: makeChromeBrowserStateObserver(desktop, deps.requestTimeoutMs, {
2163
- ...(browserLaunchIdentity?.cdpPort === undefined ? {} : { cdpPort: browserLaunchIdentity.cdpPort }),
2164
- ...(browserLaunchIdentity?.profileDir === undefined ? {} : { profileDir: browserLaunchIdentity.profileDir }),
2165
- targetUrl
2166
- }, browserTargetId,
2167
- // Once per lane: a dark observation channel is a gap in the instrument, and the
2168
- // funnel's NEVER MEASURED count needs this line to explain itself (#514).
2169
- (reason) => {
2170
- warnings.push(`Browser-state observer unavailable for lane ${spec.laneId} (${redactText(deps.scrubKnownValues(reason))}); ` +
2171
- "urlIncludes/urlPathEquals/textIncludes stop conditions and task criteria are NOT being measured this session.");
2172
- }, emulatedTargetId === undefined
2173
- ? undefined
2174
- : {
2175
- emulatedTargetId,
2176
- expectedWidth: spec.devicePreset.width,
2177
- expectTouch: appliedFidelity?.requested.touch === true,
2178
- onDrift: (reason) => {
2179
- warnings.push(`Mobile emulation drift on lane ${spec.laneId}: ${reason} (#623).`);
2180
- },
2181
- onCovered: (coveredTargetId, read) => {
2182
- // A later tab the page itself reported at the phone width: evidence that
2183
- // the emulation followed the participant (#623), kept on the bundle.
2184
- if (appliedFidelity === undefined)
2185
- return;
2186
- appliedFidelity = {
2187
- ...appliedFidelity,
2188
- laterTargets: [...(appliedFidelity.laterTargets ?? []), { targetId: coveredTargetId, ...read }]
2189
- };
2190
- }
2191
- })
2192
- }
2193
- }
2194
- : {}),
2195
- redactScreenshots: deps.redactScreenshots,
2196
- scrubText: deps.scrubKnownValues,
2197
- writeScreenshot,
2198
- ...(spec.idleSteps === undefined ? {} : { idleSteps: spec.idleSteps }),
2199
- ...(spec.noProgressSteps === undefined ? {} : { noProgressSteps: spec.noProgressSteps }),
2200
- ...(spec.stopWhen === undefined ? {} : { stopWhen: spec.stopWhen }),
2201
- ...(spec.dwell === undefined ? {} : { dwell: spec.dwell }),
2202
- ...(spec.tasks === undefined ? {} : { tasks: spec.tasks }),
2203
- // The STUDY budget (#299): this lane notes its own running estimate on the shared ledger
2204
- // and stops when the RUN total crosses the cap — independent of the per-lane maxUsd above.
2205
- ...(deps.runBudget === undefined
2206
- ? {}
2207
- : {
2208
- overRunBudget: (usage) => {
2209
- const estimate = estimateActorCost(usage, capModelId).estimatedCostUsd;
2210
- const totalUsd = deps.runBudget.note(spec.laneId, estimate);
2211
- return totalUsd > deps.runBudget.maxTotalUsd
2212
- ? `study budget reached: the run's estimated model spend $${round6(totalUsd)} crossed execution.caps.maxTotalUsd=$${deps.runBudget.maxTotalUsd}; this lane stops here and sibling lanes stop at their next turn`
2213
- : null;
2214
- }
2215
- }),
2216
- ...(deps.onObservedUrl === undefined ? {} : { onObservedUrl: deps.onObservedUrl }),
2217
- ...(deps.onMessage === undefined ? {} : { onMessage: deps.onMessage }),
2218
- ...(deps.onScreenshot === undefined ? {} : { onScreenshot: deps.onScreenshot }),
2219
- ...(deps.onTrace === undefined
2220
- ? {}
2221
- : {
2222
- // Forwards the RUNNING usage as well: the lane is where both are known, and usage
2223
- // without it never reaches the flush — which is how the live cost stayed unknown.
2224
- onTrace: (items, usage) => deps.onTrace?.(spec.laneId, items, usage)
2225
- })
2226
- };
2227
- session = await deps.runSession(sessionOptions);
2228
808
  }
809
+ // World is ready: release the pipeline gate so the remaining lanes may start.
810
+ provisioned = true;
811
+ signal(true);
812
+ const ready = await desktopLane.openSession();
813
+ // The FAIL-CLOSED spend cap (execution.caps.maxUsd) is wired into the loop as maxUsd + an
814
+ // injected pure per-turn estimator keyed on the resolved model. Preflight already refused a
815
+ // cap on an unpriced model, so the estimate is measurable whenever a cap is in force. The
816
+ // model id here matches provider.version (openai-responses-cu resolves the default when unset).
817
+ const capModelId = config.actors[0]?.model ?? DEFAULT_OPENAI_CU_MODEL;
818
+ const maxUsd = config.execution?.caps?.maxUsd;
819
+ const sessionOptions = {
820
+ instructions: ready.inbox
821
+ ? withInboxMission(spec, ready.inbox.url, ready.inbox.address, ready.inbox.receiving).instructions
822
+ : spec.instructions,
823
+ persona: spec.persona,
824
+ timeoutMs: deps.timeoutMs,
825
+ // The brain is either a keyed API client or a CLI the operator is already signed in to.
826
+ // Everything below this line — loop, executor, trace, affordances — is identical either
827
+ // way, which is what makes a local-agent run comparable to an API one.
828
+ ...(localAgentProvider === undefined ? {} : { provider: localAgentProvider }),
829
+ openai: {
830
+ apiKey: deps.openaiApiKey,
831
+ ...(config.actors[0]?.model ? { model: config.actors[0].model } : {}),
832
+ // Per-LANE, not per-actor: two lanes at different efforts is the control this exists for.
833
+ ...(spec.reasoningEffort === undefined ? {} : { reasoningEffort: spec.reasoningEffort }),
834
+ ...(spec.maxOutputTokens === undefined ? {} : { maxOutputTokens: spec.maxOutputTokens })
835
+ },
836
+ ...(maxUsd === undefined
837
+ ? {}
838
+ : {
839
+ maxUsd,
840
+ estimateTurnCostUsd: (usage) => estimateActorCost(usage, capModelId).estimatedCostUsd
841
+ }),
842
+ executor: ready.executor,
843
+ redactScreenshots: deps.redactScreenshots,
844
+ scrubText: deps.scrubKnownValues,
845
+ writeScreenshot,
846
+ ...(spec.idleSteps === undefined ? {} : { idleSteps: spec.idleSteps }),
847
+ ...(spec.noProgressSteps === undefined ? {} : { noProgressSteps: spec.noProgressSteps }),
848
+ ...(spec.stopWhen === undefined ? {} : { stopWhen: spec.stopWhen }),
849
+ ...(spec.dwell === undefined ? {} : { dwell: spec.dwell }),
850
+ ...(spec.tasks === undefined ? {} : { tasks: spec.tasks }),
851
+ // The STUDY budget (#299): this lane notes its own running estimate on the shared ledger
852
+ // and stops when the RUN total crosses the cap — independent of the per-lane maxUsd above.
853
+ ...(deps.runBudget === undefined
854
+ ? {}
855
+ : {
856
+ overRunBudget: (usage) => {
857
+ const estimate = estimateActorCost(usage, capModelId).estimatedCostUsd;
858
+ const totalUsd = deps.runBudget.note(spec.laneId, estimate);
859
+ return totalUsd > deps.runBudget.maxTotalUsd
860
+ ? `study budget reached: the run's estimated model spend $${round6(totalUsd)} crossed execution.caps.maxTotalUsd=$${deps.runBudget.maxTotalUsd}; this lane stops here and sibling lanes stop at their next turn`
861
+ : null;
862
+ }
863
+ }),
864
+ ...(deps.onObservedUrl === undefined ? {} : { onObservedUrl: deps.onObservedUrl }),
865
+ ...(deps.onMessage === undefined ? {} : { onMessage: deps.onMessage }),
866
+ ...(deps.onScreenshot === undefined ? {} : { onScreenshot: deps.onScreenshot }),
867
+ ...(deps.onTrace === undefined
868
+ ? {}
869
+ : {
870
+ // Forwards the RUNNING usage as well: the lane is where both are known, and usage
871
+ // without it never reaches the flush — which is how the live cost stayed unknown.
872
+ onTrace: (items, usage) => deps.onTrace?.(spec.laneId, items, usage)
873
+ })
874
+ };
875
+ session = await deps.runSession(sessionOptions);
2229
876
  }
2230
877
  catch (error) {
2231
878
  sessionError = redactText(deps.scrubKnownValues(toErrorMessage(error)));
2232
879
  }
2233
880
  finally {
2234
- // The local brain owns a process. Close it before anything else can throw: a leaked
2235
- // app-server per lane would outlive the run and keep a thread open on the operator's plan.
2236
- appServer?.close();
2237
- await claudeSession?.close();
2238
- // Stop the mid-run inbox-surface loop FIRST — before the teardown evidence drain below — so the two
2239
- // `cat`s never overlap and the final surface state is deterministic. A surface failure can never
2240
- // block teardown (the loop body is fully try/caught and this await is on its already-caught promise).
2241
- surfaceDisposed = true;
2242
- releaseSurface();
2243
- if (surfaceLoop)
2244
- await surfaceLoop.catch(() => undefined);
2245
- if (!provisioned) {
2246
- signal(false);
881
+ try {
882
+ appServer?.close();
2247
883
  }
2248
- if (desktop && desktopModule) {
2249
- if (browserLaunched) {
2250
- const finalGeometry = await captureDesktopBrowserGeometry({
2251
- desktop,
2252
- browserFamily: launchedBrowserFamily,
2253
- ...(browserLaunchIdentity === undefined ? {} : { launchIdentity: browserLaunchIdentity }),
2254
- ...(browserWindowId === undefined ? {} : { browserWindowId }),
2255
- ...(browserTargetId === undefined ? {} : { browserTargetId }),
2256
- laneId: spec.laneId,
2257
- targetUrl,
2258
- requestedScreen: spec.resolution,
2259
- requestTimeoutMs: deps.requestTimeoutMs,
2260
- pagePreference: "active",
2261
- resize: false
2262
- }).catch((error) => ({
2263
- warnings: [`Final browser geometry measurement failed for lane ${spec.laneId}: ${redactText(deps.scrubKnownValues(toErrorMessage(error)))}`]
2264
- }));
2265
- // Chosen capture rule: final-if-it-measured-anything, else launch-time. A final capture
2266
- // that measured EITHER field wins whole, so a partial final capture omits fields the
2267
- // launch-time capture had (honest omission); only a final capture that measured NOTHING
2268
- // falls back to the launch-time capture.
2269
- const chosenGeometry = finalGeometry.browserWindow !== undefined || finalGeometry.viewport !== undefined
2270
- ? finalGeometry
2271
- : initialBrowserGeometry ?? finalGeometry;
2272
- const geometryWarnings = [...new Set([...(initialBrowserGeometry?.warnings ?? []), ...chosenGeometry.warnings].map((warning) => deps.scrubKnownValues(warning)))];
2273
- warnings.push(...geometryWarnings);
2274
- // The emulation holder's own log, after its announce line: which later targets it
2275
- // attached to, what it sent, and any reply that came back as an error (#623). Read while
2276
- // the sandbox is alive; the first live proof had no way to say what the holder did.
2277
- if (appliedFidelity !== undefined && emulationHolderName !== undefined) {
2278
- const holderLog = await readDetachedLog(desktop, emulationHolderName, deps.requestTimeoutMs).catch(() => "");
2279
- const lines = holderLog.split("\n").map((line) => line.trim()).filter((line) => line.startsWith("{")).slice(1, 51);
2280
- if (lines.length > 0)
2281
- appliedFidelity = { ...appliedFidelity, holderLog: lines.map((line) => deps.scrubKnownValues(line)) };
2282
- }
2283
- desktopGeometry = {
2284
- screen: desktopGeometry.screen,
2285
- ...(chosenGeometry.browserWindow === undefined ? {} : { browserWindow: chosenGeometry.browserWindow }),
2286
- ...(chosenGeometry.viewport === undefined ? {} : { viewport: chosenGeometry.viewport }),
2287
- ...(appliedFidelity === undefined ? {} : { fidelity: appliedFidelity }),
2288
- ...((desktopGeometry.warnings?.length ?? 0) + geometryWarnings.length === 0
2289
- ? {}
2290
- : { warnings: [...(desktopGeometry.warnings ?? []), ...geometryWarnings] })
2291
- };
2292
- }
2293
- if (deps.receiving) {
2294
- try {
2295
- await deps.receiving.finishParticipant(spec.laneId);
2296
- }
2297
- catch {
2298
- warnings.push("Real email finalization is incomplete. Inspect communication cleanup with humanish comms recover.");
2299
- }
2300
- }
2301
- // Off-app comms evidence (#297): before this lane's sandbox is torn down, drain everything the
2302
- // in-sandbox catch captured, route it into a host fake inbox addressed to the declared
2303
- // recipients, and write the digest-only thread artifact. Wrapped so a drain failure NEVER
2304
- // breaks teardown — the sandbox must still be killed either way. Runs only for a ready catch.
2305
- if (commsEmail && deployedComms?.ready) {
2306
- try {
2307
- const commsChannel = new FakeInbox();
2308
- const commsInboxes = [];
2309
- for (const recipient of commsEmail.recipients ?? []) {
2310
- if (recipient.address !== undefined) {
2311
- commsInboxes.push(await commsChannel.provisionAddress(recipient.lane, recipient.address));
2312
- }
2313
- }
2314
- const collected = await collectCommsThread({
2315
- desktop,
2316
- deployed: deployedComms,
2317
- channel: commsChannel,
2318
- inboxes: commsInboxes,
2319
- requestTimeoutMs: deps.requestTimeoutMs
2320
- });
2321
- if (collected.artifact) {
2322
- const path = deps.laneCount === 1 ? "comms/thread.json" : `comms/${spec.streamId}.thread.json`;
2323
- await writeContainedOutputFile(deps.artifactRoot, path, `${JSON.stringify(collected.artifact, null, 2)}\n`, "utf8");
2324
- commsArtifactPath = path;
2325
- }
2326
- else if (collected.captured > 0) {
2327
- // Captured mail that matched no declared recipient must not vanish silently (invariant 6:
2328
- // honest signals): tell the operator to declare comms.email.recipients[].address to match
2329
- // the address the app actually sends to (e.g. the one the persona surface will sign up with).
2330
- warnings.push(`Comms catch captured ${collected.captured} email send(s) but none matched a declared recipient inbox — no comms evidence written. Declare comms.email.recipients[].address to match the address the app sends to.`);
2331
- }
2332
- else {
2333
- // Zero captures is the silent-broken shape (#351): the app never posted to the catch at
2334
- // all, so the personas stared at an empty inbox. Most common cause: the app does not
2335
- // actually read the declared injectEnv var for its email API base URL.
2336
- const transportHint = commsEmail.smtp
2337
- ? `Verify the app reads ${commsEmail.smtp.hostEnv}/${commsEmail.smtp.portEnv} for its SMTP host and port`
2338
- : `Verify the app reads ${commsEmail.injectEnv} for its email API base URL (an SDK that ignores it sends real mail or throws)`;
2339
- warnings.push(`Comms catch captured ZERO email sends — the app never delivered mail through the catch. ${transportHint} and that the flow reached an email step.`);
2340
- }
2341
- }
2342
- catch (error) {
2343
- warnings.push(`Comms evidence collection failed (run continues; sandbox still torn down): ${redactText(deps.scrubKnownValues(toErrorMessage(error)))}`);
2344
- }
2345
- }
2346
- const failed = sessionError !== undefined || session === undefined;
2347
- // Each route's own keep flag gates its own lane only: a clone.keep can never leak into
2348
- // a local-tree lane's teardown decision, and vice versa.
2349
- const keepReason = cloneRoute && config.subject.clone?.keep === true
2350
- ? "subject.clone.keep"
2351
- : localTreeRoute && config.subject.localTree?.keep === true
2352
- ? "subject.localTree.keep"
2353
- : undefined;
2354
- const keepForDebug = keepReason !== undefined && failed;
2355
- if (keepForDebug) {
2356
- warnings.push(`Sandbox ${desktop.sandboxId} kept for debugging (${keepReason} on failure); reclaim it via E2B or it will be killed on its server-side timeout.`);
2357
- }
2358
- else if (typeof desktopModule.Sandbox.kill === "function") {
2359
- try {
2360
- await desktopModule.Sandbox.kill(desktop.sandboxId, { requestTimeoutMs: 60_000 });
2361
- killed = true;
2362
- }
2363
- catch (error) {
2364
- warnings.push(`Sandbox teardown failed (server-side kill-on-timeout will reclaim it): ${redactText(deps.scrubKnownValues(toErrorMessage(error)))}`);
2365
- }
2366
- }
2367
- else {
2368
- warnings.push("Installed @e2b/desktop SDK does not expose Sandbox.kill; server-side kill-on-timeout will reclaim the sandbox.");
2369
- }
2370
- // Close the observed span. A kept or unconfirmed sandbox can still accrue compute cost;
2371
- // the summary records that remaining lifetime as unknown instead of calling this complete.
2372
- sandboxTornDownAtMs = deps.now();
2373
- // The lane's live stream is now a dead page whichever teardown path ran (killed, kept, or
2374
- // kill-failed-awaiting-TTL) — tell the watch overlay so the tile falls back to recorded
2375
- // evidence instead of "sandbox not found" (#357). Guarded: a viewer callback must never
2376
- // break teardown.
2377
- if (streamUrl !== undefined) {
2378
- try {
2379
- await deps.hooks.onRuntimeStreamEnded?.({ laneId: spec.laneId, simId: spec.simId, streamId: spec.streamId });
2380
- }
2381
- catch {
2382
- // viewer-side only; nothing to record
2383
- }
2384
- }
884
+ catch {
885
+ warnings.push('Codex session cleanup failed; desktop cleanup will still run.');
886
+ }
887
+ try {
888
+ await claudeSession?.close();
889
+ }
890
+ catch {
891
+ warnings.push('Claude session cleanup failed; desktop cleanup will still run.');
892
+ }
893
+ try {
894
+ if (!provisioned)
895
+ signal(false);
896
+ }
897
+ finally {
898
+ await desktopLane.finalize({ failed: sessionError !== undefined || session === undefined });
2385
899
  }
2386
900
  }
2387
- // Host-side approximation of the E2B desktop's billed lifetime; feeds the desktop-minute cost
2388
- // estimate. Never negative.
2389
- const desktopDurationMs = sandboxCreatedAtMs !== undefined && sandboxTornDownAtMs !== undefined
2390
- ? Math.max(0, sandboxTornDownAtMs - sandboxCreatedAtMs)
2391
- : undefined;
2392
901
  if (session) {
2393
902
  // Per-lane model-token cost ESTIMATE, attached to the trace before it is persisted (the model
2394
903
  // id is authoritative here — provider.version). Kept at the lab boundary so the pure loop
@@ -2419,24 +928,13 @@ export async function runCuaLane(spec, deps) {
2419
928
  spec,
2420
929
  ...(session ? { session } : {}),
2421
930
  ...(sessionError === undefined ? {} : { sessionError }),
2422
- ...(sandboxId === undefined ? {} : { sandboxId }),
2423
- ...(desktopDurationMs === undefined ? {} : { desktopDurationMs }),
2424
- ...(desktopResources === undefined ? {} : { desktopResources }),
2425
- killed,
2426
- streamUrlPresent: streamUrl !== undefined,
931
+ ...desktopLane.snapshot(),
2427
932
  screenshots,
2428
- ...(subjectCommit === undefined ? {} : { subjectCommit }),
2429
- ...(desktopBrowser === undefined ? {} : { desktopBrowser }),
2430
- desktopGeometry,
2431
- stateStepRecords,
2432
- phaseRecords,
2433
933
  warnings,
2434
934
  noEngagement,
2435
935
  selfReportedBlocker,
2436
936
  reportedFriction,
2437
937
  harnessError,
2438
- ...(failureCode === undefined ? {} : { failureCode }),
2439
- ...(commsArtifactPath === undefined ? {} : { commsArtifactPath })
2440
938
  };
2441
939
  }
2442
940
  /** Run the single IN-PROCESS lane (a custom executor + provider; NO E2B). Always one lane. */
@@ -3763,300 +2261,6 @@ function buildSingleLaneBundle(args) {
3763
2261
  phaseEvents: outcome?.phaseRecords ?? []
3764
2262
  });
3765
2263
  }
3766
- /**
3767
- * Shared post-populate provisioning pipeline (clone AND local-tree routes): (install) ->
3768
- * state(before-build) -> (build) -> state(before-start) -> detached start -> readiness probe ->
3769
- * state(after-ready). Both provisioning routes populate SUBJECT_DIR by different means (git
3770
- * clone vs. upload+extract) and then run this identical pipeline unchanged.
3771
- *
3772
- * State steps run through the same detached primitive as serve steps (author-trusted, the
3773
- * "serve commands are author-trusted" corollary) under the reserved `subject-state-<name>`
3774
- * label prefix, so a step name can never collide with subject-clone/subject-extract/install/
3775
- * build/start. after-ready steps complete BEFORE the caller opens the browser: the actor never
3776
- * drives a half-seeded subject and seeding never eats the session budget.
3777
- */
3778
- async function runSubjectServePipeline(desktop, args) {
3779
- const timers = {
3780
- ...(args.now === undefined ? {} : { now: args.now }),
3781
- ...(args.sleep === undefined ? {} : { sleep: args.sleep })
3782
- };
3783
- const now = args.now ?? Date.now;
3784
- const refresh = args.onPhaseComplete ?? (() => Promise.resolve());
3785
- const stateSteps = args.state?.seed ?? [];
3786
- const runStateSteps = async (when) => {
3787
- const steps = stateSteps.filter((step) => (step.when ?? "before-start") === when);
3788
- if (steps.length === 0) {
3789
- // No declared steps for this group: no boundary to report (avoids empty-group noise on
3790
- // every run, since before-build/before-start/after-ready are always called).
3791
- return;
3792
- }
3793
- const groupStartedAt = now();
3794
- emitPhaseStarted(args.onPhase, now, `state.${when}`, `running subject state seed steps (${when})`);
3795
- for (const step of steps) {
3796
- const stepTimeoutMs = step.timeoutMs ?? DEFAULT_STATE_STEP_TIMEOUT_MS;
3797
- const startedAt = now();
3798
- const result = await runDetachedStep(desktop, {
3799
- name: `subject-state-${step.name}`,
3800
- command: step.command,
3801
- cwd: SUBJECT_DIR,
3802
- timeoutMs: stepTimeoutMs,
3803
- requestTimeoutMs: args.requestTimeoutMs,
3804
- ...timers
3805
- });
3806
- args.onStateStep?.({
3807
- name: step.name,
3808
- when,
3809
- // Digest only (sha256-16): the command text never persists: the lab YAML in the
3810
- // consumer's repo is the plaintext source of truth.
3811
- commandDigest: commandDigestOf(step.command),
3812
- ok: result.ok,
3813
- ...(result.exitCode === undefined ? {} : { exitCode: result.exitCode }),
3814
- ...(result.timedOut ? { timedOut: true } : {}),
3815
- durationMs: Math.max(0, now() - startedAt)
3816
- });
3817
- if (!result.ok) {
3818
- emitPhaseCompleted(args.onPhase, now, groupStartedAt, `state.${when}`, false, `subject state seed steps failed (${when})`);
3819
- // Fail closed with the existing scrub-before-truncate tail chain: literal scrub of
3820
- // every provisioned value PRE-truncation, then pattern redaction + cap in tailOf.
3821
- throw new Error(`subject state step "${step.name}" ${result.timedOut ? `timed out after ${stepTimeoutMs}ms` : `failed (exit ${result.exitCode})`}: ${tailOf(args.scrub(result.logTail))}`);
3822
- }
3823
- }
3824
- emitPhaseCompleted(args.onPhase, now, groupStartedAt, `state.${when}`, true, `subject state seed steps complete (${when})`);
3825
- };
3826
- // Provide the runtime the pipeline needs before running it (#371). The stock desktop template
3827
- // ships python3 and curl but no Node, so an `npm install` here used to die at exit 127 after the
3828
- // sandbox was already paid for. Probe-first, so a template that ships its own Node pays nothing.
3829
- const serveCommands = [args.serve.install, args.serve.build, args.serve.start];
3830
- if (needsNodeRuntime(serveCommands)) {
3831
- const runtimeStartedAt = now();
3832
- emitPhaseStarted(args.onPhase, now, "runtime", "providing the Node runtime the serve pipeline needs");
3833
- const bootstrap = await runProvisioningStepWithOneRetry(desktop, {
3834
- name: "subject-runtime-node",
3835
- command: nodeBootstrapCommand(),
3836
- cwd: SUBJECT_DIR,
3837
- timeoutMs: args.serve.installTimeoutMs ?? INSTALL_TIMEOUT_MS,
3838
- requestTimeoutMs: args.requestTimeoutMs,
3839
- timers,
3840
- retryPhase: "runtime-retry",
3841
- retryMessage: "Node runtime bootstrap",
3842
- onPhase: args.onPhase,
3843
- now
3844
- });
3845
- let ok = bootstrap.ok;
3846
- const corepack = ok ? corepackCommandFor(serveCommands) : undefined;
3847
- if (corepack) {
3848
- const pm = await runDetachedStep(desktop, {
3849
- name: "subject-runtime-pm",
3850
- command: corepack,
3851
- cwd: SUBJECT_DIR,
3852
- timeoutMs: args.serve.installTimeoutMs ?? INSTALL_TIMEOUT_MS,
3853
- requestTimeoutMs: args.requestTimeoutMs,
3854
- ...timers
3855
- });
3856
- ok = pm.ok;
3857
- }
3858
- emitPhaseCompleted(args.onPhase, now, runtimeStartedAt, "runtime", ok, ok ? "Node runtime ready" : "could not provide a Node runtime");
3859
- if (!ok) {
3860
- throw new Error(`the subject's serve pipeline needs a Node runtime and this desktop template has none, and bootstrapping one failed${bootstrap.attempts === 2 ? " twice" : ""}: ${tailOf(args.scrub(bootstrap.logTail))}. Use execution.desktop.template with an image that ships Node, or change serve.install to a runtime the template provides.`);
3861
- }
3862
- }
3863
- if (args.serve.install) {
3864
- const installStartedAt = now();
3865
- emitPhaseStarted(args.onPhase, now, "install", "installing subject dependencies");
3866
- const install = await runProvisioningStepWithOneRetry(desktop, {
3867
- name: "subject-install",
3868
- command: args.serve.install,
3869
- cwd: SUBJECT_DIR,
3870
- timeoutMs: args.serve.installTimeoutMs ?? INSTALL_TIMEOUT_MS,
3871
- requestTimeoutMs: args.requestTimeoutMs,
3872
- timers,
3873
- retryPhase: "install-retry",
3874
- retryMessage: "subject install",
3875
- onPhase: args.onPhase,
3876
- now
3877
- });
3878
- emitPhaseCompleted(args.onPhase, now, installStartedAt, "install", install.ok, install.ok
3879
- ? install.attempts === 2
3880
- ? "subject dependencies installed (on the second attempt)"
3881
- : "subject dependencies installed"
3882
- : install.attempts === 2
3883
- ? "subject install failed twice"
3884
- : "subject install failed");
3885
- if (!install.ok) {
3886
- // Lead with the line a person can act on; npm's own trace follows it (#602).
3887
- const headline = install.timedOut
3888
- ? `subject install timed out after ${args.serve.installTimeoutMs ?? INSTALL_TIMEOUT_MS}ms`
3889
- : install.attempts === 2
3890
- ? `subject install failed twice (exit ${install.firstExitCode ?? "null"}, then exit ${install.exitCode ?? "null"}); the sandbox could not complete serve.install`
3891
- : `subject install failed (exit ${install.exitCode ?? "null"})`;
3892
- throw new Error(`${headline}: ${tailOf(args.scrub(install.logTail))}`);
3893
- }
3894
- await refresh();
3895
- }
3896
- // before-build: after install, before build (builds that read seeded state, e.g. SSG).
3897
- // When no build is declared this simply precedes start: equivalent to before-start.
3898
- await runStateSteps("before-build");
3899
- await refresh();
3900
- if (args.serve.build) {
3901
- const buildStartedAt = now();
3902
- emitPhaseStarted(args.onPhase, now, "build", "building subject");
3903
- const build = await runDetachedStep(desktop, {
3904
- name: "subject-build",
3905
- command: args.serve.build,
3906
- cwd: SUBJECT_DIR,
3907
- timeoutMs: args.serve.buildTimeoutMs ?? BUILD_TIMEOUT_MS,
3908
- requestTimeoutMs: args.requestTimeoutMs,
3909
- ...timers
3910
- });
3911
- emitPhaseCompleted(args.onPhase, now, buildStartedAt, "build", build.ok, build.ok ? "subject build complete" : "subject build failed");
3912
- if (!build.ok) {
3913
- throw new Error(`subject build ${build.timedOut ? "timed out" : `failed (exit ${build.exitCode})`}: ${tailOf(args.scrub(build.logTail))}`);
3914
- }
3915
- await refresh();
3916
- }
3917
- // before-start (the default phase): migrations, SQL/file fixtures, an in-sandbox DB server
3918
- // (`sudo service postgresql start && pg_isready` is a bounded step; the daemon it forks is
3919
- // reclaimed by the sandbox lifecycle like everything else).
3920
- await runStateSteps("before-start");
3921
- await refresh();
3922
- await startDetachedProcess(desktop, {
3923
- name: "subject-start",
3924
- command: args.serve.start,
3925
- cwd: SUBJECT_DIR,
3926
- requestTimeoutMs: args.requestTimeoutMs
3927
- });
3928
- // Fire-and-forget: startDetachedProcess never waits for the long-lived server to exit, so
3929
- // there is no matching completed event here (no ok/durationMs to report yet); readiness is
3930
- // the next boundary.
3931
- args.onPhase?.({ at: isoNow(now), type: "cua-lab.subject.serve.started", message: "subject server launched (detached)" });
3932
- const readyStartedAt = now();
3933
- emitPhaseStarted(args.onPhase, now, "ready", "waiting for subject to become ready");
3934
- const ready = await probeUrl(desktop, args.serve.url, {
3935
- timeoutMs: args.serve.readyTimeoutMs ?? DEFAULT_READY_TIMEOUT_MS,
3936
- requestTimeoutMs: args.requestTimeoutMs,
3937
- ...timers
3938
- });
3939
- emitPhaseCompleted(args.onPhase, now, readyStartedAt, "ready", ready, ready ? "subject is ready" : "subject did not become ready in time");
3940
- if (!ready) {
3941
- const startLog = await readDetachedLog(desktop, "subject-start", args.requestTimeoutMs).catch(() => "");
3942
- throw new Error(`subject did not answer at ${args.serve.url} within ${args.serve.readyTimeoutMs ?? DEFAULT_READY_TIMEOUT_MS}ms; server log tail: ${tailOf(args.scrub(startLog))}`);
3943
- }
3944
- // after-ready: fixture loading through the RUNNING app (loopback curl from in-sandbox:
3945
- // steps are author-trusted provisioning, not actors, so no new URL policy surface). These
3946
- // complete before the caller opens the browser and the session timer starts.
3947
- await runStateSteps("after-ready");
3948
- await refresh();
3949
- }
3950
- /**
3951
- * Provision a clone subject inside the sandbox: clone → the shared serve pipeline
3952
- * (install → state(before-build) → build → state(before-start) → start → readiness
3953
- * probe → state(after-ready)). Returns the latest subject HEAD after successful
3954
- * provisioning. Throws (with a capped log tail for the caller to redact) on any failing step:
3955
- * the lab persists that as a failed-evidence bundle.
3956
- *
3957
- * Auth: when GITHUB_TOKEN is among the declared subject env names, the clone authenticates
3958
- * via an Authorization header computed IN-SANDBOX from the provisioned env: the token never
3959
- * appears in the script text, the process argv beyond the transient git call, the clone URL,
3960
- * or .git/config.
3961
- */
3962
- export async function provisionCloneSubject(desktop, args) {
3963
- const timers = {
3964
- ...(args.now === undefined ? {} : { now: args.now }),
3965
- ...(args.sleep === undefined ? {} : { sleep: args.sleep })
3966
- };
3967
- const now = args.now ?? Date.now;
3968
- let latestCommit;
3969
- const refreshCommit = async () => {
3970
- const head = await desktop.commands.run(`git -C ${SUBJECT_DIR} rev-parse HEAD 2>/dev/null || true`, { requestTimeoutMs: args.requestTimeoutMs });
3971
- const commit = (head.stdout ?? "").trim() || undefined;
3972
- if (commit) {
3973
- latestCommit = commit;
3974
- args.onCommit?.(commit);
3975
- }
3976
- };
3977
- const cloneCommand = args.hasGithubToken
3978
- ? `auth=$(printf 'x-access-token:%s' "$GITHUB_TOKEN" | base64 -w0) && git -c http.extraHeader="Authorization: Basic $auth" clone --depth ${args.depth} https://github.com/${args.repo}.git ${SUBJECT_DIR}`
3979
- : `git clone --depth ${args.depth} https://github.com/${args.repo}.git ${SUBJECT_DIR}`;
3980
- const cloneStartedAt = now();
3981
- emitPhaseStarted(args.onPhase, now, "clone", "cloning subject repository");
3982
- const clone = await runDetachedStep(desktop, {
3983
- name: "subject-clone",
3984
- command: cloneCommand,
3985
- timeoutMs: CLONE_TIMEOUT_MS,
3986
- requestTimeoutMs: args.requestTimeoutMs,
3987
- ...timers
3988
- });
3989
- emitPhaseCompleted(args.onPhase, now, cloneStartedAt, "clone", clone.ok, clone.ok ? "subject repository cloned" : "subject clone failed");
3990
- if (!clone.ok) {
3991
- throw new Error(`subject clone ${clone.timedOut ? "timed out" : `failed (exit ${clone.exitCode})`}: ${tailOf(args.scrub(clone.logTail))}`);
3992
- }
3993
- await refreshCommit();
3994
- await runSubjectServePipeline(desktop, {
3995
- serve: args.serve,
3996
- ...(args.state === undefined ? {} : { state: args.state }),
3997
- requestTimeoutMs: args.requestTimeoutMs,
3998
- scrub: args.scrub,
3999
- ...(args.onStateStep === undefined ? {} : { onStateStep: args.onStateStep }),
4000
- ...(args.onPhase === undefined ? {} : { onPhase: args.onPhase }),
4001
- onPhaseComplete: refreshCommit,
4002
- ...timers
4003
- });
4004
- return latestCommit;
4005
- }
4006
- /**
4007
- * Provision a local-tree subject inside the sandbox: upload the once-per-run packed archive
4008
- * (identical bytes across every fan-out lane) → extract it into SUBJECT_DIR → the
4009
- * same shared serve pipeline provisionCloneSubject uses. Unlike the clone route there is no
4010
- * in-sandbox git refresh: the archive excludes .git entirely (see source-archive.ts), so
4011
- * subject identity is the host-side LocalTreeArchive captured at pack time, never anything
4012
- * resolved in-sandbox.
4013
- */
4014
- export async function provisionLocalTreeSubject(desktop, args) {
4015
- const timers = {
4016
- ...(args.now === undefined ? {} : { now: args.now }),
4017
- ...(args.sleep === undefined ? {} : { sleep: args.sleep })
4018
- };
4019
- const now = args.now ?? Date.now;
4020
- const uploadStartedAt = now();
4021
- emitPhaseStarted(args.onPhase, now, "upload", "uploading packed local-tree archive");
4022
- try {
4023
- await withOneRetryOnTransientE2BError(() => desktop.files.write(LOCAL_TREE_REMOTE_ARCHIVE_PATH, args.archiveBuffer, {
4024
- requestTimeoutMs: args.requestTimeoutMs,
4025
- useOctetStream: true
4026
- }), {
4027
- onRetry: (reason) => emitPhaseStarted(args.onPhase, now, "upload-retry", `local-tree archive upload retried once (${tailOf(args.scrub(reason))})`),
4028
- ...(args.sleep === undefined ? {} : { sleep: args.sleep })
4029
- });
4030
- }
4031
- catch (error) {
4032
- emitPhaseCompleted(args.onPhase, now, uploadStartedAt, "upload", false, "local-tree archive upload failed");
4033
- throw new Error(`subject-upload failed: ${tailOf(args.scrub(toErrorMessage(error)))}`);
4034
- }
4035
- emitPhaseCompleted(args.onPhase, now, uploadStartedAt, "upload", true, "local-tree archive uploaded");
4036
- const extractCommand = `rm -rf ${SUBJECT_DIR} && mkdir -p ${SUBJECT_DIR} && tar -xzf ${LOCAL_TREE_REMOTE_ARCHIVE_PATH} -C ${SUBJECT_DIR} && rm -f ${LOCAL_TREE_REMOTE_ARCHIVE_PATH}`;
4037
- const extractStartedAt = now();
4038
- emitPhaseStarted(args.onPhase, now, "extract", "extracting local-tree archive");
4039
- const extract = await runDetachedStep(desktop, {
4040
- name: "subject-extract",
4041
- command: extractCommand,
4042
- timeoutMs: CLONE_TIMEOUT_MS,
4043
- requestTimeoutMs: args.requestTimeoutMs,
4044
- ...timers
4045
- });
4046
- emitPhaseCompleted(args.onPhase, now, extractStartedAt, "extract", extract.ok, extract.ok ? "local-tree archive extracted" : "local-tree archive extraction failed");
4047
- if (!extract.ok) {
4048
- throw new Error(`subject extract ${extract.timedOut ? "timed out" : `failed (exit ${extract.exitCode})`}: ${tailOf(args.scrub(extract.logTail))}`);
4049
- }
4050
- await runSubjectServePipeline(desktop, {
4051
- serve: args.serve,
4052
- ...(args.state === undefined ? {} : { state: args.state }),
4053
- requestTimeoutMs: args.requestTimeoutMs,
4054
- scrub: args.scrub,
4055
- ...(args.onStateStep === undefined ? {} : { onStateStep: args.onStateStep }),
4056
- ...(args.onPhase === undefined ? {} : { onPhase: args.onPhase }),
4057
- ...timers
4058
- });
4059
- }
4060
2264
  /**
4061
2265
  * Default local-tree packing implementation: createLocalTreeArchive(root, opts) on the host,
4062
2266
  * then a single read of the produced archive file into an ArrayBuffer for upload. The DI seam
@@ -4076,10 +2280,6 @@ export async function defaultPackLocalTree(args) {
4076
2280
  await rm(path.dirname(archive.archivePath), { recursive: true, force: true }).catch(() => undefined);
4077
2281
  return { archive, buffer };
4078
2282
  }
4079
- /** sha256 hex of the exact command string, first 16 chars (the promptDigest convention). */
4080
- export function commandDigestOf(command) {
4081
- return digestText(command, 16);
4082
- }
4083
2283
  /**
4084
2284
  * Resolve the bundle's state marker from the declaration and what actually ran.
4085
2285
  * Precedence: external declared → "unpinned" (seed records, if any, stay attached — a
@@ -4135,10 +2335,6 @@ function describeSubjectState(state, dryRun) {
4135
2335
  return "external-public (operator-declared, operator-owned public deployment; neither provisioned nor seeded)";
4136
2336
  }
4137
2337
  }
4138
- // The in-sandbox `tail -c` upstream is a fundamental log-tail limit we cannot redact past.
4139
- function tailOf(log) {
4140
- return redactedTail(log, ERROR_TAIL_CHARS);
4141
- }
4142
2338
  export function buildCuaCostSummary(args) {
4143
2339
  const breakdown = [];
4144
2340
  let sumInput = 0;