humanish 0.96.1 → 0.98.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (193) hide show
  1. package/AGENTS.md +86 -79
  2. package/CONTRIBUTING.md +7 -2
  3. package/README.md +11 -2
  4. package/dist/actor-contract.d.ts +35 -1
  5. package/dist/actor-contract.js +38 -0
  6. package/dist/actor-contract.js.map +1 -1
  7. package/dist/adapter-extension.js +1 -0
  8. package/dist/adapter-extension.js.map +1 -1
  9. package/dist/automatic-analysis-config.d.ts +15 -5
  10. package/dist/automatic-analysis-config.js +25 -4
  11. package/dist/automatic-analysis-config.js.map +1 -1
  12. package/dist/automatic-study-analysis.js +4 -2
  13. package/dist/automatic-study-analysis.js.map +1 -1
  14. package/dist/browser-control-client.d.ts +14 -0
  15. package/dist/browser-control-client.js +134 -0
  16. package/dist/browser-control-client.js.map +1 -0
  17. package/dist/browser-control-dispatcher.d.ts +14 -0
  18. package/dist/browser-control-dispatcher.js +109 -0
  19. package/dist/browser-control-dispatcher.js.map +1 -0
  20. package/dist/browser-control-protocol.d.ts +371 -0
  21. package/dist/browser-control-protocol.js +155 -0
  22. package/dist/browser-control-protocol.js.map +1 -0
  23. package/dist/browser-control-transport.d.ts +24 -0
  24. package/dist/browser-control-transport.js +156 -0
  25. package/dist/browser-control-transport.js.map +1 -0
  26. package/dist/comms-lease-store.d.ts +1 -0
  27. package/dist/comms-lease-store.js +9 -3
  28. package/dist/comms-lease-store.js.map +1 -1
  29. package/dist/computer-use-actor.d.ts +2 -2
  30. package/dist/computer-use-actor.js +6 -1
  31. package/dist/computer-use-actor.js.map +1 -1
  32. package/dist/computer-use.d.ts +23 -1
  33. package/dist/computer-use.js +253 -70
  34. package/dist/computer-use.js.map +1 -1
  35. package/dist/cua-actor-lab.d.ts +24 -289
  36. package/dist/cua-actor-lab.js +203 -1983
  37. package/dist/cua-actor-lab.js.map +1 -1
  38. package/dist/cua-desktop-lane.d.ts +35 -0
  39. package/dist/cua-desktop-lane.js +13 -0
  40. package/dist/cua-desktop-lane.js.map +1 -0
  41. package/dist/cua-executor-error.d.ts +31 -0
  42. package/dist/cua-executor-error.js +48 -0
  43. package/dist/cua-executor-error.js.map +1 -0
  44. package/dist/cua-provider-error.d.ts +12 -0
  45. package/dist/cua-provider-error.js +28 -0
  46. package/dist/cua-provider-error.js.map +1 -0
  47. package/dist/desktop-session.d.ts +41 -0
  48. package/dist/desktop-session.js +46 -0
  49. package/dist/desktop-session.js.map +1 -0
  50. package/dist/doctor-lab.d.ts +8 -1
  51. package/dist/doctor-lab.js +40 -8
  52. package/dist/doctor-lab.js.map +1 -1
  53. package/dist/e2b-cua-desktop.d.ts +3 -0
  54. package/dist/e2b-cua-desktop.js +675 -0
  55. package/dist/e2b-cua-desktop.js.map +1 -0
  56. package/dist/e2b-cua-provisioning.d.ts +311 -0
  57. package/dist/e2b-cua-provisioning.js +1213 -0
  58. package/dist/e2b-cua-provisioning.js.map +1 -0
  59. package/dist/e2b-desktop-executor.d.ts +1 -24
  60. package/dist/e2b-desktop-executor.js +2 -127
  61. package/dist/e2b-desktop-executor.js.map +1 -1
  62. package/dist/e2b-desktop-session.d.ts +7 -0
  63. package/dist/e2b-desktop-session.js +29 -0
  64. package/dist/e2b-desktop-session.js.map +1 -0
  65. package/dist/e2b-terminal-lab.js +1 -0
  66. package/dist/e2b-terminal-lab.js.map +1 -1
  67. package/dist/frame-signature.d.ts +24 -0
  68. package/dist/frame-signature.js +128 -0
  69. package/dist/frame-signature.js.map +1 -0
  70. package/dist/guest-bootstrap.d.ts +43 -0
  71. package/dist/guest-bootstrap.js +240 -0
  72. package/dist/guest-bootstrap.js.map +1 -0
  73. package/dist/guest-browser-tools.d.ts +8 -0
  74. package/dist/guest-browser-tools.js +66 -0
  75. package/dist/guest-browser-tools.js.map +1 -0
  76. package/dist/guest-chromium-text.d.ts +27 -0
  77. package/dist/guest-chromium-text.js +281 -0
  78. package/dist/guest-chromium-text.js.map +1 -0
  79. package/dist/guest-desktop-executor.d.ts +22 -0
  80. package/dist/guest-desktop-executor.js +177 -0
  81. package/dist/guest-desktop-executor.js.map +1 -0
  82. package/dist/guest-desktop-native.d.ts +14 -0
  83. package/dist/guest-desktop-native.js +131 -0
  84. package/dist/guest-desktop-native.js.map +1 -0
  85. package/dist/guest-runtime-desktop.d.ts +35 -0
  86. package/dist/guest-runtime-desktop.js +231 -0
  87. package/dist/guest-runtime-desktop.js.map +1 -0
  88. package/dist/guest-runtime-main.d.ts +1 -0
  89. package/dist/guest-runtime-main.js +31 -0
  90. package/dist/guest-runtime-main.js.map +1 -0
  91. package/dist/guest-runtime-revision.d.ts +1 -0
  92. package/dist/guest-runtime-revision.js +3 -0
  93. package/dist/guest-runtime-revision.js.map +1 -0
  94. package/dist/guest-runtime.d.ts +25 -0
  95. package/dist/guest-runtime.js +96 -0
  96. package/dist/guest-runtime.js.map +1 -0
  97. package/dist/index.d.ts +1 -1
  98. package/dist/lab-config.js +10 -3
  99. package/dist/lab-config.js.map +1 -1
  100. package/dist/lab-engine.js +6 -0
  101. package/dist/lab-engine.js.map +1 -1
  102. package/dist/lab-summary.d.ts +5 -1
  103. package/dist/lab-summary.js +5 -0
  104. package/dist/lab-summary.js.map +1 -1
  105. package/dist/local-agent-cli.js +1 -1
  106. package/dist/local-agent-cli.js.map +1 -1
  107. package/dist/local-firecracker-desktop.d.ts +13 -0
  108. package/dist/local-firecracker-desktop.js +150 -0
  109. package/dist/local-firecracker-desktop.js.map +1 -0
  110. package/dist/local-firecracker-study.d.ts +9 -0
  111. package/dist/local-firecracker-study.js +93 -0
  112. package/dist/local-firecracker-study.js.map +1 -0
  113. package/dist/local-runtime-config.d.ts +6 -0
  114. package/dist/local-runtime-config.js +56 -0
  115. package/dist/local-runtime-config.js.map +1 -0
  116. package/dist/local-runtime-release.d.ts +3 -0
  117. package/dist/local-runtime-release.js +8 -0
  118. package/dist/local-runtime-release.js.map +1 -0
  119. package/dist/local-runtime.d.ts +25 -0
  120. package/dist/local-runtime.js +113 -0
  121. package/dist/local-runtime.js.map +1 -0
  122. package/dist/observer-app.html +4 -4
  123. package/dist/pricing.d.ts +22 -1
  124. package/dist/pricing.js +22 -0
  125. package/dist/pricing.js.map +1 -1
  126. package/dist/program.js +50 -9
  127. package/dist/program.js.map +1 -1
  128. package/dist/restricted-codex-analysis.d.ts +15 -0
  129. package/dist/restricted-codex-analysis.js +13 -0
  130. package/dist/restricted-codex-analysis.js.map +1 -0
  131. package/dist/restricted-codex-participant-policy.d.ts +39 -0
  132. package/dist/restricted-codex-participant-policy.js +69 -0
  133. package/dist/restricted-codex-participant-policy.js.map +1 -0
  134. package/dist/restricted-codex-participant-run.d.ts +20 -0
  135. package/dist/restricted-codex-participant-run.js +78 -0
  136. package/dist/restricted-codex-participant-run.js.map +1 -0
  137. package/dist/restricted-codex-participant.d.ts +14 -0
  138. package/dist/restricted-codex-participant.js +178 -0
  139. package/dist/restricted-codex-participant.js.map +1 -0
  140. package/dist/restricted-codex-policy.d.ts +56 -0
  141. package/dist/restricted-codex-policy.js +151 -0
  142. package/dist/restricted-codex-policy.js.map +1 -0
  143. package/dist/restricted-codex-session.d.ts +19 -0
  144. package/dist/restricted-codex-session.js +413 -0
  145. package/dist/restricted-codex-session.js.map +1 -0
  146. package/dist/restricted-codex-transport.d.ts +58 -0
  147. package/dist/restricted-codex-transport.js +233 -0
  148. package/dist/restricted-codex-transport.js.map +1 -0
  149. package/dist/run-detail.js +4 -2
  150. package/dist/run-detail.js.map +1 -1
  151. package/dist/run.d.ts +12 -5
  152. package/dist/run.js +17 -1
  153. package/dist/run.js.map +1 -1
  154. package/dist/shared-world-lab.js +2 -2
  155. package/dist/shared-world-lab.js.map +1 -1
  156. package/dist/study-analysis-codex-config.d.ts +11 -0
  157. package/dist/study-analysis-codex-config.js +34 -0
  158. package/dist/study-analysis-codex-config.js.map +1 -0
  159. package/dist/study-analysis-engine.d.ts +6 -3
  160. package/dist/study-analysis-engine.js +19 -10
  161. package/dist/study-analysis-engine.js.map +1 -1
  162. package/dist/study-analysis-job.d.ts +3 -2
  163. package/dist/study-analysis-job.js +1 -1
  164. package/dist/study-analysis-job.js.map +1 -1
  165. package/dist/study-analysis-provider.d.ts +4 -2
  166. package/dist/study-analysis-provider.js +1 -1
  167. package/dist/study-analysis-provider.js.map +1 -1
  168. package/dist/study-analysis-service.d.ts +3 -0
  169. package/dist/study-analysis-service.js +24 -6
  170. package/dist/study-analysis-service.js.map +1 -1
  171. package/dist/study-analysis-validation.d.ts +43 -19
  172. package/dist/study-analysis-validation.js +23 -10
  173. package/dist/study-analysis-validation.js.map +1 -1
  174. package/dist/study-analysis.d.ts +27 -2
  175. package/dist/study-costs.js +6 -0
  176. package/dist/study-costs.js.map +1 -1
  177. package/dist/tui-app.js +102 -102
  178. package/docs/architecture/browser-control.md +117 -0
  179. package/docs/architecture/desktop-sessions.md +80 -0
  180. package/docs/architecture/guest-desktop.md +87 -0
  181. package/docs/architecture/local-browser-runtime.md +100 -0
  182. package/docs/architecture/restricted-codex-analysis.md +91 -0
  183. package/docs/architecture/runtime-broker-core.md +30 -0
  184. package/docs/contracts/schemas.md +1 -1
  185. package/docs/contracts/study-analysis.md +44 -2
  186. package/docs/goals/current.md +25 -9
  187. package/docs/product/automatic-analysis.md +24 -3
  188. package/docs/product/open-source-install-experience.md +7 -0
  189. package/docs/ramp/README.md +30 -13
  190. package/docs/release/0.97.0-codex-account-analysis.md +45 -0
  191. package/docs/release/0.98.0-local-browser-studies.md +31 -0
  192. package/package.json +4 -2
  193. package/skills/humanish/SKILL.md +39 -0
@@ -1,7 +1,12 @@
1
- import { withTransientCommsSecrets } from "./run-narration-secrets.js";
1
+ export { inboxRecipientFor, laneHasInboxRecipient } from "./cua-desktop-lane.js";
2
+ export { CUA_ACTOR_LAB_PROVIDER_METADATA, DEFAULT_MOBILE_USER_AGENT, SANDBOX_CAMERA_PATH, SANDBOX_MEDIA_DIR, SUBJECT_DIR, SYNTHETIC_CAMERA_COMMAND, applyMobileEmulation, buildFillDesktopWindowCommand, captureDesktopBrowserGeometry, commandDigestOf, declaredScreenForRender, desktopBrowserFamily, inspectDesktopScreenGeometry, makeChromeBrowserStateObserver, makeChromeDesktopGeometryObserver, parseXwininfoGeometry, prepareDesktopMedia, provisionCloneSubject, provisionLocalTreeSubject } from "./e2b-cua-provisioning.js";
2
3
  import { prepareReceivingRun, receivingPublication } from "./comms-receiving-runtime.js";
3
- import { deployReceivingInbox } from "./comms-receiving-inbox.js";
4
+ import { laneHasInboxRecipient } from "./cua-desktop-lane.js";
5
+ import { createE2BCuaDesktopLane } from "./e2b-cua-desktop.js";
6
+ import { isLocalBrowserLab, LOCAL_BROWSER_LIFETIME_MS } from "./local-runtime-config.js";
7
+ import { DEFAULT_STATE_STEP_TIMEOUT_MS, commandDigestOf, declaredScreenForRender } from "./e2b-cua-provisioning.js";
4
8
  import { receivingEmailValidationReason } from "./lab-config.js";
9
+ import { withTransientCommsSecrets } from "./run-narration-secrets.js";
5
10
  // The computer-use lab backend: a subject (an app-url the caller provisioned, or a repo the
6
11
  // lab clones AND serves in-sandbox) driven by a REGISTRY-RESOLVED computer-use actor inside a
7
12
  // hosted E2B desktop. This is the path that makes `actors[].type` load-bearing — the
@@ -24,52 +29,47 @@ import { receivingEmailValidationReason } from "./lab-config.js";
24
29
  // harness errors are redacted at THIS boundary; the bundle's `stream.actor` carries the
25
30
  // conformant humanish.actor-trace.v1 projection, whose `redaction.screenshots` records the
26
31
  // run's actual mode ("raw" | "blurred" | "n/a") — every label downstream derives from it.
27
- import { resolveAutomaticAnalysis } from "./automatic-analysis-config.js";
28
- import { completeAutomaticAnalysis, markFinalizedStudyResult } from "./automatic-analysis-completion.js";
29
- import { desktopMediaValidationReason, taskProtocolValidationReason } from "./lab-config.js";
30
32
  import { randomBytes } from "node:crypto";
31
- import { describeMissingKeys } from "./key-resolution.js";
32
33
  import { readFile, realpath, rm } from "node:fs/promises";
33
34
  import path from "node:path";
35
+ import { completeAutomaticAnalysis, markFinalizedStudyResult } from "./automatic-analysis-completion.js";
36
+ import { resolveAutomaticAnalysis } from "./automatic-analysis-config.js";
37
+ import { describeMissingKeys } from "./key-resolution.js";
38
+ import { desktopMediaValidationReason, taskProtocolValidationReason } from "./lab-config.js";
39
+ import { pathToFileURL } from "node:url";
40
+ import { toErrorMessage } from "./command-failure.js";
34
41
  import { cuaLaneDiagnostics, summarizeCuaDiagnostics } from "./cua-diagnostics.js";
35
42
  import { feedbackProofCommands } from "./feedback-proof.js";
36
- import { runDesktopCommandOrThrow, toErrorMessage } from "./command-failure.js";
37
- import { pathToFileURL } from "node:url";
38
- import { beginRunStatus, withRunStatusScope } from "./run-status.js";
39
- import { adapterScoreFailureMessage, applyBrowserAdapterHooks } from "./adapter-extension.js";
40
43
  import { actorRegistry, isCuaActorDescriptor } from "./actor-registry.js";
41
- import { CHROMIUM_EVIDENCE_HYGIENE_FLAGS, chromiumEvidenceProfilePreferencesJson } from "./browser-evidence-hygiene.js";
42
- import { DEFAULT_OPENAI_CU_MODEL } from "./openai-responses-cu.js";
43
- import { createLocalAgentProvider, detectLocalAgents } from "./local-agent-cli.js";
44
- import { startAppServerSession } from "./local-agent-appserver.js";
45
- import { startClaudeSession } from "./local-agent-claude-session.js";
46
- import { createDesktopSandbox, withOneRetryOnTransientE2BError, loadE2BDesktopModule } from "./e2b-desktop-launch.js";
47
- import { probeUrl, readDetachedLog, runDetachedStep, startDetachedProcess } from "./e2b-detached.js";
48
- import { DEFAULT_SANDBOX_CATCH_PORT, collectCommsThread, collectExternalCommsThread, deployCommsCatch, externalCatchHealthy, externalInboxUrl, refreshInboxSurface, writeInboxSurface } from "./comms-sandbox-catch.js";
44
+ import { actorEnding } from "./actor-stop-cause.js";
45
+ import { adapterScoreFailureMessage, applyBrowserAdapterHooks } from "./adapter-extension.js";
49
46
  import { FakeInbox } from "./comms-fake-inbox.js";
50
- import { buildOriginMap, recipientInboxUrl } from "./comms-inbox.js";
51
- import { DEFAULT_DEVICE_PRESET, isDevicePresetName, resolveDevicePreset } from "./device-presets.js";
52
- import { cuaLaneValidationReason, outputTokenLimitValidationReason, isHttpUrl, isLoopbackUrl, MAX_CUA_LANES, subjectStateInvalidReason } from "./lab-config.js";
47
+ import { recipientInboxUrl } from "./comms-inbox.js";
48
+ import { collectExternalCommsThread, externalCatchHealthy, externalInboxUrl } from "./comms-sandbox-catch.js";
53
49
  import { mapWithConcurrency } from "./concurrency.js";
54
- import { appendSandboxReceipt } from "./sandbox-receipts.js";
50
+ import { DEFAULT_DEVICE_PRESET, isDevicePresetName, resolveDevicePreset } from "./device-presets.js";
51
+ import {} from "./e2b-desktop-launch.js";
52
+ import {} from "./e2b-desktop-resources.js";
53
+ import {} from "./e2b-detached.js";
55
54
  import { assertScreenshotEvidence } from "./image-evidence.js";
55
+ import { MAX_CUA_LANES, cuaLaneValidationReason, isHttpUrl, isLoopbackUrl, outputTokenLimitValidationReason, subjectStateInvalidReason } from "./lab-config.js";
56
+ import { startAppServerSession } from "./local-agent-appserver.js";
57
+ import { startClaudeSession } from "./local-agent-claude-session.js";
58
+ import { createLocalAgentProvider, detectLocalAgents } from "./local-agent-cli.js";
56
59
  import { buildObserverData } from "./observer-data.js";
57
- import { corepackCommandFor, needsNodeRuntime, nodeBootstrapCommand } from "./subject-runtime.js";
58
- import { TERMINAL_NODE_BOOTSTRAP_COMMAND } from "./terminal-node-bootstrap.js";
59
- import { chromeCdpProbeCommand, parseChromeCdpProbeOutput } from "./chrome-cdp-probe.js";
60
- import { personaToDirectives, renderPersonaPromptSection } from "./persona.js";
61
- import { labPersonaIds, resolveCommittedPersonas } from "./persona-resolve.js";
62
- import { renderTaskPrompt } from "./tasks.js";
63
60
  import { attachObserverRuntimeStreamUrls, renderObserver } from "./observer.js";
64
- import { containsSensitive, digestText, redactedTail, redactText } from "./redaction.js";
61
+ import { DEFAULT_OPENAI_CU_MODEL } from "./openai-responses-cu.js";
65
62
  import { participantAssignment } from "./participant-assignment.js";
66
- import { actorEnding } from "./actor-stop-cause.js";
67
- import { assertPreparedSelectedOutputDirectory, assertSafeOutputPathSegment, prepareContainedOutputDirectory, prepareSelectedOutputDirectory, writeContainedOutputFile, writePreparedRunLatestPointer } from "./selected-output-paths.js";
63
+ import { labPersonaIds, resolveCommittedPersonas } from "./persona-resolve.js";
64
+ import { personaToDirectives, renderPersonaPromptSection } from "./persona.js";
65
+ import { MODEL_RATES, estimateActorCost, estimateActorCostForExecution, estimateAllocatedDesktopCost, estimateDesktopCost, round6 } from "./pricing.js";
66
+ import { containsSensitive, digestText, redactText } from "./redaction.js";
68
67
  import { prepareRunArtifactPaths, validatePreparedRunArtifactPaths } from "./run-paths.js";
68
+ import { beginRunStatus, withRunStatusScope } from "./run-status.js";
69
+ import { PUBLIC_TARGET_CWD, REVIEW_SCHEMA, RUN_BUNDLE_SCHEMA, aggregateTaskFunnels, buildRunSource, formatParticipantOutcomes, formatStudyTaskFunnel, loadRunBundle, tallyParticipantOutcomes, withCuaReviewProvenance } from "./run.js";
70
+ import { assertPreparedSelectedOutputDirectory, assertSafeOutputPathSegment, prepareContainedOutputDirectory, prepareSelectedOutputDirectory, writeContainedOutputFile, writePreparedRunLatestPointer } from "./selected-output-paths.js";
69
71
  import { createLocalTreeArchive } from "./source-archive.js";
70
- import { buildRunSource, loadRunBundle, PUBLIC_TARGET_CWD, REVIEW_SCHEMA, RUN_BUNDLE_SCHEMA, aggregateTaskFunnels, formatParticipantOutcomes, formatStudyTaskFunnel, tallyParticipantOutcomes, withCuaReviewProvenance } from "./run.js";
71
- import { estimateActorCost, estimateDesktopCost, estimateAllocatedDesktopCost, MODEL_RATES, round6 } from "./pricing.js";
72
- import { observeDesktopResources } from "./e2b-desktop-resources.js";
72
+ import { renderTaskPrompt } from "./tasks.js";
73
73
  export const CUA_ACTOR_LAB_SCHEMA = "humanish.cua-lab-result.v2";
74
74
  // The only fan-out topology this slice ships: N lanes = N independent E2B desktop sandboxes,
75
75
  // each its own world (clone/serve + subject.state per lane). Shared-world is layer 7 (#164).
@@ -77,10 +77,6 @@ export const CUA_FANOUT_STRATEGY = "per-lane-worlds";
77
77
  // Env override that may only LOWER the effective concurrency (never raise concurrent paid
78
78
  // desktops — invariant 3). Read names-only into a local; the value never persists.
79
79
  const CUA_MAX_CONCURRENCY_ENV = "HUMANISH_CUA_MAX_CONCURRENCY";
80
- export const CUA_ACTOR_LAB_PROVIDER_METADATA = {
81
- mode: "cua-actor-lab",
82
- tool: "humanish"
83
- };
84
80
  // The DEFAULT session budget, sized so a study can FINISH (docs/principles/three-roles.md: a
85
81
  // session ends because the participant is done, not because a timer fired — the time-box is a
86
82
  // session-level cap a researcher sets generously; spend protection is the dollar caps' job).
@@ -103,60 +99,6 @@ function defaultSessionTimeoutMs(config) {
103
99
  const room = MAX_SANDBOX_MS - SUBJECT_PROVISION_BUDGET_MS - stateBudgetMs - SANDBOX_TIMEOUT_BUFFER_MS;
104
100
  return Math.max(MIN_DERIVED_SESSION_TIMEOUT_MS, Math.min(DEFAULT_APP_URL_SESSION_TIMEOUT_MS, room));
105
101
  }
106
- // Settle after opening the browser, before the first screenshot — long enough for a cold
107
- // browser + page load to paint (2s captured a blank desktop; the render empirically needs ~6-9s).
108
- const BROWSER_SETTLE_MS = 8_000;
109
- /** Where a lane's synthetic camera feed lives inside the sandbox: a tmpfs the sandbox user can
110
- * write, and a path that contains neither /tmp/ nor /home/, which the public-safety scan reads
111
- * as an operator's local path (this one is the harness's own and belongs in the bundle). */
112
- export const SANDBOX_MEDIA_DIR = "/dev/shm/humanish-media";
113
- export const SANDBOX_CAMERA_PATH = `${SANDBOX_MEDIA_DIR}/camera.y4m`;
114
- /** The synthetic feed: ffmpeg's test pattern, 640x480 at 10 fps, six seconds (about 28 MB of
115
- * raw Y4M on the tmpfs), looped by Chrome's fake capture device. */
116
- export const SYNTHETIC_CAMERA_COMMAND = `mkdir -p ${SANDBOX_MEDIA_DIR} && ffmpeg -y -loglevel error -f lavfi -i testsrc=size=640x480:rate=10 -t 6 -pix_fmt yuv420p ${SANDBOX_CAMERA_PATH}`;
117
- /**
118
- * Put the declared camera feed in the sandbox and return the Chromium flags that present it as a
119
- * capture device (#509). Fails CLOSED: a feed that cannot be produced (no ffmpeg on the image, an
120
- * unreadable host file) is named before the browser launches, because a participant told it has
121
- * a camera and finds none reports the instrument's gap as the product's.
122
- */
123
- export async function prepareDesktopMedia(desktop, media, permission, cwd, requestTimeoutMs, readHostFile = (absolutePath) => readFile(absolutePath)) {
124
- if (media.microphone !== undefined) {
125
- throw new Error("execution.desktop.media.microphone.source injection is unsupported; the declared microphone file cannot be delivered, including on custom templates.");
126
- }
127
- const flags = [];
128
- let camera;
129
- if (media.camera !== undefined) {
130
- if (media.camera.source === "synthetic") {
131
- const made = await desktop.commands.run(SYNTHETIC_CAMERA_COMMAND, { requestTimeoutMs, timeoutMs: 60_000 });
132
- if (made.exitCode !== undefined && made.exitCode !== 0) {
133
- throw new Error(`the synthetic camera feed could not be generated on this desktop image (ffmpeg exited ${made.exitCode}: ${tailOf(made.stderr ?? made.stdout ?? "")}); give execution.desktop.media.camera.source a .y4m file instead`);
134
- }
135
- camera = { source: "synthetic", file: SANDBOX_CAMERA_PATH };
136
- }
137
- else {
138
- const absolutePath = path.resolve(cwd, media.camera.source);
139
- let bytes;
140
- try {
141
- bytes = await readHostFile(absolutePath);
142
- }
143
- catch (error) {
144
- throw new Error(`execution.desktop.media.camera.source could not be read (${toErrorMessage(error)})`);
145
- }
146
- if (bytes.length > 64 * 1024 * 1024) {
147
- throw new Error(`execution.desktop.media.camera.source is ${bytes.length} bytes; the camera feed is capped at 64 MiB`);
148
- }
149
- await desktop.commands.run(`mkdir -p ${SANDBOX_MEDIA_DIR}`, { requestTimeoutMs, timeoutMs: 15_000 });
150
- const payload = bytes.buffer.slice(bytes.byteOffset, bytes.byteOffset + bytes.byteLength);
151
- await desktop.files.write(SANDBOX_CAMERA_PATH, payload, { requestTimeoutMs, useOctetStream: true });
152
- camera = { source: "file", file: SANDBOX_CAMERA_PATH };
153
- }
154
- flags.push("--use-fake-device-for-media-stream", `--use-file-for-fake-video-capture=${SANDBOX_CAMERA_PATH}`);
155
- }
156
- if (permission === "granted")
157
- flags.push("--use-fake-ui-for-media-stream");
158
- return { ...(camera === undefined ? {} : { camera }), permission, flags };
159
- }
160
102
  // Device/screen size comes from the named-preset registry (device-presets.ts), selectable per run
161
103
  // via execution.desktop.device (default `desktop`=1440x950). NOTE: this is run-wide for now; a
162
104
  // per-PERSONA device dimension (N personas × devices, as the bespoke sims author) lands with
@@ -172,19 +114,6 @@ const SUBJECT_PROVISION_BUDGET_MS = 30 * 60_000;
172
114
  * The derived per-lane deadline has to stay under it, and saying so at plan time beats discovering
173
115
  * it from a raw provider 400 after a plan has already printed. */
174
116
  const MAX_SANDBOX_MS = 60 * 60_000;
175
- export const SUBJECT_DIR = "/home/user/subject";
176
- // Remote path for the once-per-run packed local-tree archive; removed by the extract step
177
- // after it unpacks into SUBJECT_DIR.
178
- const LOCAL_TREE_REMOTE_ARCHIVE_PATH = "/home/user/.humanish-source.tar.gz";
179
- const CLONE_TIMEOUT_MS = 5 * 60_000;
180
- const INSTALL_TIMEOUT_MS = 10 * 60_000;
181
- const BUILD_TIMEOUT_MS = 10 * 60_000;
182
- const DEFAULT_READY_TIMEOUT_MS = 180_000;
183
- // Per-step budget for subject.state seed steps; each step's declared (or default) budget is
184
- // also summed into the default sandbox deadline so seeding never eats the session's room.
185
- const DEFAULT_STATE_STEP_TIMEOUT_MS = 5 * 60_000;
186
- // How much of a failing step's log tail rides the (redacted) error message.
187
- const ERROR_TAIL_CHARS = 2000;
188
117
  const DEFAULT_MISSION = "You are testing a web application. The browser is already open at the subject URL. Explore it, accomplish what the scenario asks, and stop when done.";
189
118
  /**
190
119
  * The participant's outcome as ONE fixed first line of its last message (#570, second half). The
@@ -262,21 +191,6 @@ export function withInboxMission(spec, inboxUrl, address, receiving = false) {
262
191
  instructions: `${spec.instructions}\n\nEmail inbox:${identity} When the app tells you it has emailed you (a verification link, confirmation code, or magic link), open ${recipientInboxUrl(inboxUrl, address)} in the browser to read that email and follow its link or enter its code. All email the app sends you arrives there. Waiting for an email is normal, not a blocker — do not end your session while waiting; open the inbox and refresh it until the email appears.`
263
192
  };
264
193
  }
265
- /** The lane's addressed comms recipient, when one exists — the gate AND the address source for the
266
- * inbox instruction (#351). A lane told to check an inbox it can never receive into would stall,
267
- * so no addressed recipient means no instruction. */
268
- export function inboxRecipientFor(commsEmail, laneId) {
269
- return (commsEmail.recipients ?? []).find((recipient) => recipient.lane === laneId && recipient.address !== undefined);
270
- }
271
- /** True when a lane has a declared comms recipient WITH an address, so the drain can actually match the
272
- * mail the persona will be told to read. Gates the inbox instruction to lanes that can receive mail —
273
- * a lane told to check an inbox it can never receive into would just stall. */
274
- export function laneHasInboxRecipient(commsEmail, laneId) {
275
- return inboxRecipientFor(commsEmail, laneId) !== undefined;
276
- }
277
- /** Mid-run inbox-surface render cadence (ms). Coarse enough that the per-tick `cat` + file writes stay
278
- * cheap; fine enough that a verification email is visible seconds after the app sends it. */
279
- const INBOX_SURFACE_CADENCE_MS = 2500;
280
194
  /**
281
195
  * The narrowest browser WINDOW Chrome/Chromium will render on the E2B desktop. Chrome refuses to
282
196
  * make its window narrower than this (~500 CSS px observed: a 414-wide X screen produced a 500-wide
@@ -292,20 +206,6 @@ export const MIN_DESKTOP_RENDER_WIDTH = 500;
292
206
  export function floorRenderResolution(resolution) {
293
207
  return [Math.max(resolution[0], MIN_DESKTOP_RENDER_WIDTH), resolution[1]];
294
208
  }
295
- /**
296
- * The DECLARED preset to record alongside the rendered screen, or undefined when the preset
297
- * rendered faithfully.
298
- *
299
- * `desktopGeometry.screen.verified` compares the FLOORED number with itself, so on its own a
300
- * floored run is indistinguishable from a faithful one: a reader sees requested 500 / verified 500
301
- * and concludes a 500-wide screen was asked for. Recording the declared preset is what makes
302
- * "the preset width did not render" legible in the bundle.
303
- */
304
- export function declaredScreenForRender(preset, presetName, rendered) {
305
- if (preset.width === rendered[0] && preset.height === rendered[1])
306
- return undefined;
307
- return { width: preset.width, height: preset.height, preset: presetName };
308
- }
309
209
  /**
310
210
  * Resolve a lane's device + rendered resolution (most-specific wins, exactly as the single-lane
311
211
  * path always has): a raw execution.desktop.resolution escape hatch (only legal when no lane
@@ -333,6 +233,8 @@ export function resolveLaneDevice(config, lane) {
333
233
  * git clone for an upload+extract, but the shared install/build/state/start/probe pipeline
334
234
  * costs the same wall-clock room either way. */
335
235
  function resolvePerLaneSandboxMs(config) {
236
+ if (isLocalBrowserLab(config))
237
+ return LOCAL_BROWSER_LIFETIME_MS;
336
238
  const timeoutMs = config.execution?.timeoutMs ?? defaultSessionTimeoutMs(config);
337
239
  const provisionedRoute = config.subject.source === "clone" || config.subject.source === "local-tree";
338
240
  const stateBudgetMs = provisionedRoute
@@ -582,65 +484,6 @@ function formatLanePlanEntry(lane) {
582
484
  ].filter((part) => part !== undefined);
583
485
  return `${lane.id}: persona=${lane.persona}${taxonomy.length > 0 ? ` ${taxonomy.join(" ")}` : ""} device=${lane.device} ${lane.resolution[0]}x${lane.resolution[1]} prompt#${lane.instructionDigest}${lane.targetDigest ? ` target#${lane.targetDigest}` : ""}`;
584
486
  }
585
- /** ISO timestamp from an injectable clock (tests freeze `now` for deterministic durationMs). */
586
- function isoNow(now) {
587
- return new Date(now()).toISOString();
588
- }
589
- /** Emit a phase-started event (no ok/durationMs: those belong to the matching completed event). */
590
- /**
591
- * Run a provisioning step and, when it fails with an EXIT CODE, run it once more (#602). A cold
592
- * install of 0.74.0 lost its whole first live study to one transient TLS error inside the
593
- * sandbox's `npm install`; the parallel install twenty seconds later passed, as had the ten
594
- * before it. One retry clears that class. A TIMEOUT is not retried: its budget is already spent,
595
- * and a second wait would double it. The retry runs under its own step name so both logs stay.
596
- */
597
- async function runProvisioningStepWithOneRetry(desktop, args) {
598
- const first = await runDetachedStep(desktop, {
599
- name: args.name,
600
- command: args.command,
601
- cwd: args.cwd,
602
- timeoutMs: args.timeoutMs,
603
- requestTimeoutMs: args.requestTimeoutMs,
604
- ...args.timers
605
- });
606
- if (first.ok || first.timedOut)
607
- return { ...first, attempts: 1 };
608
- const retryStartedAt = args.now();
609
- emitPhaseStarted(args.onPhase, args.now, args.retryPhase, `${args.retryMessage} (first attempt exited ${first.exitCode ?? "null"}; retrying once)`);
610
- const second = await runDetachedStep(desktop, {
611
- name: `${args.name}-retry`,
612
- command: args.command,
613
- cwd: args.cwd,
614
- timeoutMs: args.timeoutMs,
615
- requestTimeoutMs: args.requestTimeoutMs,
616
- ...args.timers
617
- });
618
- emitPhaseCompleted(args.onPhase, args.now, retryStartedAt, args.retryPhase, second.ok, second.ok ? `${args.retryMessage}: succeeded on the second attempt` : `${args.retryMessage}: failed twice`);
619
- return { ...second, attempts: 2, ...(first.exitCode === undefined ? {} : { firstExitCode: first.exitCode }) };
620
- }
621
- function emitPhaseStarted(onPhase, now, phase, message) {
622
- onPhase?.({ at: isoNow(now), type: `cua-lab.subject.${phase}.started`, message });
623
- }
624
- /** Emit the matching phase-completed event: always carries ok and durationMs (>= 0). */
625
- function emitPhaseCompleted(onPhase, now, startedAt, phase, ok, message) {
626
- onPhase?.({
627
- at: isoNow(now),
628
- type: `cua-lab.subject.${phase}.completed`,
629
- ok,
630
- durationMs: Math.max(0, now() - startedAt),
631
- message
632
- });
633
- }
634
- /** Default phase-boundary sink (stderr): one line per event, prefixed with the lane id ONLY
635
- * when laneCount > 1. Single-lane emission is unconditional: total single-lane silence for the
636
- * whole clone/install/build/ready boot is the bug this event stream exists to close.
637
- * Overridable via CuaActorLabHooks.onPhase so deterministic tests capture instead of writing to
638
- * the real stderr. */
639
- function defaultSubjectPhaseSink(event, ctx) {
640
- const durationSuffix = event.durationMs === undefined ? "" : ` (${event.durationMs}ms)`;
641
- const prefix = ctx.laneCount > 1 ? `humanish cua [${ctx.laneId}]` : "humanish cua";
642
- process.stderr.write(`${prefix}: ${event.message}${durationSuffix}\n`);
643
- }
644
487
  /** Short id-safe suffix for a subject-phase RunEvent: drops the shared prefix/suffix so each
645
488
  * phase gets a distinct bundle event id (e.g. "clone", "state-before-build"). */
646
489
  function phaseEventIdSuffix(type) {
@@ -681,36 +524,6 @@ export function makeLaneWriteScreenshot(artifactRoot, spec, screenshots) {
681
524
  return rel;
682
525
  };
683
526
  }
684
- /**
685
- * Verify the desktop screen geometry IN-SANDBOX (the per-lane device claim is checked, never
686
- * assumed). A parseable mismatch fails closed. Unavailable/unparseable evidence is returned as
687
- * an explicit warning: the lane may still run, but its bundle records only the requested screen
688
- * and never upgrades that request into a verified measurement.
689
- */
690
- export async function inspectDesktopScreenGeometry(args) {
691
- let out = "";
692
- try {
693
- const result = await args.desktop.commands.run("xdpyinfo 2>/dev/null | grep -i dimensions || true", { requestTimeoutMs: args.requestTimeoutMs });
694
- out = (result.stdout ?? "").trim();
695
- }
696
- catch {
697
- return { warning: `Desktop screen geometry could not be measured for lane ${args.laneId}; requested geometry remains unverified.` };
698
- }
699
- const match = out.match(/(\d+)\s*x\s*(\d+)\s*pixels/i);
700
- if (!match) {
701
- return { warning: `Desktop screen geometry could not be parsed for lane ${args.laneId}; requested geometry remains unverified.` };
702
- }
703
- const width = Number(match[1]);
704
- const height = Number(match[2]);
705
- const [expectedWidth, expectedHeight] = args.requestedScreen;
706
- if (width === expectedWidth && height === expectedHeight) {
707
- return { verified: { width, height, source: "xdpyinfo" } };
708
- }
709
- return {
710
- verified: { width, height, source: "xdpyinfo" },
711
- error: `HUMANISH_CUA_LAB_DEVICE_GEOMETRY: lane ${args.laneId} requested a ${expectedWidth}x${expectedHeight} desktop but xdpyinfo reports ${width}x${height} in-sandbox; the per-lane device geometry is unverified (fail-closed).`
712
- };
713
- }
714
527
  /** A blocked lane outcome (pipeline gate / fail-fast skipped it before it ran). */
715
528
  function blockedLaneOutcome(spec, reason) {
716
529
  return {
@@ -728,725 +541,6 @@ function blockedLaneOutcome(spec, reason) {
728
541
  harnessError: false
729
542
  };
730
543
  }
731
- async function findVisibleBrowserWindowId(desktop, requestTimeoutMs, browserFamily, launchIdentity) {
732
- if (browserFamily === "unknown")
733
- return undefined;
734
- // The candidate loop keeps the LAST identity match: with a launch identity the match is
735
- // unique anyway, and without one every family candidate matches, so the newest visible
736
- // window of the launched family wins (the window this lane just opened).
737
- const finder = browserFamily === "firefox"
738
- ? [
739
- "find_firefox_window() {",
740
- " timeout 2s xdotool search --onlyvisible --class 'firefox|Firefox' 2>/dev/null || true",
741
- "}",
742
- "window_id=",
743
- "for _ in $(seq 1 10); do",
744
- " for candidate in $(find_firefox_window); do",
745
- " window_pid=\"$(xdotool getwindowpid \"$candidate\" 2>/dev/null || true)\"",
746
- " if matches_launch_identity \"$window_pid\"; then window_id=\"$candidate\"; fi",
747
- " done",
748
- " if [ -n \"$window_id\" ]; then break; fi",
749
- " sleep 0.5",
750
- "done"
751
- ]
752
- : [
753
- "find_chrome_window() {",
754
- " timeout 2s xdotool search --onlyvisible --class 'google-chrome|Google-chrome|chromium|Chromium|chrome|Chrome' 2>/dev/null || true",
755
- "}",
756
- "window_id=",
757
- "for _ in $(seq 1 10); do",
758
- " for candidate in $(find_chrome_window); do",
759
- " window_pid=\"$(xdotool getwindowpid \"$candidate\" 2>/dev/null || true)\"",
760
- " if matches_launch_identity \"$window_pid\"; then window_id=\"$candidate\"; fi",
761
- " done",
762
- " if [ -n \"$window_id\" ]; then break; fi",
763
- " sleep 0.5",
764
- "done"
765
- ];
766
- const result = await desktop.commands.run([
767
- "set -euo pipefail",
768
- "export DISPLAY=\"${DISPLAY:-:0}\"",
769
- `launch_pid=${shellSingleQuote(launchIdentity?.processId ?? "")}`,
770
- `profile_dir=${shellSingleQuote(launchIdentity?.profileDir ?? "")}`,
771
- "matches_launch_identity() {",
772
- " if [ -z \"$launch_pid\" ] && [ -z \"$profile_dir\" ]; then return 0; fi",
773
- " local current=\"${1:-}\"",
774
- " while [[ \"$current\" =~ ^[0-9]+$ ]] && [ \"$current\" -gt 1 ]; do",
775
- " cmdline=\"$(tr '\\0' ' ' < \"/proc/$current/cmdline\" 2>/dev/null || true)\"",
776
- " if [ -n \"$profile_dir\" ] && [[ \"$cmdline\" == *\"$profile_dir\"* ]]; then return 0; fi",
777
- " if [ \"$current\" = \"$launch_pid\" ]; then return 0; fi",
778
- " current=\"$(ps -o ppid= -p \"$current\" 2>/dev/null | tr -d ' ' || true)\"",
779
- " done",
780
- " return 1",
781
- "}",
782
- ...finder,
783
- "if [ -n \"$window_id\" ]; then printf 'WINDOW_ID=%s\\n' \"$window_id\"; fi"
784
- ].join("\n"), {
785
- requestTimeoutMs,
786
- timeoutMs: 15_000
787
- });
788
- return (result.stdout ?? "").match(/^WINDOW_ID=(\S+)$/m)?.[1];
789
- }
790
- /**
791
- * Build the xdotool command that makes a browser window fill the desktop.
792
- * Exported (pure) for contract tests. A window manager can ignore Chrome's
793
- * --window-size, so xdotool is the robust path: move the window to the origin,
794
- * then size it to the exact desktop resolution so Observer screenshots carry no
795
- * dead margin around the browser.
796
- */
797
- export function buildFillDesktopWindowCommand(windowId, width, height) {
798
- return [
799
- "set -euo pipefail",
800
- `win=${shellSingleQuote(windowId)}`,
801
- `xdotool windowactivate "$win" >/dev/null 2>&1 || true`,
802
- `xdotool windowmove "$win" 0 0 >/dev/null 2>&1 || true`,
803
- `xdotool windowsize "$win" ${width} ${height} >/dev/null 2>&1 || true`,
804
- ].join("\n");
805
- }
806
- /**
807
- * Best-effort initial fill. A contained smaller window remains usable; the capture
808
- * below checks for clipping and refuses an uncorrectable window before the actor runs.
809
- */
810
- async function fillDesktopBrowserWindow(desktop, windowId, resolution, requestTimeoutMs) {
811
- const [width, height] = resolution;
812
- await desktop.commands
813
- .run(buildFillDesktopWindowCommand(windowId, width, height), {
814
- requestTimeoutMs,
815
- timeoutMs: 10_000,
816
- })
817
- .catch(() => undefined);
818
- }
819
- async function openDesktopBrowserTarget(desktop, targetUrl, requestTimeoutMs, browserPreference,
820
- /** Launch-time flags that make mobile fidelity (#221) hold across every tab: the user agent and
821
- * touch events are browser-wide here, where the CDP holder covers only the launch page. */
822
- extraChromiumFlags = []) {
823
- const requestedBrowser = browserPreference ?? "default";
824
- if (isHttpUrl(targetUrl)) {
825
- const chromiumFlags = [...CHROMIUM_EVIDENCE_HYGIENE_FLAGS, ...extraChromiumFlags].map(shellSingleQuote).join(" ");
826
- const browserLaunchCommand = [
827
- "set -euo pipefail",
828
- `target_url=${shellSingleQuote(targetUrl)}`,
829
- `browser_preference=${shellSingleQuote(requestedBrowser)}`,
830
- "chrome_profile_dir=",
831
- `chrome_preferences_json=${shellSingleQuote(chromiumEvidenceProfilePreferencesJson())}`,
832
- "prepare_chrome_profile() {",
833
- " chrome_profile_dir=\"$(mktemp -d /tmp/humanish-chrome-profile.XXXXXX)\"",
834
- " mkdir -p \"$chrome_profile_dir/Default\"",
835
- " printf '%s\\n' \"$chrome_preferences_json\" > \"$chrome_profile_dir/Default/Preferences\"",
836
- "}",
837
- "launch_browser() {",
838
- " local label=\"$1\"",
839
- " local binary=\"$2\"",
840
- " shift 2",
841
- " if command -v \"$binary\" >/dev/null 2>&1; then",
842
- " nohup \"$binary\" \"$@\" \"$target_url\" >/tmp/humanish-browser-open.log 2>&1 &",
843
- " local launch_pid=$!",
844
- " printf 'HUMANISH_BROWSER_RESOLVED=%s\\n' \"$label\"",
845
- " printf 'HUMANISH_BROWSER_PID=%s\\n' \"$launch_pid\"",
846
- " printf 'HUMANISH_BROWSER_PROFILE_DIR=%s\\n' \"$chrome_profile_dir\"",
847
- " if [[ \"$label\" =~ ^(google-chrome|google-chrome-stable|chromium|chromium-browser)$ ]]; then",
848
- " for _ in $(seq 1 30); do",
849
- " if [ -s \"$chrome_profile_dir/DevToolsActivePort\" ]; then",
850
- " head -n 1 \"$chrome_profile_dir/DevToolsActivePort\" | sed 's/^/HUMANISH_BROWSER_CDP_PORT=/'",
851
- " break",
852
- " fi",
853
- " sleep 0.1",
854
- " done",
855
- " fi",
856
- " return 0",
857
- " fi",
858
- " return 1",
859
- "}",
860
- // Fixed CDP port (not :0/random): each seat has its OWN desktop sandbox, so a known port
861
- // cannot conflict, and it makes the observer's port resolution deterministic. With :0 the
862
- // real port lives only in DevToolsActivePort; when the launch-time capture misses on a cold
863
- // start the observer falls back to 9222 and — being wrong — every CDP read fails for the
864
- // whole run (the lobby-code handoff then never sees the host's /lobby URL). 9222 is already
865
- // the fallback, so making it the actual port aligns launch, capture, and fallback.
866
- `chrome_debug_flags=(--remote-debugging-address=127.0.0.1 --remote-debugging-port=9222 ${chromiumFlags})`,
867
- "open_target() {",
868
- " case \"$browser_preference\" in",
869
- " chrome)",
870
- " prepare_chrome_profile",
871
- " launch_browser google-chrome google-chrome --new-window \"--user-data-dir=$chrome_profile_dir\" \"${chrome_debug_flags[@]}\" && return 0",
872
- " launch_browser google-chrome-stable google-chrome-stable --new-window \"--user-data-dir=$chrome_profile_dir\" \"${chrome_debug_flags[@]}\" && return 0",
873
- " echo 'requested browser chrome was not found' >&2",
874
- " return 127",
875
- " ;;",
876
- " chromium)",
877
- " prepare_chrome_profile",
878
- " launch_browser chromium chromium --new-window \"--user-data-dir=$chrome_profile_dir\" \"${chrome_debug_flags[@]}\" && return 0",
879
- " launch_browser chromium-browser chromium-browser --new-window \"--user-data-dir=$chrome_profile_dir\" \"${chrome_debug_flags[@]}\" && return 0",
880
- " echo 'requested browser chromium was not found' >&2",
881
- " return 127",
882
- " ;;",
883
- " firefox)",
884
- " prepare_chrome_profile",
885
- " launch_browser firefox firefox --new-instance --no-remote --new-window --profile \"$chrome_profile_dir\" && return 0",
886
- " echo 'requested browser firefox was not found' >&2",
887
- " return 127",
888
- " ;;",
889
- " default)",
890
- " prepare_chrome_profile",
891
- " launch_browser google-chrome google-chrome --new-window \"--user-data-dir=$chrome_profile_dir\" \"${chrome_debug_flags[@]}\" && return 0",
892
- " launch_browser google-chrome-stable google-chrome-stable --new-window \"--user-data-dir=$chrome_profile_dir\" \"${chrome_debug_flags[@]}\" && return 0",
893
- " launch_browser chromium chromium --new-window \"--user-data-dir=$chrome_profile_dir\" \"${chrome_debug_flags[@]}\" && return 0",
894
- " launch_browser chromium-browser chromium-browser --new-window \"--user-data-dir=$chrome_profile_dir\" \"${chrome_debug_flags[@]}\" && return 0",
895
- " launch_browser firefox firefox --new-instance --no-remote --new-window --profile \"$chrome_profile_dir\" && return 0",
896
- " launch_browser xdg-open xdg-open && return 0",
897
- " echo 'no browser opener found' >&2",
898
- " return 127",
899
- " ;;",
900
- " esac",
901
- "}",
902
- "open_target"
903
- ].join("\n");
904
- const result = await runDesktopCommandOrThrow(() => desktop.commands.run(browserLaunchCommand, {
905
- requestTimeoutMs,
906
- timeoutMs: 15_000,
907
- }), ({ exitCode, stderrTail }) => new Error(`browser launch failed${exitCode === undefined ? "" : ` with exit ${exitCode}`}: ${stderrTail}`));
908
- if (result.exitCode !== undefined && result.exitCode !== 0) {
909
- throw new Error(`browser launch failed with exit ${result.exitCode}: ${tailOf(result.stderr ?? result.stdout ?? "")}`);
910
- }
911
- const resolved = (result.stdout ?? "").match(/^HUMANISH_BROWSER_RESOLVED=(\S+)$/m)?.[1];
912
- const processId = (result.stdout ?? "").match(/^HUMANISH_BROWSER_PID=(\d+)$/m)?.[1];
913
- const profileDir = (result.stdout ?? "").match(/^HUMANISH_BROWSER_PROFILE_DIR=(\S+)$/m)?.[1];
914
- const cdpPortRaw = (result.stdout ?? "").match(/^HUMANISH_BROWSER_CDP_PORT=(\d+)$/m)?.[1];
915
- const cdpPort = cdpPortRaw === undefined ? undefined : Number(cdpPortRaw);
916
- return {
917
- family: desktopBrowserFamily(resolved ?? requestedBrowser),
918
- ...(processId === undefined || profileDir === undefined
919
- ? {}
920
- : { identity: { processId, profileDir, targetUrl, ...(cdpPort === undefined ? {} : { cdpPort }) } }),
921
- ...(browserPreference === undefined
922
- ? {}
923
- : { evidence: { requested: requestedBrowser, ...(resolved === undefined ? {} : { resolved }) } })
924
- };
925
- }
926
- if (browserPreference === undefined || browserPreference === "default") {
927
- if (desktop.open) {
928
- await desktop.open(targetUrl);
929
- }
930
- else {
931
- await desktop.launch("google-chrome", targetUrl);
932
- }
933
- return {
934
- family: desktop.open ? "unknown" : "chromium",
935
- ...(browserPreference === undefined ? {} : { evidence: { requested: requestedBrowser } })
936
- };
937
- }
938
- const launchTarget = requestedBrowser === "chrome" ? "google-chrome"
939
- : requestedBrowser === "chromium" ? "chromium"
940
- : requestedBrowser === "firefox" ? "firefox"
941
- : "google-chrome";
942
- await desktop.launch(launchTarget, targetUrl);
943
- return {
944
- family: desktopBrowserFamily(launchTarget),
945
- evidence: { requested: requestedBrowser, resolved: launchTarget }
946
- };
947
- }
948
- export function desktopBrowserFamily(value) {
949
- if (value === "firefox")
950
- return "firefox";
951
- if (value === "chrome" || value === "chromium" || value === "google-chrome" || value === "google-chrome-stable" || value === "chromium-browser") {
952
- return "chromium";
953
- }
954
- return "unknown";
955
- }
956
- /**
957
- * The URL / title / page-text / scroll observer behind stopWhen and task criteria. One probe per
958
- * observation, run on the sandbox's python3 (see chrome-cdp-probe.ts for why not node: #514).
959
- *
960
- * "active": follow the participant to whatever tab they are driving now — never pin the state
961
- * observer to the launch tab (a verification link that opened in a NEW tab left a pinned observer
962
- * reading the old tab forever).
963
- *
964
- * `onUnavailable` fires ONCE, on the first probe that could not read the page, with the reason.
965
- * The observer still degrades to `{}` for the loop; the callback is how a lane says out loud that
966
- * url/text criteria are not being measured, instead of letting the funnel report 0/N (#514).
967
- */
968
- export function makeChromeBrowserStateObserver(desktop, requestTimeoutMs, endpoint, targetId, onUnavailable,
969
- /**
970
- * Mobile emulation on later tabs (#623): the holder attaches to every page target Chrome opens
971
- * after the launch page, so a tab the participant opens later should lay out at the phone width
972
- * too. The first observation on each new target reads that page's OWN report; a target that
973
- * reports the requested width is recorded through `onCovered`, and one that does not (or cannot
974
- * be read) fires `onDrift` once, so a phone-labelled lane that spent part of its session at
975
- * desktop layout says so with the number the page gave.
976
- */
977
- drift) {
978
- let reported = false;
979
- let drifted = false;
980
- const checkedTargets = new Set(drift === undefined ? [] : [drift.emulatedTargetId]);
981
- const unavailable = (reason) => {
982
- if (!reported) {
983
- reported = true;
984
- onUnavailable?.(reason);
985
- }
986
- return {};
987
- };
988
- const checkLaterTarget = async (newTargetId) => {
989
- if (drift === undefined || checkedTargets.has(newTargetId))
990
- return;
991
- checkedTargets.add(newTargetId);
992
- const read = await desktop.commands.run(chromeCdpProbeCommand({ ...endpoint, targetId: newTargetId, prefer: "pinned", mode: "fidelity" }), { requestTimeoutMs, timeoutMs: 5_000 });
993
- const fidelity = read.exitCode !== undefined && read.exitCode !== 0 ? undefined : parseChromeCdpProbeOutput(read.stdout).fidelity;
994
- if (fidelity !== undefined && fidelity.innerWidth === drift.expectedWidth) {
995
- drift.onCovered?.(newTargetId, { innerWidth: fidelity.innerWidth, devicePixelRatio: fidelity.devicePixelRatio, maxTouchPoints: fidelity.maxTouchPoints });
996
- if (drift.expectTouch === true && fidelity.maxTouchPoints === 0 && !drifted) {
997
- // The viewport followed; touch did not (yet): the holder reloads a later tab once after its
998
- // first navigation commits, and this observation may have landed before that reload.
999
- drifted = true;
1000
- drift.onDrift(`a later page target reports the ${fidelity.innerWidth} px viewport but navigator.maxTouchPoints 0 on its first observation; touch reaches a document only when it loads under the override`);
1001
- }
1002
- return;
1003
- }
1004
- if (drifted)
1005
- return;
1006
- drifted = true;
1007
- drift.onDrift(fidelity === undefined
1008
- ? "the participant drove a page target other than the emulated launch tab and that page's own read-back could not be taken; whether it laid out at the phone width is not known"
1009
- : `the participant drove a page target other than the emulated launch tab and that page reports a ${fidelity.innerWidth} px viewport where ${drift.expectedWidth} px was requested (DPR ${fidelity.devicePixelRatio}); the mobile user agent and touch events are browser-wide, the viewport override was not re-applied to it`);
1010
- };
1011
- return async () => {
1012
- const result = await desktop.commands.run(chromeCdpProbeCommand({ ...endpoint, ...(targetId === undefined ? {} : { targetId }), prefer: "active", mode: "state" }), { requestTimeoutMs, timeoutMs: 5_000 });
1013
- if (result.exitCode !== undefined && result.exitCode !== 0) {
1014
- return unavailable(`probe exited ${result.exitCode}: ${tailOf(result.stderr ?? result.stdout ?? "")}`);
1015
- }
1016
- const parsed = parseChromeCdpProbeOutput(result.stdout);
1017
- if (parsed.unavailable !== undefined)
1018
- return unavailable(parsed.unavailable);
1019
- if (parsed.targetId !== undefined)
1020
- await checkLaterTarget(parsed.targetId);
1021
- return {
1022
- ...(parsed.url === undefined ? {} : { url: parsed.url }),
1023
- ...(parsed.title === undefined ? {} : { title: parsed.title }),
1024
- ...(parsed.text === undefined ? {} : { text: parsed.text }),
1025
- ...(parsed.scrollY === undefined ? {} : { scrollY: parsed.scrollY })
1026
- };
1027
- };
1028
- }
1029
- /**
1030
- * Read the running browser's actual outer-window bounds and CSS layout viewport through the
1031
- * already-enabled local Chrome DevTools endpoint. The returned values come from `window.*` in
1032
- * the target page; requested E2B resolution is deliberately not an input to this function.
1033
- * Missing channels report their reason via `onUnavailable`, so the geometry warning can name
1034
- * the cause (a dead CDP endpoint, no python3) instead of only the symptom. Returns `undefined`
1035
- * only when neither channel could be measured.
1036
- * Outer bounds and CSS dimensions are independent channels: a background page can report zero
1037
- * outer dimensions while still reporting a CSS viewport. Final captures follow the active tab;
1038
- * launch captures and emulation attribution keep the pinned target.
1039
- */
1040
- export function makeChromeDesktopGeometryObserver(desktop, requestTimeoutMs, endpoint, targetId, onUnavailable, prefer = "pinned") {
1041
- return async () => {
1042
- const result = await desktop.commands.run(chromeCdpProbeCommand({ ...endpoint, ...(targetId === undefined ? {} : { targetId }), prefer, mode: "geometry" }), { requestTimeoutMs, timeoutMs: 5_000 });
1043
- if (result.exitCode !== undefined && result.exitCode !== 0) {
1044
- onUnavailable?.(`probe exited ${result.exitCode}: ${tailOf(result.stderr ?? result.stdout ?? "")}`);
1045
- return undefined;
1046
- }
1047
- const parsed = parseChromeCdpProbeOutput(result.stdout);
1048
- if (parsed.unavailable !== undefined) {
1049
- onUnavailable?.(parsed.unavailable);
1050
- return undefined;
1051
- }
1052
- const browserWindow = isMeasuredRect(parsed.browserWindow) ? { ...parsed.browserWindow, source: "cdp" } : undefined;
1053
- const viewport = isMeasuredViewport(parsed.viewport) ? { ...parsed.viewport, source: "cdp" } : undefined;
1054
- if (browserWindow === undefined && viewport === undefined) {
1055
- onUnavailable?.("the page reported no usable window or viewport dimensions");
1056
- return undefined;
1057
- }
1058
- if (browserWindow === undefined)
1059
- onUnavailable?.("the page reported no usable outer-window dimensions");
1060
- if (viewport === undefined)
1061
- onUnavailable?.("the page reported no usable CSS viewport dimensions");
1062
- return {
1063
- ...(browserWindow === undefined ? {} : { browserWindow }),
1064
- ...(viewport === undefined ? {} : { viewport }),
1065
- ...(parsed.targetId === undefined ? {} : { targetId: parsed.targetId })
1066
- };
1067
- };
1068
- }
1069
- /** The user agent a mobile-emulated lane presents unless the lab sets its own. */
1070
- export const DEFAULT_MOBILE_USER_AGENT = "Mozilla/5.0 (iPhone; CPU iPhone OS 17_0 like Mac OS X) AppleWebKit/605.1.15 (KHTML, like Gecko) Version/17.0 Mobile/15E148 Safari/604.1";
1071
- /**
1072
- * Apply mobile emulation (#221) to the lane's launch page and read back what the page reports.
1073
- * Fails CLOSED: a request that cannot be applied throws, because a desktop run labelled mobile is
1074
- * the over-trust this feature exists to prevent. A read-back that cannot be taken is a warning
1075
- * (the emulation was applied; only the proof is missing).
1076
- */
1077
- export async function applyMobileEmulation(desktop, requestTimeoutMs, endpoint, targetId, request) {
1078
- const command = (mode) => chromeCdpProbeCommand({ ...endpoint, ...(targetId === undefined ? {} : { targetId }), prefer: "pinned", mode, emulation: request });
1079
- const read = async () => {
1080
- const result = await desktop.commands.run(command("fidelity"), { requestTimeoutMs, timeoutMs: 15_000 });
1081
- if (result.exitCode !== undefined && result.exitCode !== 0) {
1082
- return { unavailable: `probe exited ${result.exitCode}: ${tailOf(result.stderr ?? result.stdout ?? "")}` };
1083
- }
1084
- return parseChromeCdpProbeOutput(result.stdout);
1085
- };
1086
- // The UA / touch / DPR overrides are bound to the DevTools session that set them and lapse the
1087
- // moment its socket closes (measured: only the viewport width survived a one-shot apply). So the
1088
- // applier stays attached for the lane's whole life as a detached process; the sandbox teardown
1089
- // ends it. Its first stdout line says what was applied.
1090
- const holderName = `mobile-emulation-${Date.now().toString(36)}`;
1091
- await startDetachedProcess(desktop, { name: holderName, command: command("hold"), requestTimeoutMs });
1092
- let announced;
1093
- for (let attempt = 0; attempt < 30 && announced === undefined; attempt += 1) {
1094
- await new Promise((resolve) => setTimeout(resolve, 500));
1095
- const log = await readDetachedLog(desktop, holderName, requestTimeoutMs).catch(() => "");
1096
- const line = log.split("\n").find((candidate) => candidate.trim().startsWith("{"));
1097
- if (line !== undefined)
1098
- announced = parseChromeCdpProbeOutput(line);
1099
- }
1100
- if (announced === undefined) {
1101
- throw new Error("mobile emulation could not be applied: the in-sandbox applier printed nothing within 15 s");
1102
- }
1103
- if (announced.unavailable !== undefined) {
1104
- throw new Error(`mobile emulation could not be applied (${announced.unavailable}); applied before failing: ${(announced.applied ?? []).join(", ") || "nothing"}`);
1105
- }
1106
- const applied = announced;
1107
- // Viewport/touch read-back proves context settings, not gesture equivalence. Two hosted
1108
- // replicas and a native-X conversion-toggle control reproduced reset click counts (#676).
1109
- const warnings = request.touch
1110
- ? ["Mobile emulation uses desktop pointer-to-touch conversion, which can differ for repeated taps. Confirm gesture failures with direct or native touch input before attributing them to the app."]
1111
- : [];
1112
- // The reload inside the applier takes a moment; the read-back is retried until the page reports
1113
- // the requested viewport and user agent, so a slow page does not read as "no proof".
1114
- let readBack = await read();
1115
- for (let attempt = 0; attempt < 20 && (readBack.fidelity === undefined || readBack.fidelity.innerWidth !== request.width || !readBack.fidelity.userAgent.includes(request.userAgent.slice(0, 24))); attempt += 1) {
1116
- await new Promise((resolve) => setTimeout(resolve, 500));
1117
- readBack = await read();
1118
- }
1119
- const fidelityRead = readBack;
1120
- const requested = {
1121
- width: request.width,
1122
- height: request.height,
1123
- deviceScaleFactor: request.deviceScaleFactor,
1124
- touch: request.touch,
1125
- userAgent: request.userAgent
1126
- };
1127
- const emulatedTargetId = applied.targetId ?? fidelityRead.targetId;
1128
- if (fidelityRead.fidelity === undefined) {
1129
- warnings.push(`Mobile emulation was applied but the page's own report could not be read (${fidelityRead.unavailable ?? "no fidelity read"}); desktopGeometry.fidelity carries the request without a resolved block.`);
1130
- return { fidelity: { tier: "mobile-emulated", requested, applied: applied.applied ?? [] }, warnings, holderName, ...(emulatedTargetId === undefined ? {} : { targetId: emulatedTargetId }) };
1131
- }
1132
- const resolved = { ...fidelityRead.fidelity, source: "cdp" };
1133
- if (resolved.innerWidth !== request.width) {
1134
- warnings.push(`Mobile emulation requested a ${request.width} px viewport; the page reports ${resolved.innerWidth} px.`);
1135
- }
1136
- if (resolved.devicePixelRatio !== request.deviceScaleFactor) {
1137
- warnings.push(`Mobile emulation requested devicePixelRatio ${request.deviceScaleFactor}; the page reports ${resolved.devicePixelRatio}.`);
1138
- }
1139
- if (request.touch && resolved.maxTouchPoints === 0) {
1140
- warnings.push("Mobile emulation requested touch; the page reports navigator.maxTouchPoints 0.");
1141
- }
1142
- if (!resolved.userAgent.includes("Mobile") && !resolved.userAgent.includes("Android") && !resolved.userAgent.includes("iPhone")) {
1143
- warnings.push("Mobile emulation requested a mobile user agent; the page reports a desktop one.");
1144
- }
1145
- return { fidelity: { tier: "mobile-emulated", requested, applied: applied.applied ?? [], resolved }, warnings, holderName, ...(emulatedTargetId === undefined ? {} : { targetId: emulatedTargetId }) };
1146
- }
1147
- function isMeasuredRect(value) {
1148
- if (!value || typeof value !== "object")
1149
- return false;
1150
- const record = value;
1151
- return Number.isFinite(record.x)
1152
- && Number.isFinite(record.y)
1153
- && isPositiveMeasurement(record.width)
1154
- && isPositiveMeasurement(record.height);
1155
- }
1156
- function isMeasuredViewport(value) {
1157
- if (!value || typeof value !== "object")
1158
- return false;
1159
- const record = value;
1160
- return isPositiveMeasurement(record.width)
1161
- && isPositiveMeasurement(record.height)
1162
- && isPositiveMeasurement(record.deviceScaleFactor);
1163
- }
1164
- function isPositiveMeasurement(value) {
1165
- return typeof value === "number" && Number.isFinite(value) && value > 0;
1166
- }
1167
- async function measureBrowserWindowWithXwininfo(desktop, windowId, requestTimeoutMs) {
1168
- const result = await desktop.commands.run([
1169
- "set -euo pipefail",
1170
- `win=${shellSingleQuote(windowId)}`,
1171
- // Older xdotool builds translate parent-relative offsets twice. With window
1172
- // decorations that falsely reports a visible client as clipped, triggering
1173
- // fullscreen and hiding the participant's address bar. Read root-relative
1174
- // client coordinates directly; never substitute emulated CDP outer bounds.
1175
- "LC_ALL=C xwininfo -id \"$win\" -stats 2>/dev/null"
1176
- ].join("\n"), { requestTimeoutMs, timeoutMs: 5_000 });
1177
- if (result.exitCode !== undefined && result.exitCode !== 0)
1178
- return undefined;
1179
- return parseXwininfoGeometry(result.stdout ?? "");
1180
- }
1181
- /** Root-relative physical client bounds from xwininfo's C-locale stats. */
1182
- export function parseXwininfoGeometry(output) {
1183
- const read = (label) => {
1184
- const matches = [...output.matchAll(new RegExp(`^\\s*${label}:\\s*(-?\\d+)\\s*$`, "gm"))];
1185
- if (matches.length !== 1)
1186
- return undefined;
1187
- const value = Number(matches[0][1]);
1188
- return Number.isSafeInteger(value) ? value : undefined;
1189
- };
1190
- const x = read("Absolute upper-left X");
1191
- const y = read("Absolute upper-left Y");
1192
- const width = read("Width");
1193
- const height = read("Height");
1194
- const mapStates = [...output.matchAll(/^\s*Map State:\s*(\S+)\s*$/gm)];
1195
- if (mapStates.length !== 1 || mapStates[0][1] !== "IsViewable")
1196
- return undefined;
1197
- if (x === undefined || y === undefined || width === undefined || height === undefined || width <= 0 || height <= 0) {
1198
- return undefined;
1199
- }
1200
- return { x, y, width, height, source: "xwininfo" };
1201
- }
1202
- /** Physical X client bounds, never the page's emulated window.outerWidth/Height. */
1203
- function isBrowserWindowContained(bounds, [width, height]) {
1204
- return bounds.x >= 0 && bounds.y >= 0
1205
- && bounds.x + bounds.width <= width && bounds.y + bounds.height <= height;
1206
- }
1207
- /** Bounded repair. Resizing can clear a window-manager maximize state and move the
1208
- * client origin as decorations return, so remeasure before a second adjustment. */
1209
- async function fitBrowserWindowWithinDesktop(desktop, windowId, resolution, requestTimeoutMs) {
1210
- const run = (command) => desktop.commands.run([
1211
- "set -euo pipefail",
1212
- `win=${shellSingleQuote(windowId)}`,
1213
- command
1214
- ].join("\n"), { requestTimeoutMs, timeoutMs: 5_000 }).catch(() => undefined);
1215
- await run('xdotool windowmove "$win" 0 0');
1216
- await desktop.wait(250).catch(() => undefined);
1217
- const moved = await measureBrowserWindowWithXwininfo(desktop, windowId, requestTimeoutMs).catch(() => undefined);
1218
- if (moved === undefined)
1219
- return moved;
1220
- let resized = moved;
1221
- // Resizing alone cannot fix an offscreen client origin. The window manager
1222
- // can also center a minimum-width client at a negative x on a narrow screen.
1223
- for (let attempt = 0; attempt < 2; attempt += 1) {
1224
- if (isBrowserWindowContained(resized, resolution))
1225
- return resized;
1226
- const width = resolution[0] - resized.x;
1227
- const height = resolution[1] - resized.y;
1228
- if (resized.x < 0 || resized.y < 0 || width <= 0 || height <= 0)
1229
- break;
1230
- await run(`xdotool windowsize "$win" ${width} ${height}`);
1231
- await desktop.wait(250).catch(() => undefined);
1232
- const measured = await measureBrowserWindowWithXwininfo(desktop, windowId, requestTimeoutMs).catch(() => undefined);
1233
- if (measured === undefined)
1234
- return measured;
1235
- resized = measured;
1236
- }
1237
- if (resized === undefined || isBrowserWindowContained(resized, resolution))
1238
- return resized;
1239
- // Chrome's minimum client width can equal the whole desktop. Window-manager
1240
- // borders then make a decorated window impossible to contain, even after a
1241
- // successful move/resize. Request fullscreen once and prove the physical result.
1242
- // xprop/xdotool ship with the desktop template; wmctrl is not required.
1243
- // Check state first so the fullscreen shortcut cannot toggle an existing state off.
1244
- await run([
1245
- 'state=$(xprop -id "$win" _NET_WM_STATE)',
1246
- 'case "$state" in',
1247
- ' *_NET_WM_STATE_FULLSCREEN*) ;;',
1248
- ' *) xdotool windowactivate --sync "$win"; xdotool key --clearmodifiers F11 ;;',
1249
- 'esac'
1250
- ].join("\n"));
1251
- // The fullscreen animation may report its new origin before its final width.
1252
- // Give the window manager a bounded settling window, keeping missing reads unverified.
1253
- for (let attempt = 0; attempt < 4; attempt += 1) {
1254
- await desktop.wait(250).catch(() => undefined);
1255
- const measured = await measureBrowserWindowWithXwininfo(desktop, windowId, requestTimeoutMs).catch(() => undefined);
1256
- if (measured === undefined || isBrowserWindowContained(measured, resolution))
1257
- return measured;
1258
- resized = measured;
1259
- }
1260
- return resized;
1261
- }
1262
- /** Shared hosted-browser geometry capture used by per-lane and sequential shared-world routes. */
1263
- export async function captureDesktopBrowserGeometry(args) {
1264
- const warnings = [];
1265
- let browserWindowId = args.browserWindowId;
1266
- if (browserWindowId === undefined && args.browserFamily !== "unknown") {
1267
- browserWindowId = await findVisibleBrowserWindowId(args.desktop, args.requestTimeoutMs, args.browserFamily, args.launchIdentity).catch((error) => {
1268
- warnings.push(`Browser window lookup failed for lane ${args.laneId}: ${redactText(toErrorMessage(error))}`);
1269
- return undefined;
1270
- });
1271
- }
1272
- let physicalWindow;
1273
- if (browserWindowId !== undefined) {
1274
- if (args.resize !== false) {
1275
- await fillDesktopBrowserWindow(args.desktop, browserWindowId, args.requestedScreen, args.requestTimeoutMs);
1276
- // Let the window manager apply the resize before querying both X and page layout geometry.
1277
- await args.desktop.wait(250).catch(() => undefined);
1278
- }
1279
- physicalWindow = await measureBrowserWindowWithXwininfo(args.desktop, browserWindowId, args.requestTimeoutMs)
1280
- .catch(() => undefined);
1281
- }
1282
- else {
1283
- warnings.push(`Browser window bounds could not be measured for lane ${args.laneId}; the live stream will use the full desktop.`);
1284
- }
1285
- let unusable;
1286
- if (physicalWindow !== undefined && !isBrowserWindowContained(physicalWindow, args.requestedScreen)) {
1287
- const before = physicalWindow;
1288
- if (args.resize !== false && browserWindowId !== undefined) {
1289
- physicalWindow = await fitBrowserWindowWithinDesktop(args.desktop, browserWindowId, args.requestedScreen, args.requestTimeoutMs);
1290
- if (physicalWindow === undefined) {
1291
- // Keep the last measured bad state; a missing observation cannot prove a successful fix.
1292
- physicalWindow = before;
1293
- unusable = `Physical browser containment could not be verified after correction for lane ${args.laneId}; the last measured window was clipped.`;
1294
- }
1295
- else if (isBrowserWindowContained(physicalWindow, args.requestedScreen)) {
1296
- warnings.push(`Browser window clipping corrected for lane ${args.laneId}; physical bounds are ${physicalWindow.width}x${physicalWindow.height} at (${physicalWindow.x}, ${physicalWindow.y}).`);
1297
- }
1298
- }
1299
- if (unusable === undefined && !isBrowserWindowContained(physicalWindow, args.requestedScreen)) {
1300
- unusable = `Browser window is outside the captured ${args.requestedScreen[0]}x${args.requestedScreen[1]} desktop for lane ${args.laneId}: physical bounds ${physicalWindow.width}x${physicalWindow.height} at (${physicalWindow.x}, ${physicalWindow.y}), right=${physicalWindow.x + physicalWindow.width}, bottom=${physicalWindow.y + physicalWindow.height}.`;
1301
- }
1302
- if (unusable !== undefined)
1303
- warnings.push(unusable);
1304
- }
1305
- if (physicalWindow === undefined) {
1306
- warnings.push(`Physical browser containment is unverified for lane ${args.laneId}; X window bounds could not be measured. Page-reported outer dimensions can be emulated and do not prove physical visibility.`);
1307
- }
1308
- let cdpUnavailable;
1309
- const chromeGeometry = args.browserFamily === "chromium"
1310
- ? await makeChromeDesktopGeometryObserver(args.desktop, args.requestTimeoutMs, {
1311
- ...(args.launchIdentity?.cdpPort === undefined ? {} : { cdpPort: args.launchIdentity.cdpPort }),
1312
- ...(args.launchIdentity?.profileDir === undefined ? {} : { profileDir: args.launchIdentity.profileDir }),
1313
- targetUrl: args.targetUrl
1314
- }, args.browserTargetId, (reason) => {
1315
- cdpUnavailable = reason;
1316
- }, args.pagePreference ?? "pinned")().catch((error) => {
1317
- cdpUnavailable = toErrorMessage(error);
1318
- return undefined;
1319
- })
1320
- : undefined;
1321
- const browserWindow = physicalWindow ?? chromeGeometry?.browserWindow;
1322
- const viewport = chromeGeometry?.viewport;
1323
- // The fill check reads the X window when it was measured: under mobile emulation (#221) the
1324
- // page's window.outerWidth reports the EMULATED screen (414), which is not a fill failure.
1325
- const fillBounds = physicalWindow;
1326
- if (!browserWindow) {
1327
- warnings.push(`Browser outer bounds could not be measured for lane ${args.laneId}.`);
1328
- }
1329
- else if (unusable === undefined && fillBounds !== undefined && (fillBounds.x !== 0 || fillBounds.y !== 0 || fillBounds.width !== args.requestedScreen[0] || fillBounds.height !== args.requestedScreen[1])) {
1330
- warnings.push(`Browser window fill did not reach the requested ${args.requestedScreen[0]}x${args.requestedScreen[1]} screen for lane ${args.laneId}; measured physical bounds are ${fillBounds.width}x${fillBounds.height} at (${fillBounds.x}, ${fillBounds.y}).`);
1331
- }
1332
- if (!viewport) {
1333
- // Name the cause, not only the symptom: the same dead DevTools channel that loses the viewport
1334
- // loses every url/text observation, and a reader of the bundle should learn that here (#514).
1335
- const cause = cdpUnavailable === undefined ? "" : ` DevTools probe: ${redactText(cdpUnavailable)}.`;
1336
- warnings.push(args.browserFamily === "firefox"
1337
- ? `Browser CSS viewport measurement is unavailable for Firefox on lane ${args.laneId}; stream.viewport is omitted instead of reading a different browser's CDP endpoint.`
1338
- : `Browser CSS viewport could not be measured for lane ${args.laneId}; stream.viewport is omitted instead of copying the requested screen resolution.${cause}`);
1339
- }
1340
- return {
1341
- ...(unusable === undefined ? {} : { unusable }),
1342
- ...(browserWindowId === undefined ? {} : { browserWindowId }),
1343
- ...(chromeGeometry?.targetId === undefined ? {} : { browserTargetId: chromeGeometry.targetId }),
1344
- ...(browserWindow === undefined ? {} : { browserWindow }),
1345
- ...(viewport === undefined ? {} : { viewport }),
1346
- warnings
1347
- };
1348
- }
1349
- function shellSingleQuote(value) {
1350
- return `'${value.replace(/'/g, "'\\''")}'`;
1351
- }
1352
- /**
1353
- * Prepare a CLI study's runtime and, only when declared, its product (#495, #515).
1354
- *
1355
- * The install runs UNKEYED and before the session starts, for the same reason the clone route
1356
- * provisions its subject first: what is being studied begins when the participant looks at the
1357
- * screen. Omitting install deliberately studies product installation; Node/npm remain a
1358
- * harness prerequisite so the participant can follow the product's public npm instructions.
1359
- */
1360
- async function provisionDesktopCli(desktop, args) {
1361
- const install = args.install;
1362
- const now = () => Date.now();
1363
- if (install === undefined || needsNodeRuntime([install])) {
1364
- const startedAt = now();
1365
- emitPhaseStarted(args.onPhase, now, "runtime", "providing Node/npm for the desktop CLI study");
1366
- const bootstrap = await runDetachedStep(desktop, {
1367
- name: "desktop-cli-runtime-node",
1368
- command: TERMINAL_NODE_BOOTSTRAP_COMMAND,
1369
- cwd: "/home/user",
1370
- timeoutMs: INSTALL_TIMEOUT_MS,
1371
- requestTimeoutMs: args.requestTimeoutMs
1372
- });
1373
- emitPhaseCompleted(args.onPhase, now, startedAt, "runtime", bootstrap.ok, bootstrap.ok
1374
- ? "Node runtime ready"
1375
- : "Node runtime bootstrap failed");
1376
- if (!bootstrap.ok) {
1377
- throw new Error(`desktop-cli runtime bootstrap failed for "${args.product}"`);
1378
- }
1379
- }
1380
- if (install === undefined)
1381
- return;
1382
- const startedAt = now();
1383
- emitPhaseStarted(args.onPhase, now, "install", `installing ${args.product} on the desktop`);
1384
- const result = await runDetachedStep(desktop, {
1385
- name: "desktop-cli-install",
1386
- command: install,
1387
- cwd: "/home/user",
1388
- timeoutMs: INSTALL_TIMEOUT_MS,
1389
- requestTimeoutMs: args.requestTimeoutMs
1390
- });
1391
- emitPhaseCompleted(args.onPhase, now, startedAt, "install", result.ok, result.ok
1392
- ? `${args.product} installed`
1393
- : `installing ${args.product} failed`);
1394
- if (!result.ok) {
1395
- // Fail closed: a participant handed a desktop where the product is not installed would produce
1396
- // a transcript about a missing command, and that finding belongs to the harness, not the tool.
1397
- // The tail rides along, scrubbed before truncation like every other provisioning failure — a
1398
- // bare "install failed" is unactionable to whoever wrote the command.
1399
- throw new Error(args.scrub(`desktop-cli install failed for "${args.product}" (${result.timedOut ? "timed out" : `exit ${result.exitCode ?? "?"}`}): ${tailOf(args.scrub(result.logTail))}`));
1400
- }
1401
- }
1402
- /**
1403
- * Open a terminal window on the desktop.
1404
- *
1405
- * The stock template is XFCE and ships xfce4-terminal (also aliased x-terminal-emulator), verified
1406
- * live before this route was built. `x-terminal-emulator` is tried first so a template that swaps
1407
- * the emulator still works; a desktop with neither is a template problem and fails closed rather
1408
- * than handing a participant an empty screen and calling it a study.
1409
- */
1410
- async function openDesktopTerminal(desktop, requestTimeoutMs, workdir) {
1411
- const dir = workdir ?? "/home/user";
1412
- const result = await runDetachedStep(desktop, {
1413
- name: "desktop-cli-terminal",
1414
- command: [
1415
- "for candidate in x-terminal-emulator xfce4-terminal gnome-terminal konsole xterm; do",
1416
- ' if command -v "$candidate" >/dev/null 2>&1; then',
1417
- // LANG is set on the terminal we open, not globally: the stock image declares no locale, and
1418
- // a study that measures our own mojibake against an unconfigured template would be measuring
1419
- // the template. The PRODUCT-side fix (an ASCII fallback when the locale is not UTF-8) is in
1420
- // src/terminal-encoding.ts, and it is the one that matters for real users.
1421
- ` (cd ${shellSingleQuote(dir)} 2>/dev/null || cd /home/user; DISPLAY=:0 LANG=C.UTF-8 LC_ALL=C.UTF-8 HUMANISH_STUDY_PARTICIPANT=1 nohup "$candidate" >/dev/null 2>&1 &)`,
1422
- " sleep 3",
1423
- ' echo "humanish: opened $candidate"',
1424
- " exit 0",
1425
- " fi",
1426
- "done",
1427
- "echo 'humanish: no terminal emulator on this desktop template' >&2",
1428
- "exit 1"
1429
- ].join("\n"),
1430
- cwd: "/home/user",
1431
- timeoutMs: 60_000,
1432
- requestTimeoutMs
1433
- });
1434
- if (!result.ok) {
1435
- throw new Error("desktop-cli lane could not open a terminal on this desktop template");
1436
- }
1437
- }
1438
- async function startDesktopStream(desktop, browserWindowId) {
1439
- if (!browserWindowId) {
1440
- await desktop.stream.start({ requireAuth: true });
1441
- return;
1442
- }
1443
- try {
1444
- await desktop.stream.start({ requireAuth: true, windowId: browserWindowId });
1445
- }
1446
- catch {
1447
- await desktop.stream.start({ requireAuth: true });
1448
- }
1449
- }
1450
544
  // "can't" followed by a PERCEPTION verb describes what the screen showed, not an inability to
1451
545
  // proceed: "the canvas truncates it so you can't even read the whole thing", "I can't tell from
1452
546
  // the screen whether the rename is persisted", "so I could not read its full description". Five
@@ -1656,118 +750,18 @@ export function resolveSelfReportedFriction(session) {
1656
750
  return reports.join("\n\n");
1657
751
  return undefined;
1658
752
  }
1659
- /**
1660
- * Run ONE E2B desktop lane end-to-end: create the sandbox (per-lane metadata + the lane's device
1661
- * resolution), prepareDesktop, verify geometry, (clone+serve+seed the subject per lane), open the
1662
- * browser, run the session, and ALWAYS tear down THIS lane's sandbox BY ID in a finally. Never
1663
- * enumerates sandboxes. Extracted from the former single-lane block; at N=1 it writes the exact
1664
- * same artifacts (actor.json, screenshots/<name>) the bundle has always referenced.
1665
- */
753
+ /** Run one participant against a prepared desktop. The adapter owns provisioning, final
754
+ * evidence and cleanup; this runner owns the model, trace and participant outcome. */
1666
755
  export async function runCuaLane(spec, deps) {
1667
- const { config, appUrl, cloneRoute, localTreeRoute, serve, subjectRepo, subjectEnvNames } = deps;
1668
- const desktopCliRoute = deps.desktopCliRoute === true;
1669
- // The local brain, when there is one. `appServer` / `claudeSession` own a process, so the lane
1670
- // closes it.
756
+ const { config, env } = deps;
1671
757
  let appServer;
1672
758
  let claudeSession;
1673
759
  let localAgentProvider;
1674
- const subjectEnvValues = config.subject.envValues ?? {};
1675
- const targetUrl = spec.targetUrl ?? appUrl;
1676
- const env = deps.env;
1677
- // Off-app comms (#297): on an in-sandbox subject route, redirect the app's email-API sends into an
1678
- // in-sandbox catch (loopback) so its verification mail is CAPTURED, not sent to the internet. Gated
1679
- // ENTIRELY on config.comms — no comms declared → zero change. The base-URL env is injected at
1680
- // sandbox-create (below, so the app reads it at boot); the catch is started right after create.
1681
- const commsEmail = (cloneRoute || localTreeRoute) && config.comms?.email?.kind === "fake" ? config.comms.email : undefined;
1682
- const commsPort = commsEmail ? (commsEmail.port ?? DEFAULT_SANDBOX_CATCH_PORT) : undefined;
1683
- // Hoisted so the finally can drain the catch before teardown; `commsArtifactPath` is the written
1684
- // evidence path folded into the lane outcome.
1685
- let deployedComms;
1686
- let commsArtifactPath;
1687
- let receivingInboxUrl;
1688
- // injectEnv is absent on an adopter-hosted plane (#328): there is no subject env to inject
1689
- // because the operator points their own app at their own catch.
1690
- const commsEnv = commsEmail?.injectEnv !== undefined && commsPort !== undefined
1691
- ? { [commsEmail.injectEnv]: `http://127.0.0.1:${commsPort}` }
1692
- : {};
1693
- // SMTP transport: the same idea as injectEnv, but an app that speaks SMTP needs a host and a port
1694
- // rather than a base URL. The catch accepts any credentials (loopback only), yet many apps refuse
1695
- // to boot unless the user/password vars exist at all, so those are injected when declared.
1696
- const commsSmtpPort = commsEmail?.smtp?.port;
1697
- if (commsEmail?.smtp && commsSmtpPort !== undefined) {
1698
- commsEnv[commsEmail.smtp.hostEnv] = "127.0.0.1";
1699
- commsEnv[commsEmail.smtp.portEnv] = String(commsSmtpPort);
1700
- if (commsEmail.smtp.userEnv)
1701
- commsEnv[commsEmail.smtp.userEnv] = commsEmail.smtp.user ?? "humanish";
1702
- if (commsEmail.smtp.passwordEnv)
1703
- commsEnv[commsEmail.smtp.passwordEnv] = commsEmail.smtp.password ?? "humanish";
1704
- }
1705
- // Persona inbox SURFACE (#297 slice B): the loopback URL the persona opens to read captured mail; the
1706
- // origin-rewrite map (identity on this same-sandbox route, but covers localhost/0.0.0.0 alias skew + an
1707
- // operator-declared linkOrigin); and a disposable background loop that renders the surface DURING the
1708
- // session so the inbox is live when the persona checks. The surface uses its OWN FakeInbox + cursor,
1709
- // independent of the teardown evidence drain (two readers of the append-only NDJSON — no double-count).
1710
- const commsInboxUrl = commsEmail && commsPort !== undefined ? `http://127.0.0.1:${commsPort}/inbox` : undefined;
1711
- const commsOriginMap = commsEmail
1712
- ? buildOriginMap({
1713
- ...(config.subject.serve?.url === undefined ? {} : { internalServeUrl: config.subject.serve.url }),
1714
- reachableBaseUrl: targetUrl,
1715
- ...(commsEmail.linkOrigin === undefined ? {} : { linkOrigin: commsEmail.linkOrigin })
1716
- })
1717
- : [];
1718
- const surfaceRecipients = (commsEmail?.recipients ?? [])
1719
- .filter((recipient) => recipient.address !== undefined)
1720
- .map((recipient) => ({ lane: recipient.lane, address: recipient.address }));
1721
- let surfaceRenderedCount = 0;
1722
- let surfaceDisposed = false;
1723
- let releaseSurface = () => { };
1724
- const surfaceDispose = new Promise((resolve) => { releaseSurface = resolve; });
1725
- let surfaceLoop;
1726
760
  const warnings = [];
1727
761
  const screenshots = [];
1728
762
  const writeScreenshot = makeLaneWriteScreenshot(deps.artifactRoot, spec, screenshots);
1729
- const stateStepRecords = [];
1730
- // Completed-only trail (durationMs/ok are set on completed events, never on started ones):
1731
- // this is what survives into bundle.events. The default/injected sink below sees EVERY event,
1732
- // started and completed alike, so an operator watching stderr sees both halves of each phase.
1733
- const phaseRecords = [];
1734
- const onSubjectPhase = (event) => {
1735
- if (event.ok !== undefined) {
1736
- phaseRecords.push(event);
1737
- }
1738
- (deps.hooks.onPhase ?? defaultSubjectPhaseSink)(event, { laneId: spec.laneId, laneCount: deps.laneCount });
1739
- };
1740
763
  let session;
1741
764
  let sessionError;
1742
- let failureCode;
1743
- let sandboxId;
1744
- // Host-side E2B desktop billed-span endpoints, measured via the injected clock. Captured right
1745
- // after create() succeeds and again in the finally after teardown resolves (both the killed and
1746
- // kept-for-debug paths). This measured span excludes allocation before the acquired handle;
1747
- // a kept/unconfirmed allocation gets an extra unknown lifetime cost line.
1748
- let sandboxCreatedAtMs;
1749
- let sandboxTornDownAtMs;
1750
- let desktopResources;
1751
- let killed = false;
1752
- let streamUrl;
1753
- let subjectCommit;
1754
- let desktopBrowser;
1755
- let launchedBrowserFamily = "unknown";
1756
- let browserLaunchIdentity;
1757
- let browserLaunched = false;
1758
- let initialBrowserGeometry;
1759
- let appliedFidelity;
1760
- let emulatedTargetId;
1761
- let emulationHolderName;
1762
- let browserWindowId;
1763
- let browserTargetId;
1764
- const declaredScreen = declaredScreenForRender(spec.devicePreset, spec.deviceName, spec.resolution);
1765
- let desktopGeometry = {
1766
- screen: {
1767
- requested: { width: spec.resolution[0], height: spec.resolution[1] },
1768
- ...(declaredScreen ? { declared: declaredScreen } : {})
1769
- }
1770
- };
1771
765
  let provisioned = false;
1772
766
  let signaled = false;
1773
767
  const signal = (ok) => {
@@ -1776,641 +770,153 @@ export async function runCuaLane(spec, deps) {
1776
770
  deps.signalProvisioned(ok);
1777
771
  }
1778
772
  };
1779
- let desktopModule;
1780
- let desktop;
773
+ const desktopLane = deps.createDesktopLane?.(spec, warnings) ?? createE2BCuaDesktopLane(spec, deps, warnings);
1781
774
  try {
1782
- desktopModule = await (deps.hooks.loadDesktopModule ?? loadE2BDesktopModule)();
1783
- // Optional custom desktop template (image): present → Sandbox.create(template, opts); absent →
1784
- // the byte-stable Sandbox.create(opts) default (stock `desktop` template).
1785
- desktop = await createDesktopSandbox(desktopModule, {
1786
- apiKey: deps.e2bApiKey,
1787
- requestTimeoutMs: deps.requestTimeoutMs,
1788
- timeoutMs: deps.perLaneSandboxMs,
1789
- metadata: {
1790
- ...CUA_ACTOR_LAB_PROVIDER_METADATA,
1791
- labId: config.id,
1792
- simId: spec.simId,
1793
- laneId: spec.laneId,
1794
- laneIndex: String(spec.laneIndex),
1795
- laneCount: String(deps.laneCount)
1796
- },
1797
- // Env placement per the doctrine: the ACTOR's key never enters the sandbox (the model drives
1798
- // from outside). The SUBJECT's declared env NAMES are provisioned here on the clone route.
1799
- // Three sources, in precedence order: committed non-secret config (subject.envValues), then
1800
- // secret values forwarded from the caller's environment (subject.env), then the harness's own
1801
- // comms wiring, which must win because only it knows the catch's address.
1802
- ...(subjectEnvNames.length > 0 || Object.keys(subjectEnvValues).length > 0 || Object.keys(commsEnv).length > 0
1803
- ? {
1804
- envs: {
1805
- ...subjectEnvValues,
1806
- ...Object.fromEntries(subjectEnvNames.map((name) => [name, env[name]])),
1807
- ...commsEnv
1808
- }
1809
- }
1810
- : {}),
1811
- resolution: spec.resolution,
1812
- dpi: 96,
1813
- lifecycle: { onTimeout: "kill" }
1814
- }, config.execution?.desktop?.template, {
1815
- // The default loader reclaims an acquired handle before retrying failed desktop startup.
1816
- // Its error names the cleanup outcome; pre-construction allocation failures remain unowned.
1817
- onRetry: (reason) => {
1818
- const named = redactText(deps.scrubKnownValues(reason));
1819
- warnings.push(`Sandbox create for lane ${spec.laneId} retried once after a transient provider error (${named}).`);
1820
- onSubjectPhase({ at: new Date(deps.now()).toISOString(), type: "cua-lab.sandbox.create.retry", message: `sandbox create retried once (${named})` });
1821
- }
1822
- });
1823
- sandboxId = desktop.sandboxId;
1824
- // #358 salvage: journal the id to disk before any work — an interrupted run reclaims by
1825
- // exact recorded id (`humanish reclaim`), never by enumerating the account.
1826
- await appendSandboxReceipt(deps.artifactRoot, { at: new Date(deps.now()).toISOString(), laneId: spec.laneId, sandboxId, timeoutMs: deps.perLaneSandboxMs });
1827
- // The billed span starts the instant the sandbox exists.
1828
- sandboxCreatedAtMs = deps.now();
1829
- desktopResources = await observeDesktopResources(desktop);
1830
- if ("reason" in desktopResources) {
1831
- warnings.push(`Desktop resource size unavailable (${desktopResources.reason}); compute cost remains unpriced.`);
1832
- }
1833
- if (deps.hooks.prepareDesktop) {
1834
- await deps.hooks.prepareDesktop(desktop, { laneId: spec.laneId, laneIndex: spec.laneIndex, laneCount: deps.laneCount });
1835
- }
1836
- // Start the in-sandbox email catch BEFORE the subject serve, so the app's send-API base URL (injected
1837
- // into its env at create) resolves the moment it boots. A comms-declared lab that can't stand the
1838
- // catch up is a setup failure (fail closed) rather than silently sending real mail.
1839
- if (deps.receiving) {
1840
- const surface = await deployReceivingInbox(desktop, { leaseId: spec.streamId, requestTimeoutMs: Math.min(deps.requestTimeoutMs, 30_000) });
1841
- receivingInboxUrl = surface.url;
1842
- const email = config.comms?.email;
1843
- try {
1844
- await deps.receiving.attach(spec.laneId, {
1845
- surface,
1846
- allowedOrigins: [...new Set([new URL(targetUrl).origin, ...(email?.allowedOrigins ?? [])])],
1847
- originMap: buildOriginMap({
1848
- ...(config.subject.serve?.url === undefined ? {} : { internalServeUrl: config.subject.serve.url }),
1849
- reachableBaseUrl: targetUrl,
1850
- ...(email?.linkOrigin === undefined ? {} : { linkOrigin: email.linkOrigin })
1851
- })
1852
- });
1853
- commsArtifactPath = "comms/receiving.json";
1854
- }
1855
- catch (error) {
1856
- await surface.stop().catch(() => { });
1857
- throw error;
1858
- }
1859
- }
1860
- if (commsEmail && commsPort !== undefined) {
1861
- deployedComms = await deployCommsCatch(desktop, {
1862
- port: commsPort,
1863
- ...(commsSmtpPort === undefined ? {} : { smtpPort: commsSmtpPort }),
1864
- requestTimeoutMs: deps.requestTimeoutMs
775
+ await desktopLane.prepare();
776
+ // Start the brain BEFORE the first screenshot: the app-server handshake is ~500ms, and it
777
+ // is paid here, while the sandbox is still settling, rather than inside turn one.
778
+ if (deps.hooks.buildProvider) {
779
+ localAgentProvider = await deps.hooks.buildProvider({ config, actor: deps.descriptor, lane: spec });
780
+ }
781
+ else if (deps.localAgent === "codex") {
782
+ appServer = await startAppServerSession({
783
+ ...(spec.reasoningEffort === undefined ? {} : { reasoningEffort: spec.reasoningEffort }),
784
+ ...(config.actors[0]?.model === undefined ? {} : { model: config.actors[0].model }),
785
+ // The persona lives on the THREAD, so it is stated once instead of re-sent every turn.
786
+ baseInstructions: spec.instructions
1865
787
  });
1866
- if (!deployedComms.ready) {
1867
- throw new Error(`comms email catch did not become ready on 127.0.0.1:${commsPort} in the subject sandbox`);
1868
- }
1869
- // Write the EMPTY inbox once up front so the persona's /inbox always resolves to the "No messages
1870
- // yet." page — never a bare 404 — the instant it navigates there, even before any mail arrives OR if
1871
- // the app sends to an address no declared recipient matches (the loop only re-renders on new mail).
1872
- await writeInboxSurface(desktop, deployedComms.surfaceDir, [], { originMap: commsOriginMap, requestTimeoutMs: deps.requestTimeoutMs });
1873
- const deployedRef = deployedComms;
1874
- surfaceLoop = (async () => {
1875
- // Render-first (so even a short session gets a populated inbox), then refresh on a cadence. The
1876
- // cadence uses a REAL timer, NOT the injected instant clock: this loop is unbounded, so an instant
1877
- // sleep would busy-spin and starve the session's own timers. The wait is interruptible by
1878
- // surfaceDispose (and the timer cleared) so teardown never blocks for a full cadence. Each refresh
1879
- // is a full, idempotent rebuild; `surfaceRenderedCount` only advances on a SUCCESSFUL render so a
1880
- // transient failure retries cleanly (no duplicate emails).
1881
- for (;;) {
1882
- try {
1883
- const refreshed = await refreshInboxSurface({
1884
- desktop,
1885
- deployed: deployedRef,
1886
- recipients: surfaceRecipients,
1887
- sinceCount: surfaceRenderedCount,
1888
- originMap: commsOriginMap,
1889
- requestTimeoutMs: deps.requestTimeoutMs
1890
- });
1891
- if (refreshed.rendered)
1892
- surfaceRenderedCount = refreshed.count;
1893
- }
1894
- catch {
1895
- // Never throw into the render loop; the teardown drain + by-id teardown must still run.
1896
- }
1897
- if (surfaceDisposed)
1898
- break;
1899
- await new Promise((resolve) => {
1900
- const timer = setTimeout(resolve, INBOX_SURFACE_CADENCE_MS);
1901
- void surfaceDispose.then(() => { clearTimeout(timer); resolve(); });
1902
- });
1903
- if (surfaceDisposed)
1904
- break;
1905
- }
1906
- })();
1907
- }
1908
- // Per-lane geometry assertion (fail-closed) — the device claim is verified in-sandbox.
1909
- const screenGeometry = await inspectDesktopScreenGeometry({
1910
- desktop,
1911
- laneId: spec.laneId,
1912
- requestedScreen: spec.resolution,
1913
- requestTimeoutMs: deps.requestTimeoutMs
1914
- });
1915
- if (screenGeometry.verified) {
1916
- desktopGeometry = {
1917
- ...desktopGeometry,
1918
- screen: { ...desktopGeometry.screen, verified: screenGeometry.verified }
1919
- };
1920
- }
1921
- if (screenGeometry.warning) {
1922
- warnings.push(screenGeometry.warning);
1923
- desktopGeometry = { ...desktopGeometry, warnings: [screenGeometry.warning] };
1924
- }
1925
- if (screenGeometry.error && deps.screenMismatchPolicy !== "record-evidence") {
1926
- sessionError = screenGeometry.error;
1927
- failureCode = "HUMANISH_CUA_LAB_DEVICE_GEOMETRY";
1928
- }
1929
- else {
1930
- if (screenGeometry.error && screenGeometry.verified) {
1931
- // record-evidence policy: the bundle keeps requested vs verified as separate facts and
1932
- // discloses the divergence instead of failing this lane's world mid-flight.
1933
- const mismatchWarning = deps.scrubKnownValues(`Lane ${spec.laneId} requested a ${spec.resolution[0]}x${spec.resolution[1]} screen but xdpyinfo reports ${screenGeometry.verified.width}x${screenGeometry.verified.height}; recording requested vs verified separately instead of failing the lane closed.`);
1934
- warnings.push(mismatchWarning);
1935
- desktopGeometry = {
1936
- ...desktopGeometry,
1937
- warnings: [...(desktopGeometry.warnings ?? []), mismatchWarning]
1938
- };
1939
- }
1940
- if (desktopCliRoute) {
1941
- // Prepare the runtime and any declared product install, UNKEYED. With install omitted,
1942
- // the participant discovers and installs the product from its public surfaces.
1943
- await provisionDesktopCli(desktop, {
1944
- product: config.subject.product?.name ?? "",
1945
- ...(config.subject.product?.install === undefined ? {} : { install: config.subject.product.install }),
1946
- requestTimeoutMs: deps.requestTimeoutMs,
1947
- scrub: deps.scrubKnownValues,
1948
- onPhase: onSubjectPhase
1949
- });
1950
- }
1951
- if (cloneRoute && serve && subjectRepo) {
1952
- subjectCommit = await provisionCloneSubject(desktop, {
1953
- repo: subjectRepo,
1954
- depth: config.subject.clone?.depth ?? 1,
1955
- serve,
1956
- ...(config.subject.state === undefined ? {} : { state: config.subject.state }),
1957
- hasGithubToken: deps.hasGithubToken,
1958
- requestTimeoutMs: deps.requestTimeoutMs,
1959
- scrub: deps.scrubKnownValues,
1960
- onCommit: (commit) => {
1961
- subjectCommit = commit;
1962
- },
1963
- onStateStep: (record) => {
1964
- stateStepRecords.push(record);
1965
- },
1966
- onPhase: onSubjectPhase,
1967
- ...(deps.hooks.detachedTimers ?? {})
1968
- });
1969
- }
1970
- else if (localTreeRoute && serve && deps.localTreeArchiveBuffer) {
1971
- await provisionLocalTreeSubject(desktop, {
1972
- archiveBuffer: deps.localTreeArchiveBuffer,
1973
- serve,
1974
- ...(config.subject.state === undefined ? {} : { state: config.subject.state }),
1975
- requestTimeoutMs: deps.requestTimeoutMs,
1976
- scrub: deps.scrubKnownValues,
1977
- onStateStep: (record) => {
1978
- stateStepRecords.push(record);
1979
- },
1980
- onPhase: onSubjectPhase,
1981
- ...(deps.hooks.detachedTimers ?? {})
788
+ localAgentProvider = appServer.provider;
789
+ }
790
+ else if (deps.localAgent === "claude") {
791
+ // One session for the whole run, like the codex thread above (#520). The one-shot
792
+ // provider (createLocalAgentProvider) spawned `claude -p` per turn, and every turn
793
+ // started with no memory of the last. HUMANISH_LOCAL_AGENT_ONE_SHOT=1 keeps that path
794
+ // reachable as a MEASUREMENT switch: MemTrapBench (2026-08) reports memory frameworks
795
+ // degrading agent performance by 10-40% on some tasks, so "remembers" has to be measured
796
+ // against "does not" on the same lab, not assumed. The trace records which one ran.
797
+ const oneShot = env.HUMANISH_LOCAL_AGENT_ONE_SHOT !== undefined
798
+ && env.HUMANISH_LOCAL_AGENT_ONE_SHOT !== ""
799
+ && env.HUMANISH_LOCAL_AGENT_ONE_SHOT !== "0";
800
+ if (oneShot) {
801
+ localAgentProvider = createLocalAgentProvider({
802
+ agent: "claude",
803
+ ...(spec.reasoningEffort === undefined ? {} : { reasoningEffort: spec.reasoningEffort }),
804
+ ...(config.actors[0]?.model === undefined ? {} : { model: config.actors[0].model })
1982
805
  });
1983
806
  }
1984
- if (!desktopCliRoute) {
1985
- const requestedFidelity = config.execution?.desktop?.fidelity;
1986
- // A declared camera (#509) is in place before the browser starts: the feed is generated or
1987
- // uploaded first, and a feed that cannot be produced fails the lane closed here.
1988
- const requestedMedia = config.execution?.desktop?.media;
1989
- const mediaEvidence = requestedMedia === undefined
1990
- ? undefined
1991
- : await prepareDesktopMedia(desktop, requestedMedia, config.policies?.mediaPermission ?? "prompt", deps.labCwd, deps.requestTimeoutMs);
1992
- const browserLaunch = await openDesktopBrowserTarget(desktop, targetUrl, deps.requestTimeoutMs, config.execution?.desktop?.browser, [
1993
- ...(requestedFidelity?.mobileEmulation && spec.devicePreset.isMobile
1994
- ? [
1995
- `--user-agent=${requestedFidelity.userAgent ?? DEFAULT_MOBILE_USER_AGENT}`,
1996
- ...(requestedFidelity.touch === false ? [] : ["--touch-events=enabled"])
1997
- ]
1998
- : []),
1999
- ...(mediaEvidence?.flags ?? [])
2000
- ]);
2001
- desktopBrowser = mediaEvidence === undefined
2002
- ? browserLaunch.evidence
2003
- : { requested: config.execution?.desktop?.browser ?? "default", ...(browserLaunch.evidence ?? {}), media: mediaEvidence };
2004
- if (mediaEvidence !== undefined && browserLaunch.family !== "chromium") {
2005
- throw new Error(`execution.desktop.media needs Chrome or Chromium on lane ${spec.laneId} (the fake-device flags are Chromium's); the launched browser family is ${browserLaunch.family}. Set execution.desktop.browser: chrome.`);
2006
- }
2007
- launchedBrowserFamily = browserLaunch.family;
2008
- browserLaunchIdentity = browserLaunch.identity;
2009
- browserLaunched = true;
2010
- await desktop.wait(BROWSER_SETTLE_MS).catch(() => undefined);
2011
- // Mobile fidelity beyond viewport size (#221): applied to the launch page before the
2012
- // geometry capture and the participant's first observation, OUTSIDE the stream/geometry
2013
- // try below (whose catch degrades to a warning): a request that cannot be applied fails
2014
- // the lane closed with the reason.
2015
- // Only lanes on a mobile preset are emulated: a run-wide flag must not hand a desktop or
2016
- // tablet lane an iPhone user agent (the first live proof did exactly that to the desktop
2017
- // newcomer beside the phone lane). Those lanes carry no fidelity block, which is honest.
2018
- const fidelityRequest = config.execution?.desktop?.fidelity;
2019
- if (fidelityRequest?.mobileEmulation && spec.devicePreset.isMobile) {
2020
- if (launchedBrowserFamily !== "chromium") {
2021
- throw new Error(`execution.desktop.fidelity.mobileEmulation needs Chrome or Chromium on lane ${spec.laneId}; the launched browser family is ${launchedBrowserFamily}. Set execution.desktop.browser: chrome.`);
2022
- }
2023
- const applied = await applyMobileEmulation(desktop, deps.requestTimeoutMs, {
2024
- ...(browserLaunchIdentity?.cdpPort === undefined ? {} : { cdpPort: browserLaunchIdentity.cdpPort }),
2025
- ...(browserLaunchIdentity?.profileDir === undefined ? {} : { profileDir: browserLaunchIdentity.profileDir }),
2026
- targetUrl
2027
- }, browserTargetId, {
2028
- width: spec.devicePreset.width,
2029
- height: spec.devicePreset.height,
2030
- deviceScaleFactor: fidelityRequest.deviceScaleFactor ?? spec.devicePreset.deviceScaleFactor,
2031
- touch: fidelityRequest.touch ?? true,
2032
- userAgent: fidelityRequest.userAgent ?? DEFAULT_MOBILE_USER_AGENT
2033
- });
2034
- appliedFidelity = applied.fidelity;
2035
- emulatedTargetId = applied.targetId;
2036
- emulationHolderName = applied.holderName;
2037
- warnings.push(...applied.warnings);
2038
- }
2039
- }
2040
807
  else {
2041
- // A terminal window, opened the way the browser is opened on every other route: the
2042
- // participant arrives at a desktop with the thing they were asked to use already in front
2043
- // of them. They can still open another from the dock — that is the point of a desktop.
2044
- await openDesktopTerminal(desktop, deps.requestTimeoutMs, config.subject.product?.workdir);
2045
- await desktop.wait(BROWSER_SETTLE_MS).catch(() => undefined);
2046
- }
2047
- // Start the brain BEFORE the first screenshot: the app-server handshake is ~500ms, and it
2048
- // is paid here, while the sandbox is still settling, rather than inside turn one.
2049
- if (deps.localAgent === "codex") {
2050
- appServer = await startAppServerSession({
808
+ claudeSession = await startClaudeSession({
2051
809
  ...(spec.reasoningEffort === undefined ? {} : { reasoningEffort: spec.reasoningEffort }),
2052
- ...(config.actors[0]?.model === undefined ? {} : { model: config.actors[0].model }),
2053
- // The persona lives on the THREAD, so it is stated once instead of re-sent every turn.
2054
- baseInstructions: spec.instructions
2055
- });
2056
- localAgentProvider = appServer.provider;
2057
- }
2058
- else if (deps.localAgent === "claude") {
2059
- // One session for the whole run, like the codex thread above (#520). The one-shot
2060
- // provider (createLocalAgentProvider) spawned `claude -p` per turn, and every turn
2061
- // started with no memory of the last. HUMANISH_LOCAL_AGENT_ONE_SHOT=1 keeps that path
2062
- // reachable as a MEASUREMENT switch: MemTrapBench (2026-08) reports memory frameworks
2063
- // degrading agent performance by 10-40% on some tasks, so "remembers" has to be measured
2064
- // against "does not" on the same lab, not assumed. The trace records which one ran.
2065
- const oneShot = env.HUMANISH_LOCAL_AGENT_ONE_SHOT !== undefined
2066
- && env.HUMANISH_LOCAL_AGENT_ONE_SHOT !== ""
2067
- && env.HUMANISH_LOCAL_AGENT_ONE_SHOT !== "0";
2068
- if (oneShot) {
2069
- localAgentProvider = createLocalAgentProvider({
2070
- agent: "claude",
2071
- ...(spec.reasoningEffort === undefined ? {} : { reasoningEffort: spec.reasoningEffort }),
2072
- ...(config.actors[0]?.model === undefined ? {} : { model: config.actors[0].model })
2073
- });
2074
- }
2075
- else {
2076
- claudeSession = await startClaudeSession({
2077
- ...(spec.reasoningEffort === undefined ? {} : { reasoningEffort: spec.reasoningEffort }),
2078
- ...(config.actors[0]?.model === undefined ? {} : { model: config.actors[0].model })
2079
- });
2080
- localAgentProvider = claudeSession.provider;
2081
- }
2082
- }
2083
- // World is ready: release the pipeline gate so the remaining lanes may start.
2084
- provisioned = true;
2085
- signal(true);
2086
- try {
2087
- // No browser means no browser geometry, and none is invented: the CSS-viewport facts a
2088
- // browser reports have no counterpart in a terminal window, and an empty record shaped like
2089
- // a measurement would read as one. The screen geometry above is still verified.
2090
- if (!desktopCliRoute) {
2091
- const browserGeometry = await captureDesktopBrowserGeometry({
2092
- desktop,
2093
- browserFamily: launchedBrowserFamily,
2094
- ...(browserLaunchIdentity === undefined ? {} : { launchIdentity: browserLaunchIdentity }),
2095
- laneId: spec.laneId,
2096
- targetUrl,
2097
- requestedScreen: spec.resolution,
2098
- requestTimeoutMs: deps.requestTimeoutMs
2099
- });
2100
- initialBrowserGeometry = browserGeometry;
2101
- browserWindowId = browserGeometry.browserWindowId;
2102
- browserTargetId = browserGeometry.browserTargetId;
2103
- }
2104
- // The WHOLE desktop, not one window: a person studying a terminal app opens other windows,
2105
- // and a stream bound to the first one would quietly stop being evidence.
2106
- await startDesktopStream(desktop, browserWindowId);
2107
- const candidateStreamUrl = desktop.stream.getUrl({
2108
- authKey: desktop.stream.getAuthKey(),
2109
- autoConnect: true,
2110
- viewOnly: true,
2111
- resize: "scale"
810
+ ...(config.actors[0]?.model === undefined ? {} : { model: config.actors[0].model })
2112
811
  });
2113
- if (typeof candidateStreamUrl === "string" && candidateStreamUrl.trim().length > 0) {
2114
- streamUrl = candidateStreamUrl;
2115
- await deps.hooks.onRuntimeStreamReady?.({
2116
- laneId: spec.laneId,
2117
- sandboxId: desktop.sandboxId,
2118
- simId: spec.simId,
2119
- streamId: spec.streamId,
2120
- url: streamUrl
2121
- });
2122
- }
2123
- else {
2124
- warnings.push("Live desktop stream started but did not return a usable watch URL; Observer will fall back to screenshots.");
2125
- }
812
+ localAgentProvider = claudeSession.provider;
2126
813
  }
2127
- catch (error) {
2128
- warnings.push(`Live desktop stream unavailable (run continues; evidence still captured): ${redactText(deps.scrubKnownValues(toErrorMessage(error)))}`);
2129
- }
2130
- // This is outside the stream's best-effort catch: unusable geometry is a harness failure,
2131
- // never a participant finding about missing controls. Both per-lane and concurrent seats
2132
- // use this route; sequential seats enforce the same capture result in shared-world-lab.
2133
- if (initialBrowserGeometry?.unusable !== undefined) {
2134
- failureCode = "HUMANISH_CUA_LAB_DEVICE_GEOMETRY";
2135
- throw new Error(`${failureCode}: ${initialBrowserGeometry.unusable} Participant actions were not started.`);
2136
- }
2137
- // The FAIL-CLOSED spend cap (execution.caps.maxUsd) is wired into the loop as maxUsd + an
2138
- // injected pure per-turn estimator keyed on the resolved model. Preflight already refused a
2139
- // cap on an unpriced model, so the estimate is measurable whenever a cap is in force. The
2140
- // model id here matches provider.version (openai-responses-cu resolves the default when unset).
2141
- const capModelId = config.actors[0]?.model ?? DEFAULT_OPENAI_CU_MODEL;
2142
- const maxUsd = config.execution?.caps?.maxUsd;
2143
- const sessionOptions = {
2144
- // Tell the persona where its inbox is — but only when comms is live AND this lane has a declared
2145
- // recipient it can actually receive mail into (else it would stall on an inbox that stays
2146
- // empty). Two comms planes, mutually exclusive by parse: the in-sandbox catch humanish
2147
- // deployed, or the adopter-hosted one (#380).
2148
- instructions: deps.receiving && receivingInboxUrl
2149
- ? withInboxMission(spec, receivingInboxUrl, deps.receiving.address(spec.laneId), true).instructions
2150
- : commsEmail && commsInboxUrl && deployedComms?.ready && laneHasInboxRecipient(commsEmail, spec.laneId)
2151
- ? withInboxMission(spec, commsInboxUrl, inboxRecipientFor(commsEmail, spec.laneId)?.address).instructions
2152
- : deps.externalComms && laneHasInboxRecipient(deps.externalComms.email, spec.laneId)
2153
- ? withInboxMission(spec, deps.externalComms.inboxUrl, inboxRecipientFor(deps.externalComms.email, spec.laneId)?.address).instructions
2154
- : spec.instructions,
2155
- persona: spec.persona,
2156
- timeoutMs: deps.timeoutMs,
2157
- // The brain is either a keyed API client or a CLI the operator is already signed in to.
2158
- // Everything below this line — loop, executor, trace, affordances — is identical either
2159
- // way, which is what makes a local-agent run comparable to an API one.
2160
- ...(localAgentProvider === undefined ? {} : { provider: localAgentProvider }),
2161
- openai: {
2162
- apiKey: deps.openaiApiKey,
2163
- ...(config.actors[0]?.model ? { model: config.actors[0].model } : {}),
2164
- // Per-LANE, not per-actor: two lanes at different efforts is the control this exists for.
2165
- ...(spec.reasoningEffort === undefined ? {} : { reasoningEffort: spec.reasoningEffort }),
2166
- ...(spec.maxOutputTokens === undefined ? {} : { maxOutputTokens: spec.maxOutputTokens })
2167
- },
2168
- ...(maxUsd === undefined
2169
- ? {}
2170
- : {
2171
- maxUsd,
2172
- estimateTurnCostUsd: (usage) => estimateActorCost(usage, capModelId).estimatedCostUsd
2173
- }),
2174
- desktop: desktop,
2175
- ...(launchedBrowserFamily === "chromium"
2176
- ? {
2177
- executorOptions: {
2178
- observeBrowserState: makeChromeBrowserStateObserver(desktop, deps.requestTimeoutMs, {
2179
- ...(browserLaunchIdentity?.cdpPort === undefined ? {} : { cdpPort: browserLaunchIdentity.cdpPort }),
2180
- ...(browserLaunchIdentity?.profileDir === undefined ? {} : { profileDir: browserLaunchIdentity.profileDir }),
2181
- targetUrl
2182
- }, browserTargetId,
2183
- // Once per lane: a dark observation channel is a gap in the instrument, and the
2184
- // funnel's NEVER MEASURED count needs this line to explain itself (#514).
2185
- (reason) => {
2186
- warnings.push(`Browser-state observer unavailable for lane ${spec.laneId} (${redactText(deps.scrubKnownValues(reason))}); ` +
2187
- "urlIncludes/urlPathEquals/textIncludes stop conditions and task criteria are NOT being measured this session.");
2188
- }, emulatedTargetId === undefined
2189
- ? undefined
2190
- : {
2191
- emulatedTargetId,
2192
- expectedWidth: spec.devicePreset.width,
2193
- expectTouch: appliedFidelity?.requested.touch === true,
2194
- onDrift: (reason) => {
2195
- warnings.push(`Mobile emulation drift on lane ${spec.laneId}: ${reason} (#623).`);
2196
- },
2197
- onCovered: (coveredTargetId, read) => {
2198
- // A later tab the page itself reported at the phone width: evidence that
2199
- // the emulation followed the participant (#623), kept on the bundle.
2200
- if (appliedFidelity === undefined)
2201
- return;
2202
- appliedFidelity = {
2203
- ...appliedFidelity,
2204
- laterTargets: [...(appliedFidelity.laterTargets ?? []), { targetId: coveredTargetId, ...read }]
2205
- };
2206
- }
2207
- })
2208
- }
2209
- }
2210
- : {}),
2211
- redactScreenshots: deps.redactScreenshots,
2212
- scrubText: deps.scrubKnownValues,
2213
- writeScreenshot,
2214
- ...(spec.idleSteps === undefined ? {} : { idleSteps: spec.idleSteps }),
2215
- ...(spec.noProgressSteps === undefined ? {} : { noProgressSteps: spec.noProgressSteps }),
2216
- ...(spec.stopWhen === undefined ? {} : { stopWhen: spec.stopWhen }),
2217
- ...(spec.dwell === undefined ? {} : { dwell: spec.dwell }),
2218
- ...(spec.tasks === undefined ? {} : { tasks: spec.tasks }),
2219
- // The STUDY budget (#299): this lane notes its own running estimate on the shared ledger
2220
- // and stops when the RUN total crosses the cap — independent of the per-lane maxUsd above.
2221
- ...(deps.runBudget === undefined
2222
- ? {}
2223
- : {
2224
- overRunBudget: (usage) => {
2225
- const estimate = estimateActorCost(usage, capModelId).estimatedCostUsd;
2226
- const totalUsd = deps.runBudget.note(spec.laneId, estimate);
2227
- return totalUsd > deps.runBudget.maxTotalUsd
2228
- ? `study budget reached: the run's estimated model spend $${round6(totalUsd)} crossed execution.caps.maxTotalUsd=$${deps.runBudget.maxTotalUsd}; this lane stops here and sibling lanes stop at their next turn`
2229
- : null;
2230
- }
2231
- }),
2232
- ...(deps.onObservedUrl === undefined ? {} : { onObservedUrl: deps.onObservedUrl }),
2233
- ...(deps.onMessage === undefined ? {} : { onMessage: deps.onMessage }),
2234
- ...(deps.onScreenshot === undefined ? {} : { onScreenshot: deps.onScreenshot }),
2235
- ...(deps.onTrace === undefined
2236
- ? {}
2237
- : {
2238
- // Forwards the RUNNING usage as well: the lane is where both are known, and usage
2239
- // without it never reaches the flush — which is how the live cost stayed unknown.
2240
- onTrace: (items, usage) => deps.onTrace?.(spec.laneId, items, usage)
2241
- })
2242
- };
2243
- session = await deps.runSession(sessionOptions);
2244
814
  }
815
+ // World is ready: release the pipeline gate so the remaining lanes may start.
816
+ provisioned = true;
817
+ signal(true);
818
+ const ready = await desktopLane.openSession();
819
+ // The FAIL-CLOSED spend cap (execution.caps.maxUsd) is wired into the loop as maxUsd + an
820
+ // injected pure per-turn estimator keyed on the resolved model. Preflight already refused a
821
+ // cap on an unpriced model, so the estimate is measurable whenever a cap is in force. The
822
+ // model id here matches provider.version (openai-responses-cu resolves the default when unset).
823
+ const capModelId = config.actors[0]?.model ?? DEFAULT_OPENAI_CU_MODEL;
824
+ const maxUsd = config.execution?.caps?.maxUsd;
825
+ const sessionOptions = {
826
+ instructions: ready.inbox
827
+ ? withInboxMission(spec, ready.inbox.url, ready.inbox.address, ready.inbox.receiving).instructions
828
+ : spec.instructions,
829
+ persona: spec.persona,
830
+ timeoutMs: deps.timeoutMs,
831
+ // The brain is either a keyed API client or a CLI the operator is already signed in to.
832
+ // Everything below this line — loop, executor, trace, affordances — is identical either
833
+ // way, which is what makes a local-agent run comparable to an API one.
834
+ ...(localAgentProvider === undefined ? {} : { provider: localAgentProvider }),
835
+ openai: {
836
+ apiKey: deps.openaiApiKey,
837
+ ...(config.actors[0]?.model ? { model: config.actors[0].model } : {}),
838
+ // Per-LANE, not per-actor: two lanes at different efforts is the control this exists for.
839
+ ...(spec.reasoningEffort === undefined ? {} : { reasoningEffort: spec.reasoningEffort }),
840
+ ...(spec.maxOutputTokens === undefined ? {} : { maxOutputTokens: spec.maxOutputTokens })
841
+ },
842
+ ...(maxUsd === undefined
843
+ ? {}
844
+ : {
845
+ maxUsd,
846
+ estimateTurnCostUsd: (usage) => estimateActorCost(usage, capModelId).estimatedCostUsd
847
+ }),
848
+ executor: ready.executor,
849
+ redactScreenshots: deps.redactScreenshots,
850
+ scrubText: deps.scrubKnownValues,
851
+ writeScreenshot,
852
+ ...(spec.idleSteps === undefined ? {} : { idleSteps: spec.idleSteps }),
853
+ ...(spec.noProgressSteps === undefined ? {} : { noProgressSteps: spec.noProgressSteps }),
854
+ ...(spec.stopWhen === undefined ? {} : { stopWhen: spec.stopWhen }),
855
+ ...(spec.dwell === undefined ? {} : { dwell: spec.dwell }),
856
+ ...(spec.tasks === undefined ? {} : { tasks: spec.tasks }),
857
+ // The STUDY budget (#299): this lane notes its own running estimate on the shared ledger
858
+ // and stops when the RUN total crosses the cap — independent of the per-lane maxUsd above.
859
+ ...(deps.runBudget === undefined
860
+ ? {}
861
+ : {
862
+ overRunBudget: (usage) => {
863
+ const estimate = estimateActorCost(usage, capModelId).estimatedCostUsd;
864
+ const totalUsd = deps.runBudget.note(spec.laneId, estimate);
865
+ return totalUsd > deps.runBudget.maxTotalUsd
866
+ ? `study budget reached: the run's estimated model spend $${round6(totalUsd)} crossed execution.caps.maxTotalUsd=$${deps.runBudget.maxTotalUsd}; this lane stops here and sibling lanes stop at their next turn`
867
+ : null;
868
+ }
869
+ }),
870
+ ...(deps.onObservedUrl === undefined ? {} : { onObservedUrl: deps.onObservedUrl }),
871
+ ...(deps.onMessage === undefined ? {} : { onMessage: deps.onMessage }),
872
+ ...(deps.onScreenshot === undefined ? {} : { onScreenshot: deps.onScreenshot }),
873
+ ...(deps.onTrace === undefined
874
+ ? {}
875
+ : {
876
+ // Forwards the RUNNING usage as well: the lane is where both are known, and usage
877
+ // without it never reaches the flush — which is how the live cost stayed unknown.
878
+ onTrace: (items, usage, metadata) => deps.onTrace?.(spec.laneId, items, usage, metadata)
879
+ })
880
+ };
881
+ session = await deps.runSession(sessionOptions);
2245
882
  }
2246
883
  catch (error) {
2247
884
  sessionError = redactText(deps.scrubKnownValues(toErrorMessage(error)));
2248
885
  }
2249
886
  finally {
2250
- // The local brain owns a process. Close it before anything else can throw: a leaked
2251
- // app-server per lane would outlive the run and keep a thread open on the operator's plan.
2252
- appServer?.close();
2253
- await claudeSession?.close();
2254
- // Stop the mid-run inbox-surface loop FIRST — before the teardown evidence drain below — so the two
2255
- // `cat`s never overlap and the final surface state is deterministic. A surface failure can never
2256
- // block teardown (the loop body is fully try/caught and this await is on its already-caught promise).
2257
- surfaceDisposed = true;
2258
- releaseSurface();
2259
- if (surfaceLoop)
2260
- await surfaceLoop.catch(() => undefined);
2261
- if (!provisioned) {
2262
- signal(false);
887
+ try {
888
+ await localAgentProvider?.close?.();
2263
889
  }
2264
- if (desktop && desktopModule) {
2265
- if (browserLaunched) {
2266
- const finalGeometry = await captureDesktopBrowserGeometry({
2267
- desktop,
2268
- browserFamily: launchedBrowserFamily,
2269
- ...(browserLaunchIdentity === undefined ? {} : { launchIdentity: browserLaunchIdentity }),
2270
- ...(browserWindowId === undefined ? {} : { browserWindowId }),
2271
- ...(browserTargetId === undefined ? {} : { browserTargetId }),
2272
- laneId: spec.laneId,
2273
- targetUrl,
2274
- requestedScreen: spec.resolution,
2275
- requestTimeoutMs: deps.requestTimeoutMs,
2276
- pagePreference: "active",
2277
- resize: false
2278
- }).catch((error) => ({
2279
- warnings: [`Final browser geometry measurement failed for lane ${spec.laneId}: ${redactText(deps.scrubKnownValues(toErrorMessage(error)))}`]
2280
- }));
2281
- // Chosen capture rule: final-if-it-measured-anything, else launch-time. A final capture
2282
- // that measured EITHER field wins whole, so a partial final capture omits fields the
2283
- // launch-time capture had (honest omission); only a final capture that measured NOTHING
2284
- // falls back to the launch-time capture.
2285
- const chosenGeometry = finalGeometry.browserWindow !== undefined || finalGeometry.viewport !== undefined
2286
- ? finalGeometry
2287
- : initialBrowserGeometry ?? finalGeometry;
2288
- const geometryWarnings = [...new Set([...(initialBrowserGeometry?.warnings ?? []), ...chosenGeometry.warnings].map((warning) => deps.scrubKnownValues(warning)))];
2289
- warnings.push(...geometryWarnings);
2290
- // The emulation holder's own log, after its announce line: which later targets it
2291
- // attached to, what it sent, and any reply that came back as an error (#623). Read while
2292
- // the sandbox is alive; the first live proof had no way to say what the holder did.
2293
- if (appliedFidelity !== undefined && emulationHolderName !== undefined) {
2294
- const holderLog = await readDetachedLog(desktop, emulationHolderName, deps.requestTimeoutMs).catch(() => "");
2295
- const lines = holderLog.split("\n").map((line) => line.trim()).filter((line) => line.startsWith("{")).slice(1, 51);
2296
- if (lines.length > 0)
2297
- appliedFidelity = { ...appliedFidelity, holderLog: lines.map((line) => deps.scrubKnownValues(line)) };
2298
- }
2299
- desktopGeometry = {
2300
- screen: desktopGeometry.screen,
2301
- ...(chosenGeometry.browserWindow === undefined ? {} : { browserWindow: chosenGeometry.browserWindow }),
2302
- ...(chosenGeometry.viewport === undefined ? {} : { viewport: chosenGeometry.viewport }),
2303
- ...(appliedFidelity === undefined ? {} : { fidelity: appliedFidelity }),
2304
- ...((desktopGeometry.warnings?.length ?? 0) + geometryWarnings.length === 0
2305
- ? {}
2306
- : { warnings: [...(desktopGeometry.warnings ?? []), ...geometryWarnings] })
2307
- };
2308
- }
2309
- if (deps.receiving) {
2310
- try {
2311
- await deps.receiving.finishParticipant(spec.laneId);
2312
- }
2313
- catch {
2314
- warnings.push("Real email finalization is incomplete. Inspect communication cleanup with humanish comms recover.");
2315
- }
2316
- }
2317
- // Off-app comms evidence (#297): before this lane's sandbox is torn down, drain everything the
2318
- // in-sandbox catch captured, route it into a host fake inbox addressed to the declared
2319
- // recipients, and write the digest-only thread artifact. Wrapped so a drain failure NEVER
2320
- // breaks teardown — the sandbox must still be killed either way. Runs only for a ready catch.
2321
- if (commsEmail && deployedComms?.ready) {
2322
- try {
2323
- const commsChannel = new FakeInbox();
2324
- const commsInboxes = [];
2325
- for (const recipient of commsEmail.recipients ?? []) {
2326
- if (recipient.address !== undefined) {
2327
- commsInboxes.push(await commsChannel.provisionAddress(recipient.lane, recipient.address));
2328
- }
2329
- }
2330
- const collected = await collectCommsThread({
2331
- desktop,
2332
- deployed: deployedComms,
2333
- channel: commsChannel,
2334
- inboxes: commsInboxes,
2335
- requestTimeoutMs: deps.requestTimeoutMs
2336
- });
2337
- if (collected.artifact) {
2338
- const path = deps.laneCount === 1 ? "comms/thread.json" : `comms/${spec.streamId}.thread.json`;
2339
- await writeContainedOutputFile(deps.artifactRoot, path, `${JSON.stringify(collected.artifact, null, 2)}\n`, "utf8");
2340
- commsArtifactPath = path;
2341
- }
2342
- else if (collected.captured > 0) {
2343
- // Captured mail that matched no declared recipient must not vanish silently (invariant 6:
2344
- // honest signals): tell the operator to declare comms.email.recipients[].address to match
2345
- // the address the app actually sends to (e.g. the one the persona surface will sign up with).
2346
- warnings.push(`Comms catch captured ${collected.captured} email send(s) but none matched a declared recipient inbox — no comms evidence written. Declare comms.email.recipients[].address to match the address the app sends to.`);
2347
- }
2348
- else {
2349
- // Zero captures is the silent-broken shape (#351): the app never posted to the catch at
2350
- // all, so the personas stared at an empty inbox. Most common cause: the app does not
2351
- // actually read the declared injectEnv var for its email API base URL.
2352
- const transportHint = commsEmail.smtp
2353
- ? `Verify the app reads ${commsEmail.smtp.hostEnv}/${commsEmail.smtp.portEnv} for its SMTP host and port`
2354
- : `Verify the app reads ${commsEmail.injectEnv} for its email API base URL (an SDK that ignores it sends real mail or throws)`;
2355
- warnings.push(`Comms catch captured ZERO email sends — the app never delivered mail through the catch. ${transportHint} and that the flow reached an email step.`);
2356
- }
2357
- }
2358
- catch (error) {
2359
- warnings.push(`Comms evidence collection failed (run continues; sandbox still torn down): ${redactText(deps.scrubKnownValues(toErrorMessage(error)))}`);
2360
- }
2361
- }
2362
- const failed = sessionError !== undefined || session === undefined;
2363
- // Each route's own keep flag gates its own lane only: a clone.keep can never leak into
2364
- // a local-tree lane's teardown decision, and vice versa.
2365
- const keepReason = cloneRoute && config.subject.clone?.keep === true
2366
- ? "subject.clone.keep"
2367
- : localTreeRoute && config.subject.localTree?.keep === true
2368
- ? "subject.localTree.keep"
2369
- : undefined;
2370
- const keepForDebug = keepReason !== undefined && failed;
2371
- if (keepForDebug) {
2372
- warnings.push(`Sandbox ${desktop.sandboxId} kept for debugging (${keepReason} on failure); reclaim it via E2B or it will be killed on its server-side timeout.`);
2373
- }
2374
- else if (typeof desktopModule.Sandbox.kill === "function") {
2375
- try {
2376
- await desktopModule.Sandbox.kill(desktop.sandboxId, { requestTimeoutMs: 60_000 });
2377
- killed = true;
2378
- }
2379
- catch (error) {
2380
- warnings.push(`Sandbox teardown failed (server-side kill-on-timeout will reclaim it): ${redactText(deps.scrubKnownValues(toErrorMessage(error)))}`);
2381
- }
2382
- }
2383
- else {
2384
- warnings.push("Installed @e2b/desktop SDK does not expose Sandbox.kill; server-side kill-on-timeout will reclaim the sandbox.");
2385
- }
2386
- // Close the observed span. A kept or unconfirmed sandbox can still accrue compute cost;
2387
- // the summary records that remaining lifetime as unknown instead of calling this complete.
2388
- sandboxTornDownAtMs = deps.now();
2389
- // The lane's live stream is now a dead page whichever teardown path ran (killed, kept, or
2390
- // kill-failed-awaiting-TTL) — tell the watch overlay so the tile falls back to recorded
2391
- // evidence instead of "sandbox not found" (#357). Guarded: a viewer callback must never
2392
- // break teardown.
2393
- if (streamUrl !== undefined) {
2394
- try {
2395
- await deps.hooks.onRuntimeStreamEnded?.({ laneId: spec.laneId, simId: spec.simId, streamId: spec.streamId });
2396
- }
2397
- catch {
2398
- // viewer-side only; nothing to record
2399
- }
2400
- }
890
+ catch {
891
+ warnings.push("Model provider cleanup is unconfirmed.");
892
+ sessionError ??= "Model provider cleanup is unconfirmed.";
893
+ }
894
+ try {
895
+ appServer?.close();
896
+ }
897
+ catch {
898
+ warnings.push('Codex session cleanup failed; desktop cleanup will still run.');
899
+ }
900
+ try {
901
+ await claudeSession?.close();
902
+ }
903
+ catch {
904
+ warnings.push('Claude session cleanup failed; desktop cleanup will still run.');
905
+ }
906
+ try {
907
+ if (!provisioned)
908
+ signal(false);
909
+ }
910
+ finally {
911
+ await desktopLane.finalize({ failed: sessionError !== undefined || session === undefined });
2401
912
  }
2402
913
  }
2403
- // Host-side approximation of the E2B desktop's billed lifetime; feeds the desktop-minute cost
2404
- // estimate. Never negative.
2405
- const desktopDurationMs = sandboxCreatedAtMs !== undefined && sandboxTornDownAtMs !== undefined
2406
- ? Math.max(0, sandboxTornDownAtMs - sandboxCreatedAtMs)
2407
- : undefined;
2408
914
  if (session) {
2409
915
  // Per-lane model-token cost ESTIMATE, attached to the trace before it is persisted (the model
2410
916
  // id is authoritative here — provider.version). Kept at the lab boundary so the pure loop
2411
917
  // never depends on the operator rate table. estimateActorCost declares absent (null) for an
2412
918
  // unknown rate / missing usage rather than guessing.
2413
- session.trace.estimatedCost = estimateActorCost(session.trace.tokenUsage, session.trace.ids.model);
919
+ session.trace.estimatedCost = estimateActorCostForExecution(session.trace.tokenUsage, session.trace.ids.model, session.trace.executionProfile);
2414
920
  await writeContainedOutputFile(deps.artifactRoot, spec.traceArtifactPath, `${JSON.stringify(session.trace, null, 2)}\n`, "utf8");
2415
921
  if (session.trace.redaction.screenshots === "raw") {
2416
922
  warnings.push("Screenshots are full-fidelity (raw) for local use — the bundle stays in gitignored .humanish and nothing scans these pixels; review them before sharing anywhere. Set policies.redactScreenshots: true to blur a share-as-is bundle.");
@@ -2435,24 +941,13 @@ export async function runCuaLane(spec, deps) {
2435
941
  spec,
2436
942
  ...(session ? { session } : {}),
2437
943
  ...(sessionError === undefined ? {} : { sessionError }),
2438
- ...(sandboxId === undefined ? {} : { sandboxId }),
2439
- ...(desktopDurationMs === undefined ? {} : { desktopDurationMs }),
2440
- ...(desktopResources === undefined ? {} : { desktopResources }),
2441
- killed,
2442
- streamUrlPresent: streamUrl !== undefined,
944
+ ...desktopLane.snapshot(),
2443
945
  screenshots,
2444
- ...(subjectCommit === undefined ? {} : { subjectCommit }),
2445
- ...(desktopBrowser === undefined ? {} : { desktopBrowser }),
2446
- desktopGeometry,
2447
- stateStepRecords,
2448
- phaseRecords,
2449
946
  warnings,
2450
947
  noEngagement,
2451
948
  selfReportedBlocker,
2452
949
  reportedFriction,
2453
950
  harnessError,
2454
- ...(failureCode === undefined ? {} : { failureCode }),
2455
- ...(commsArtifactPath === undefined ? {} : { commsArtifactPath })
2456
951
  };
2457
952
  }
2458
953
  /** Run the single IN-PROCESS lane (a custom executor + provider; NO E2B). Always one lane. */
@@ -2462,9 +957,10 @@ async function runInProcessLane(spec, deps) {
2462
957
  const writeScreenshot = makeLaneWriteScreenshot(deps.artifactRoot, spec, screenshots);
2463
958
  let session;
2464
959
  let sessionError;
960
+ let provider;
2465
961
  try {
2466
962
  const executor = await deps.hooks.buildExecutor({ config: deps.config, actor: deps.descriptor, appUrl: deps.appUrl });
2467
- const provider = await deps.hooks.buildProvider({ config: deps.config, actor: deps.descriptor });
963
+ provider = await deps.hooks.buildProvider({ config: deps.config, actor: deps.descriptor, lane: spec });
2468
964
  const sessionOptions = {
2469
965
  instructions: spec.instructions,
2470
966
  persona: spec.persona,
@@ -2474,6 +970,7 @@ async function runInProcessLane(spec, deps) {
2474
970
  redactScreenshots: deps.redactScreenshots,
2475
971
  scrubText: deps.scrubKnownValues,
2476
972
  writeScreenshot,
973
+ ...(deps.onTrace === undefined ? {} : { onTrace: (items, usage, metadata) => deps.onTrace?.(spec.laneId, items, usage, metadata) }),
2477
974
  ...(spec.stopWhen === undefined ? {} : { stopWhen: spec.stopWhen }),
2478
975
  ...(spec.dwell === undefined ? {} : { dwell: spec.dwell }),
2479
976
  ...(spec.tasks === undefined ? {} : { tasks: spec.tasks })
@@ -2483,6 +980,14 @@ async function runInProcessLane(spec, deps) {
2483
980
  catch (error) {
2484
981
  sessionError = redactText(deps.scrubKnownValues(toErrorMessage(error)));
2485
982
  }
983
+ finally {
984
+ try {
985
+ await provider?.close?.();
986
+ }
987
+ catch {
988
+ sessionError ??= "Model provider cleanup is unconfirmed.";
989
+ }
990
+ }
2486
991
  if (session) {
2487
992
  await writeContainedOutputFile(deps.artifactRoot, spec.traceArtifactPath, `${JSON.stringify(session.trace, null, 2)}\n`, "utf8");
2488
993
  if (session.trace.redaction.screenshots === "raw") {
@@ -2918,6 +1423,9 @@ async function runCuaActorLabInScope(options) {
2918
1423
  if (localAppSubject && !inProcessRoute) {
2919
1424
  return fail("HUMANISH_CUA_LAB_LOCAL_APP_NO_EXECUTOR", "subject.source: local-app requires a library caller to supply cuaHooks.buildExecutor + buildProvider; there is no built-in driver for an in-process JS contract. (Drive the app via runLab(..., { cuaHooks: { buildExecutor, buildProvider } }).)", descriptor.id);
2920
1425
  }
1426
+ if (config.subject.source === "app-url" && config.execution?.target === "local" && !hooks.createDesktopLane) {
1427
+ return fail("HUMANISH_CUA_LAB_LOCAL_APP_NO_EXECUTOR", "Local browser studies require a configured local desktop runtime.", descriptor.id);
1428
+ }
2921
1429
  // Re-enforce the fan-out cross-validation (library API surface): lanes XOR count/laneFocus,
2922
1430
  // device XOR raw resolution, cap, unique ids, allowPublicTargets+N>1, clone.fanout.
2923
1431
  const fanoutReason = cuaLaneValidationReason(config);
@@ -3010,8 +1518,8 @@ async function runCuaActorLabInScope(options) {
3010
1518
  // the local-agent route uses a CLI the operator has already signed in to.
3011
1519
  if (!dryRun && !inProcessRoute) {
3012
1520
  const missingKeys = [
3013
- ...(openaiApiKey || localAgentRoute ? [] : ["OPENAI_API_KEY"]),
3014
- ...(e2bApiKey ? [] : ["E2B_API_KEY"])
1521
+ ...(openaiApiKey || localAgentRoute || hooks.buildProvider ? [] : ["OPENAI_API_KEY"]),
1522
+ ...(e2bApiKey || hooks.createDesktopLane ? [] : ["E2B_API_KEY"])
3015
1523
  ];
3016
1524
  if (missingKeys.length > 0) {
3017
1525
  // The moment someone new actually hits the wall. If a signed-in coding agent is sitting
@@ -3028,7 +1536,7 @@ async function runCuaActorLabInScope(options) {
3028
1536
  : "";
3029
1537
  return fail("HUMANISH_CUA_LAB_KEYS_MISSING", `Live computer-use labs need ${missingKeys.join(" and ")} in the environment (values are never persisted). ${describeMissingKeys(missingKeys, env)}${suggestion}`, descriptor.id);
3030
1538
  }
3031
- if (localAgentRoute) {
1539
+ if (localAgentRoute && !hooks.buildProvider) {
3032
1540
  // Refuse HERE, before a sandbox exists. "codex is not installed" discovered after the
3033
1541
  // machine is paid for is the same information delivered at the worst possible moment.
3034
1542
  const available = await detectLocalAgents({ env });
@@ -3125,7 +1633,8 @@ async function runCuaActorLabInScope(options) {
3125
1633
  let flushLiveTrace;
3126
1634
  let stopLiveFlush;
3127
1635
  const deps = {
3128
- onTrace: (laneId, items, usage) => flushLiveTrace?.(laneId, items, usage),
1636
+ ...(hooks.createDesktopLane ? { createDesktopLane: hooks.createDesktopLane } : {}),
1637
+ onTrace: (laneId, items, usage, metadata) => flushLiveTrace?.(laneId, items, usage, metadata),
3129
1638
  config,
3130
1639
  descriptor,
3131
1640
  appUrl,
@@ -3241,6 +1750,7 @@ async function runCuaActorLabInScope(options) {
3241
1750
  // Running token usage per lane, so a run in flight can price itself instead of reporting the
3242
1751
  // cost as unknown until the moment it ends.
3243
1752
  const liveUsageByStream = new Map();
1753
+ const liveMetadataByStream = new Map();
3244
1754
  // The rate the running usage prices at. Usage without its model is not a cost, so both travel
3245
1755
  // together or neither does.
3246
1756
  const modelForLiveCost = config.actors[0]?.model ?? DEFAULT_OPENAI_CU_MODEL;
@@ -3279,6 +1789,7 @@ async function runCuaActorLabInScope(options) {
3279
1789
  // The model too: usage without the rate it prices at is not a cost.
3280
1790
  ids: { model: modelForLiveCost }
3281
1791
  }),
1792
+ ...liveMetadataByStream.get(stream.id),
3282
1793
  items: [...liveItems]
3283
1794
  }
3284
1795
  };
@@ -3309,7 +1820,7 @@ async function runCuaActorLabInScope(options) {
3309
1820
  flushTimer.unref?.();
3310
1821
  }
3311
1822
  };
3312
- flushLiveTrace = (laneId, items, usage) => {
1823
+ flushLiveTrace = (laneId, items, usage, metadata) => {
3313
1824
  // An empty snapshot (the initial observation on a frameless route) carries no
3314
1825
  // evidence worth a disk write; the first real item triggers the first flush.
3315
1826
  if (items.length === 0)
@@ -3318,6 +1829,8 @@ async function runCuaActorLabInScope(options) {
3318
1829
  if (streamId === undefined)
3319
1830
  return;
3320
1831
  liveItemsByStream.set(streamId, items.slice());
1832
+ if (metadata !== undefined)
1833
+ liveMetadataByStream.set(streamId, metadata);
3321
1834
  if (usage !== undefined)
3322
1835
  liveUsageByStream.set(streamId, usage);
3323
1836
  flushDirty = true;
@@ -3736,6 +2249,7 @@ function buildSingleLaneBundle(args) {
3736
2249
  persona: spec.persona,
3737
2250
  resolution: spec.resolution,
3738
2251
  desktopRoute: !args.inProcessRoute,
2252
+ feedbackSubstrate: args.inProcessRoute ? "local-filesystem" : args.config.execution?.target === "local" ? "local-desktop" : "e2b-desktop",
3739
2253
  ...(outcome?.desktopGeometry === undefined ? {} : { desktopGeometry: outcome.desktopGeometry }),
3740
2254
  isMobile: spec.devicePreset.isMobile,
3741
2255
  runId: args.runId,
@@ -3779,300 +2293,6 @@ function buildSingleLaneBundle(args) {
3779
2293
  phaseEvents: outcome?.phaseRecords ?? []
3780
2294
  });
3781
2295
  }
3782
- /**
3783
- * Shared post-populate provisioning pipeline (clone AND local-tree routes): (install) ->
3784
- * state(before-build) -> (build) -> state(before-start) -> detached start -> readiness probe ->
3785
- * state(after-ready). Both provisioning routes populate SUBJECT_DIR by different means (git
3786
- * clone vs. upload+extract) and then run this identical pipeline unchanged.
3787
- *
3788
- * State steps run through the same detached primitive as serve steps (author-trusted, the
3789
- * "serve commands are author-trusted" corollary) under the reserved `subject-state-<name>`
3790
- * label prefix, so a step name can never collide with subject-clone/subject-extract/install/
3791
- * build/start. after-ready steps complete BEFORE the caller opens the browser: the actor never
3792
- * drives a half-seeded subject and seeding never eats the session budget.
3793
- */
3794
- async function runSubjectServePipeline(desktop, args) {
3795
- const timers = {
3796
- ...(args.now === undefined ? {} : { now: args.now }),
3797
- ...(args.sleep === undefined ? {} : { sleep: args.sleep })
3798
- };
3799
- const now = args.now ?? Date.now;
3800
- const refresh = args.onPhaseComplete ?? (() => Promise.resolve());
3801
- const stateSteps = args.state?.seed ?? [];
3802
- const runStateSteps = async (when) => {
3803
- const steps = stateSteps.filter((step) => (step.when ?? "before-start") === when);
3804
- if (steps.length === 0) {
3805
- // No declared steps for this group: no boundary to report (avoids empty-group noise on
3806
- // every run, since before-build/before-start/after-ready are always called).
3807
- return;
3808
- }
3809
- const groupStartedAt = now();
3810
- emitPhaseStarted(args.onPhase, now, `state.${when}`, `running subject state seed steps (${when})`);
3811
- for (const step of steps) {
3812
- const stepTimeoutMs = step.timeoutMs ?? DEFAULT_STATE_STEP_TIMEOUT_MS;
3813
- const startedAt = now();
3814
- const result = await runDetachedStep(desktop, {
3815
- name: `subject-state-${step.name}`,
3816
- command: step.command,
3817
- cwd: SUBJECT_DIR,
3818
- timeoutMs: stepTimeoutMs,
3819
- requestTimeoutMs: args.requestTimeoutMs,
3820
- ...timers
3821
- });
3822
- args.onStateStep?.({
3823
- name: step.name,
3824
- when,
3825
- // Digest only (sha256-16): the command text never persists: the lab YAML in the
3826
- // consumer's repo is the plaintext source of truth.
3827
- commandDigest: commandDigestOf(step.command),
3828
- ok: result.ok,
3829
- ...(result.exitCode === undefined ? {} : { exitCode: result.exitCode }),
3830
- ...(result.timedOut ? { timedOut: true } : {}),
3831
- durationMs: Math.max(0, now() - startedAt)
3832
- });
3833
- if (!result.ok) {
3834
- emitPhaseCompleted(args.onPhase, now, groupStartedAt, `state.${when}`, false, `subject state seed steps failed (${when})`);
3835
- // Fail closed with the existing scrub-before-truncate tail chain: literal scrub of
3836
- // every provisioned value PRE-truncation, then pattern redaction + cap in tailOf.
3837
- throw new Error(`subject state step "${step.name}" ${result.timedOut ? `timed out after ${stepTimeoutMs}ms` : `failed (exit ${result.exitCode})`}: ${tailOf(args.scrub(result.logTail))}`);
3838
- }
3839
- }
3840
- emitPhaseCompleted(args.onPhase, now, groupStartedAt, `state.${when}`, true, `subject state seed steps complete (${when})`);
3841
- };
3842
- // Provide the runtime the pipeline needs before running it (#371). The stock desktop template
3843
- // ships python3 and curl but no Node, so an `npm install` here used to die at exit 127 after the
3844
- // sandbox was already paid for. Probe-first, so a template that ships its own Node pays nothing.
3845
- const serveCommands = [args.serve.install, args.serve.build, args.serve.start];
3846
- if (needsNodeRuntime(serveCommands)) {
3847
- const runtimeStartedAt = now();
3848
- emitPhaseStarted(args.onPhase, now, "runtime", "providing the Node runtime the serve pipeline needs");
3849
- const bootstrap = await runProvisioningStepWithOneRetry(desktop, {
3850
- name: "subject-runtime-node",
3851
- command: nodeBootstrapCommand(),
3852
- cwd: SUBJECT_DIR,
3853
- timeoutMs: args.serve.installTimeoutMs ?? INSTALL_TIMEOUT_MS,
3854
- requestTimeoutMs: args.requestTimeoutMs,
3855
- timers,
3856
- retryPhase: "runtime-retry",
3857
- retryMessage: "Node runtime bootstrap",
3858
- onPhase: args.onPhase,
3859
- now
3860
- });
3861
- let ok = bootstrap.ok;
3862
- const corepack = ok ? corepackCommandFor(serveCommands) : undefined;
3863
- if (corepack) {
3864
- const pm = await runDetachedStep(desktop, {
3865
- name: "subject-runtime-pm",
3866
- command: corepack,
3867
- cwd: SUBJECT_DIR,
3868
- timeoutMs: args.serve.installTimeoutMs ?? INSTALL_TIMEOUT_MS,
3869
- requestTimeoutMs: args.requestTimeoutMs,
3870
- ...timers
3871
- });
3872
- ok = pm.ok;
3873
- }
3874
- emitPhaseCompleted(args.onPhase, now, runtimeStartedAt, "runtime", ok, ok ? "Node runtime ready" : "could not provide a Node runtime");
3875
- if (!ok) {
3876
- throw new Error(`the subject's serve pipeline needs a Node runtime and this desktop template has none, and bootstrapping one failed${bootstrap.attempts === 2 ? " twice" : ""}: ${tailOf(args.scrub(bootstrap.logTail))}. Use execution.desktop.template with an image that ships Node, or change serve.install to a runtime the template provides.`);
3877
- }
3878
- }
3879
- if (args.serve.install) {
3880
- const installStartedAt = now();
3881
- emitPhaseStarted(args.onPhase, now, "install", "installing subject dependencies");
3882
- const install = await runProvisioningStepWithOneRetry(desktop, {
3883
- name: "subject-install",
3884
- command: args.serve.install,
3885
- cwd: SUBJECT_DIR,
3886
- timeoutMs: args.serve.installTimeoutMs ?? INSTALL_TIMEOUT_MS,
3887
- requestTimeoutMs: args.requestTimeoutMs,
3888
- timers,
3889
- retryPhase: "install-retry",
3890
- retryMessage: "subject install",
3891
- onPhase: args.onPhase,
3892
- now
3893
- });
3894
- emitPhaseCompleted(args.onPhase, now, installStartedAt, "install", install.ok, install.ok
3895
- ? install.attempts === 2
3896
- ? "subject dependencies installed (on the second attempt)"
3897
- : "subject dependencies installed"
3898
- : install.attempts === 2
3899
- ? "subject install failed twice"
3900
- : "subject install failed");
3901
- if (!install.ok) {
3902
- // Lead with the line a person can act on; npm's own trace follows it (#602).
3903
- const headline = install.timedOut
3904
- ? `subject install timed out after ${args.serve.installTimeoutMs ?? INSTALL_TIMEOUT_MS}ms`
3905
- : install.attempts === 2
3906
- ? `subject install failed twice (exit ${install.firstExitCode ?? "null"}, then exit ${install.exitCode ?? "null"}); the sandbox could not complete serve.install`
3907
- : `subject install failed (exit ${install.exitCode ?? "null"})`;
3908
- throw new Error(`${headline}: ${tailOf(args.scrub(install.logTail))}`);
3909
- }
3910
- await refresh();
3911
- }
3912
- // before-build: after install, before build (builds that read seeded state, e.g. SSG).
3913
- // When no build is declared this simply precedes start: equivalent to before-start.
3914
- await runStateSteps("before-build");
3915
- await refresh();
3916
- if (args.serve.build) {
3917
- const buildStartedAt = now();
3918
- emitPhaseStarted(args.onPhase, now, "build", "building subject");
3919
- const build = await runDetachedStep(desktop, {
3920
- name: "subject-build",
3921
- command: args.serve.build,
3922
- cwd: SUBJECT_DIR,
3923
- timeoutMs: args.serve.buildTimeoutMs ?? BUILD_TIMEOUT_MS,
3924
- requestTimeoutMs: args.requestTimeoutMs,
3925
- ...timers
3926
- });
3927
- emitPhaseCompleted(args.onPhase, now, buildStartedAt, "build", build.ok, build.ok ? "subject build complete" : "subject build failed");
3928
- if (!build.ok) {
3929
- throw new Error(`subject build ${build.timedOut ? "timed out" : `failed (exit ${build.exitCode})`}: ${tailOf(args.scrub(build.logTail))}`);
3930
- }
3931
- await refresh();
3932
- }
3933
- // before-start (the default phase): migrations, SQL/file fixtures, an in-sandbox DB server
3934
- // (`sudo service postgresql start && pg_isready` is a bounded step; the daemon it forks is
3935
- // reclaimed by the sandbox lifecycle like everything else).
3936
- await runStateSteps("before-start");
3937
- await refresh();
3938
- await startDetachedProcess(desktop, {
3939
- name: "subject-start",
3940
- command: args.serve.start,
3941
- cwd: SUBJECT_DIR,
3942
- requestTimeoutMs: args.requestTimeoutMs
3943
- });
3944
- // Fire-and-forget: startDetachedProcess never waits for the long-lived server to exit, so
3945
- // there is no matching completed event here (no ok/durationMs to report yet); readiness is
3946
- // the next boundary.
3947
- args.onPhase?.({ at: isoNow(now), type: "cua-lab.subject.serve.started", message: "subject server launched (detached)" });
3948
- const readyStartedAt = now();
3949
- emitPhaseStarted(args.onPhase, now, "ready", "waiting for subject to become ready");
3950
- const ready = await probeUrl(desktop, args.serve.url, {
3951
- timeoutMs: args.serve.readyTimeoutMs ?? DEFAULT_READY_TIMEOUT_MS,
3952
- requestTimeoutMs: args.requestTimeoutMs,
3953
- ...timers
3954
- });
3955
- emitPhaseCompleted(args.onPhase, now, readyStartedAt, "ready", ready, ready ? "subject is ready" : "subject did not become ready in time");
3956
- if (!ready) {
3957
- const startLog = await readDetachedLog(desktop, "subject-start", args.requestTimeoutMs).catch(() => "");
3958
- throw new Error(`subject did not answer at ${args.serve.url} within ${args.serve.readyTimeoutMs ?? DEFAULT_READY_TIMEOUT_MS}ms; server log tail: ${tailOf(args.scrub(startLog))}`);
3959
- }
3960
- // after-ready: fixture loading through the RUNNING app (loopback curl from in-sandbox:
3961
- // steps are author-trusted provisioning, not actors, so no new URL policy surface). These
3962
- // complete before the caller opens the browser and the session timer starts.
3963
- await runStateSteps("after-ready");
3964
- await refresh();
3965
- }
3966
- /**
3967
- * Provision a clone subject inside the sandbox: clone → the shared serve pipeline
3968
- * (install → state(before-build) → build → state(before-start) → start → readiness
3969
- * probe → state(after-ready)). Returns the latest subject HEAD after successful
3970
- * provisioning. Throws (with a capped log tail for the caller to redact) on any failing step:
3971
- * the lab persists that as a failed-evidence bundle.
3972
- *
3973
- * Auth: when GITHUB_TOKEN is among the declared subject env names, the clone authenticates
3974
- * via an Authorization header computed IN-SANDBOX from the provisioned env: the token never
3975
- * appears in the script text, the process argv beyond the transient git call, the clone URL,
3976
- * or .git/config.
3977
- */
3978
- export async function provisionCloneSubject(desktop, args) {
3979
- const timers = {
3980
- ...(args.now === undefined ? {} : { now: args.now }),
3981
- ...(args.sleep === undefined ? {} : { sleep: args.sleep })
3982
- };
3983
- const now = args.now ?? Date.now;
3984
- let latestCommit;
3985
- const refreshCommit = async () => {
3986
- const head = await desktop.commands.run(`git -C ${SUBJECT_DIR} rev-parse HEAD 2>/dev/null || true`, { requestTimeoutMs: args.requestTimeoutMs });
3987
- const commit = (head.stdout ?? "").trim() || undefined;
3988
- if (commit) {
3989
- latestCommit = commit;
3990
- args.onCommit?.(commit);
3991
- }
3992
- };
3993
- const cloneCommand = args.hasGithubToken
3994
- ? `auth=$(printf 'x-access-token:%s' "$GITHUB_TOKEN" | base64 -w0) && git -c http.extraHeader="Authorization: Basic $auth" clone --depth ${args.depth} https://github.com/${args.repo}.git ${SUBJECT_DIR}`
3995
- : `git clone --depth ${args.depth} https://github.com/${args.repo}.git ${SUBJECT_DIR}`;
3996
- const cloneStartedAt = now();
3997
- emitPhaseStarted(args.onPhase, now, "clone", "cloning subject repository");
3998
- const clone = await runDetachedStep(desktop, {
3999
- name: "subject-clone",
4000
- command: cloneCommand,
4001
- timeoutMs: CLONE_TIMEOUT_MS,
4002
- requestTimeoutMs: args.requestTimeoutMs,
4003
- ...timers
4004
- });
4005
- emitPhaseCompleted(args.onPhase, now, cloneStartedAt, "clone", clone.ok, clone.ok ? "subject repository cloned" : "subject clone failed");
4006
- if (!clone.ok) {
4007
- throw new Error(`subject clone ${clone.timedOut ? "timed out" : `failed (exit ${clone.exitCode})`}: ${tailOf(args.scrub(clone.logTail))}`);
4008
- }
4009
- await refreshCommit();
4010
- await runSubjectServePipeline(desktop, {
4011
- serve: args.serve,
4012
- ...(args.state === undefined ? {} : { state: args.state }),
4013
- requestTimeoutMs: args.requestTimeoutMs,
4014
- scrub: args.scrub,
4015
- ...(args.onStateStep === undefined ? {} : { onStateStep: args.onStateStep }),
4016
- ...(args.onPhase === undefined ? {} : { onPhase: args.onPhase }),
4017
- onPhaseComplete: refreshCommit,
4018
- ...timers
4019
- });
4020
- return latestCommit;
4021
- }
4022
- /**
4023
- * Provision a local-tree subject inside the sandbox: upload the once-per-run packed archive
4024
- * (identical bytes across every fan-out lane) → extract it into SUBJECT_DIR → the
4025
- * same shared serve pipeline provisionCloneSubject uses. Unlike the clone route there is no
4026
- * in-sandbox git refresh: the archive excludes .git entirely (see source-archive.ts), so
4027
- * subject identity is the host-side LocalTreeArchive captured at pack time, never anything
4028
- * resolved in-sandbox.
4029
- */
4030
- export async function provisionLocalTreeSubject(desktop, args) {
4031
- const timers = {
4032
- ...(args.now === undefined ? {} : { now: args.now }),
4033
- ...(args.sleep === undefined ? {} : { sleep: args.sleep })
4034
- };
4035
- const now = args.now ?? Date.now;
4036
- const uploadStartedAt = now();
4037
- emitPhaseStarted(args.onPhase, now, "upload", "uploading packed local-tree archive");
4038
- try {
4039
- await withOneRetryOnTransientE2BError(() => desktop.files.write(LOCAL_TREE_REMOTE_ARCHIVE_PATH, args.archiveBuffer, {
4040
- requestTimeoutMs: args.requestTimeoutMs,
4041
- useOctetStream: true
4042
- }), {
4043
- onRetry: (reason) => emitPhaseStarted(args.onPhase, now, "upload-retry", `local-tree archive upload retried once (${tailOf(args.scrub(reason))})`),
4044
- ...(args.sleep === undefined ? {} : { sleep: args.sleep })
4045
- });
4046
- }
4047
- catch (error) {
4048
- emitPhaseCompleted(args.onPhase, now, uploadStartedAt, "upload", false, "local-tree archive upload failed");
4049
- throw new Error(`subject-upload failed: ${tailOf(args.scrub(toErrorMessage(error)))}`);
4050
- }
4051
- emitPhaseCompleted(args.onPhase, now, uploadStartedAt, "upload", true, "local-tree archive uploaded");
4052
- const extractCommand = `rm -rf ${SUBJECT_DIR} && mkdir -p ${SUBJECT_DIR} && tar -xzf ${LOCAL_TREE_REMOTE_ARCHIVE_PATH} -C ${SUBJECT_DIR} && rm -f ${LOCAL_TREE_REMOTE_ARCHIVE_PATH}`;
4053
- const extractStartedAt = now();
4054
- emitPhaseStarted(args.onPhase, now, "extract", "extracting local-tree archive");
4055
- const extract = await runDetachedStep(desktop, {
4056
- name: "subject-extract",
4057
- command: extractCommand,
4058
- timeoutMs: CLONE_TIMEOUT_MS,
4059
- requestTimeoutMs: args.requestTimeoutMs,
4060
- ...timers
4061
- });
4062
- emitPhaseCompleted(args.onPhase, now, extractStartedAt, "extract", extract.ok, extract.ok ? "local-tree archive extracted" : "local-tree archive extraction failed");
4063
- if (!extract.ok) {
4064
- throw new Error(`subject extract ${extract.timedOut ? "timed out" : `failed (exit ${extract.exitCode})`}: ${tailOf(args.scrub(extract.logTail))}`);
4065
- }
4066
- await runSubjectServePipeline(desktop, {
4067
- serve: args.serve,
4068
- ...(args.state === undefined ? {} : { state: args.state }),
4069
- requestTimeoutMs: args.requestTimeoutMs,
4070
- scrub: args.scrub,
4071
- ...(args.onStateStep === undefined ? {} : { onStateStep: args.onStateStep }),
4072
- ...(args.onPhase === undefined ? {} : { onPhase: args.onPhase }),
4073
- ...timers
4074
- });
4075
- }
4076
2296
  /**
4077
2297
  * Default local-tree packing implementation: createLocalTreeArchive(root, opts) on the host,
4078
2298
  * then a single read of the produced archive file into an ArrayBuffer for upload. The DI seam
@@ -4092,10 +2312,6 @@ export async function defaultPackLocalTree(args) {
4092
2312
  await rm(path.dirname(archive.archivePath), { recursive: true, force: true }).catch(() => undefined);
4093
2313
  return { archive, buffer };
4094
2314
  }
4095
- /** sha256 hex of the exact command string, first 16 chars (the promptDigest convention). */
4096
- export function commandDigestOf(command) {
4097
- return digestText(command, 16);
4098
- }
4099
2315
  /**
4100
2316
  * Resolve the bundle's state marker from the declaration and what actually ran.
4101
2317
  * Precedence: external declared → "unpinned" (seed records, if any, stay attached — a
@@ -4151,10 +2367,6 @@ function describeSubjectState(state, dryRun) {
4151
2367
  return "external-public (operator-declared, operator-owned public deployment; neither provisioned nor seeded)";
4152
2368
  }
4153
2369
  }
4154
- // The in-sandbox `tail -c` upstream is a fundamental log-tail limit we cannot redact past.
4155
- function tailOf(log) {
4156
- return redactedTail(log, ERROR_TAIL_CHARS);
4157
- }
4158
2370
  export function buildCuaCostSummary(args) {
4159
2371
  const breakdown = [];
4160
2372
  let sumInput = 0;
@@ -4188,7 +2400,8 @@ export function buildCuaCostSummary(args) {
4188
2400
  ratesAsOf: null
4189
2401
  });
4190
2402
  }
4191
- const est = lane.trace.estimatedCost;
2403
+ const est = lane.trace.executionProfile?.billing === "account-unknown"
2404
+ ? estimateActorCostForExecution(usage, lane.trace.ids.model, lane.trace.executionProfile) : lane.trace.estimatedCost;
4192
2405
  if (!est) {
4193
2406
  continue;
4194
2407
  }
@@ -4264,7 +2477,7 @@ export function buildCuaCostSummary(args) {
4264
2477
  }
4265
2478
  const estimatedTotalUsd = anyKnown ? round6(knownSum) : null;
4266
2479
  const estimateNote = estimatedTotalUsd === null
4267
- ? `No priced spend lines this run — every cost line is DECLARED ABSENT (unknown rate / no usage / no duration); nothing is guessed. Add a rate to src/pricing.ts to estimate this model.`
2480
+ ? `No priced spend lines this run — every cost line is DECLARED ABSENT (unknown rate / no usage / no duration); nothing is guessed. ${args.lanes.some(lane => lane.trace.executionProfile?.billing === "account-unknown") ? "Account billing remains unknown; API prices do not measure account spend." : "Add a rate to src/pricing.ts to estimate this model."}`
4268
2481
  : `Estimated ${estimatedTotalUsd} USD total${anyNull ? " (LOWER BOUND — some lines unmeasured/unpriced)" : ""}${placeholder ? "; includes PLACEHOLDER rate(s) — confirm before trusting the magnitude" : ""}. Every figure is an ESTIMATE (rates as of ${minRatesAsOf} — the OLDEST contributing rate, since an aggregate is only as fresh as its stalest input), a rate-table multiply, NOT an authoritative provider charge.`;
4269
2482
  const note = estimateNote + ((args.desktops?.length ?? 0) > 0
4270
2483
  ? args.desktops.some(usage => usage.observation !== undefined && "resources" in usage.observation)
@@ -4279,7 +2492,14 @@ export function buildCuaCostSummary(args) {
4279
2492
  fullyEstimated: !anyNull,
4280
2493
  placeholder,
4281
2494
  breakdown,
4282
- tokenUsage: { input: sumInput, output: sumOutput, total: sumInput + sumOutput },
2495
+ tokenUsage: args.lanes.some(lane => lane.trace.executionProfile?.billing === "account-unknown") ? {
2496
+ ...(args.lanes.some(lane => lane.trace.tokenUsage?.input !== undefined) ? { input: sumInput } : {}),
2497
+ ...(args.lanes.some(lane => lane.trace.tokenUsage?.output !== undefined) ? { output: sumOutput } : {}),
2498
+ ...(args.lanes.every(lane => lane.trace.tokenUsage?.input !== undefined && lane.trace.tokenUsage?.output !== undefined &&
2499
+ lane.trace.interactionUsageIncomplete !== true && lane.trace.debrief?.usageReported !== false &&
2500
+ (lane.trace.executionProfile === undefined || lane.trace.providerRequests?.every(r => r.usageComplete) === true))
2501
+ ? { total: sumInput + sumOutput } : {})
2502
+ } : { input: sumInput, output: sumOutput, total: sumInput + sumOutput },
4283
2503
  desktopMinutes: args.desktops === undefined ? args.desktopMinutes ?? null
4284
2504
  : args.desktops.some(usage => usage.minutes !== undefined)
4285
2505
  ? round6(args.desktops.reduce((sum, usage) => sum + (usage.minutes ?? 0), 0)) : null,
@@ -4674,7 +2894,7 @@ export function buildCuaBundle(args) {
4674
2894
  scenarioId: `cua-${args.labId}`,
4675
2895
  adapterId: args.labId,
4676
2896
  goal: redactText(args.mission),
4677
- substrate: args.desktopRoute === false ? "local-filesystem" : "e2b-desktop",
2897
+ substrate: args.feedbackSubstrate ?? (args.desktopRoute === false ? "local-filesystem" : "e2b-desktop"),
4678
2898
  lanes: [{
4679
2899
  laneId: args.laneId ?? "lane-01",
4680
2900
  streamId: "stream-001",
@@ -5170,7 +3390,7 @@ export function buildCuaFanoutBundle(args) {
5170
3390
  scenarioId: `cua-${config.id}`,
5171
3391
  adapterId: config.id,
5172
3392
  goal: redactText(specs[0].evidenceInstructions ?? specs[0].instructions),
5173
- substrate: "e2b-desktop",
3393
+ substrate: config.execution?.target === "local" ? "local-desktop" : "e2b-desktop",
5174
3394
  lanes: specs.map((spec, index) => {
5175
3395
  const outcome = outcomes?.[index];
5176
3396
  return {