humanish 0.96.1 → 0.98.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (193) hide show
  1. package/AGENTS.md +86 -79
  2. package/CONTRIBUTING.md +7 -2
  3. package/README.md +11 -2
  4. package/dist/actor-contract.d.ts +35 -1
  5. package/dist/actor-contract.js +38 -0
  6. package/dist/actor-contract.js.map +1 -1
  7. package/dist/adapter-extension.js +1 -0
  8. package/dist/adapter-extension.js.map +1 -1
  9. package/dist/automatic-analysis-config.d.ts +15 -5
  10. package/dist/automatic-analysis-config.js +25 -4
  11. package/dist/automatic-analysis-config.js.map +1 -1
  12. package/dist/automatic-study-analysis.js +4 -2
  13. package/dist/automatic-study-analysis.js.map +1 -1
  14. package/dist/browser-control-client.d.ts +14 -0
  15. package/dist/browser-control-client.js +134 -0
  16. package/dist/browser-control-client.js.map +1 -0
  17. package/dist/browser-control-dispatcher.d.ts +14 -0
  18. package/dist/browser-control-dispatcher.js +109 -0
  19. package/dist/browser-control-dispatcher.js.map +1 -0
  20. package/dist/browser-control-protocol.d.ts +371 -0
  21. package/dist/browser-control-protocol.js +155 -0
  22. package/dist/browser-control-protocol.js.map +1 -0
  23. package/dist/browser-control-transport.d.ts +24 -0
  24. package/dist/browser-control-transport.js +156 -0
  25. package/dist/browser-control-transport.js.map +1 -0
  26. package/dist/comms-lease-store.d.ts +1 -0
  27. package/dist/comms-lease-store.js +9 -3
  28. package/dist/comms-lease-store.js.map +1 -1
  29. package/dist/computer-use-actor.d.ts +2 -2
  30. package/dist/computer-use-actor.js +6 -1
  31. package/dist/computer-use-actor.js.map +1 -1
  32. package/dist/computer-use.d.ts +23 -1
  33. package/dist/computer-use.js +253 -70
  34. package/dist/computer-use.js.map +1 -1
  35. package/dist/cua-actor-lab.d.ts +24 -289
  36. package/dist/cua-actor-lab.js +203 -1983
  37. package/dist/cua-actor-lab.js.map +1 -1
  38. package/dist/cua-desktop-lane.d.ts +35 -0
  39. package/dist/cua-desktop-lane.js +13 -0
  40. package/dist/cua-desktop-lane.js.map +1 -0
  41. package/dist/cua-executor-error.d.ts +31 -0
  42. package/dist/cua-executor-error.js +48 -0
  43. package/dist/cua-executor-error.js.map +1 -0
  44. package/dist/cua-provider-error.d.ts +12 -0
  45. package/dist/cua-provider-error.js +28 -0
  46. package/dist/cua-provider-error.js.map +1 -0
  47. package/dist/desktop-session.d.ts +41 -0
  48. package/dist/desktop-session.js +46 -0
  49. package/dist/desktop-session.js.map +1 -0
  50. package/dist/doctor-lab.d.ts +8 -1
  51. package/dist/doctor-lab.js +40 -8
  52. package/dist/doctor-lab.js.map +1 -1
  53. package/dist/e2b-cua-desktop.d.ts +3 -0
  54. package/dist/e2b-cua-desktop.js +675 -0
  55. package/dist/e2b-cua-desktop.js.map +1 -0
  56. package/dist/e2b-cua-provisioning.d.ts +311 -0
  57. package/dist/e2b-cua-provisioning.js +1213 -0
  58. package/dist/e2b-cua-provisioning.js.map +1 -0
  59. package/dist/e2b-desktop-executor.d.ts +1 -24
  60. package/dist/e2b-desktop-executor.js +2 -127
  61. package/dist/e2b-desktop-executor.js.map +1 -1
  62. package/dist/e2b-desktop-session.d.ts +7 -0
  63. package/dist/e2b-desktop-session.js +29 -0
  64. package/dist/e2b-desktop-session.js.map +1 -0
  65. package/dist/e2b-terminal-lab.js +1 -0
  66. package/dist/e2b-terminal-lab.js.map +1 -1
  67. package/dist/frame-signature.d.ts +24 -0
  68. package/dist/frame-signature.js +128 -0
  69. package/dist/frame-signature.js.map +1 -0
  70. package/dist/guest-bootstrap.d.ts +43 -0
  71. package/dist/guest-bootstrap.js +240 -0
  72. package/dist/guest-bootstrap.js.map +1 -0
  73. package/dist/guest-browser-tools.d.ts +8 -0
  74. package/dist/guest-browser-tools.js +66 -0
  75. package/dist/guest-browser-tools.js.map +1 -0
  76. package/dist/guest-chromium-text.d.ts +27 -0
  77. package/dist/guest-chromium-text.js +281 -0
  78. package/dist/guest-chromium-text.js.map +1 -0
  79. package/dist/guest-desktop-executor.d.ts +22 -0
  80. package/dist/guest-desktop-executor.js +177 -0
  81. package/dist/guest-desktop-executor.js.map +1 -0
  82. package/dist/guest-desktop-native.d.ts +14 -0
  83. package/dist/guest-desktop-native.js +131 -0
  84. package/dist/guest-desktop-native.js.map +1 -0
  85. package/dist/guest-runtime-desktop.d.ts +35 -0
  86. package/dist/guest-runtime-desktop.js +231 -0
  87. package/dist/guest-runtime-desktop.js.map +1 -0
  88. package/dist/guest-runtime-main.d.ts +1 -0
  89. package/dist/guest-runtime-main.js +31 -0
  90. package/dist/guest-runtime-main.js.map +1 -0
  91. package/dist/guest-runtime-revision.d.ts +1 -0
  92. package/dist/guest-runtime-revision.js +3 -0
  93. package/dist/guest-runtime-revision.js.map +1 -0
  94. package/dist/guest-runtime.d.ts +25 -0
  95. package/dist/guest-runtime.js +96 -0
  96. package/dist/guest-runtime.js.map +1 -0
  97. package/dist/index.d.ts +1 -1
  98. package/dist/lab-config.js +10 -3
  99. package/dist/lab-config.js.map +1 -1
  100. package/dist/lab-engine.js +6 -0
  101. package/dist/lab-engine.js.map +1 -1
  102. package/dist/lab-summary.d.ts +5 -1
  103. package/dist/lab-summary.js +5 -0
  104. package/dist/lab-summary.js.map +1 -1
  105. package/dist/local-agent-cli.js +1 -1
  106. package/dist/local-agent-cli.js.map +1 -1
  107. package/dist/local-firecracker-desktop.d.ts +13 -0
  108. package/dist/local-firecracker-desktop.js +150 -0
  109. package/dist/local-firecracker-desktop.js.map +1 -0
  110. package/dist/local-firecracker-study.d.ts +9 -0
  111. package/dist/local-firecracker-study.js +93 -0
  112. package/dist/local-firecracker-study.js.map +1 -0
  113. package/dist/local-runtime-config.d.ts +6 -0
  114. package/dist/local-runtime-config.js +56 -0
  115. package/dist/local-runtime-config.js.map +1 -0
  116. package/dist/local-runtime-release.d.ts +3 -0
  117. package/dist/local-runtime-release.js +8 -0
  118. package/dist/local-runtime-release.js.map +1 -0
  119. package/dist/local-runtime.d.ts +25 -0
  120. package/dist/local-runtime.js +113 -0
  121. package/dist/local-runtime.js.map +1 -0
  122. package/dist/observer-app.html +4 -4
  123. package/dist/pricing.d.ts +22 -1
  124. package/dist/pricing.js +22 -0
  125. package/dist/pricing.js.map +1 -1
  126. package/dist/program.js +50 -9
  127. package/dist/program.js.map +1 -1
  128. package/dist/restricted-codex-analysis.d.ts +15 -0
  129. package/dist/restricted-codex-analysis.js +13 -0
  130. package/dist/restricted-codex-analysis.js.map +1 -0
  131. package/dist/restricted-codex-participant-policy.d.ts +39 -0
  132. package/dist/restricted-codex-participant-policy.js +69 -0
  133. package/dist/restricted-codex-participant-policy.js.map +1 -0
  134. package/dist/restricted-codex-participant-run.d.ts +20 -0
  135. package/dist/restricted-codex-participant-run.js +78 -0
  136. package/dist/restricted-codex-participant-run.js.map +1 -0
  137. package/dist/restricted-codex-participant.d.ts +14 -0
  138. package/dist/restricted-codex-participant.js +178 -0
  139. package/dist/restricted-codex-participant.js.map +1 -0
  140. package/dist/restricted-codex-policy.d.ts +56 -0
  141. package/dist/restricted-codex-policy.js +151 -0
  142. package/dist/restricted-codex-policy.js.map +1 -0
  143. package/dist/restricted-codex-session.d.ts +19 -0
  144. package/dist/restricted-codex-session.js +413 -0
  145. package/dist/restricted-codex-session.js.map +1 -0
  146. package/dist/restricted-codex-transport.d.ts +58 -0
  147. package/dist/restricted-codex-transport.js +233 -0
  148. package/dist/restricted-codex-transport.js.map +1 -0
  149. package/dist/run-detail.js +4 -2
  150. package/dist/run-detail.js.map +1 -1
  151. package/dist/run.d.ts +12 -5
  152. package/dist/run.js +17 -1
  153. package/dist/run.js.map +1 -1
  154. package/dist/shared-world-lab.js +2 -2
  155. package/dist/shared-world-lab.js.map +1 -1
  156. package/dist/study-analysis-codex-config.d.ts +11 -0
  157. package/dist/study-analysis-codex-config.js +34 -0
  158. package/dist/study-analysis-codex-config.js.map +1 -0
  159. package/dist/study-analysis-engine.d.ts +6 -3
  160. package/dist/study-analysis-engine.js +19 -10
  161. package/dist/study-analysis-engine.js.map +1 -1
  162. package/dist/study-analysis-job.d.ts +3 -2
  163. package/dist/study-analysis-job.js +1 -1
  164. package/dist/study-analysis-job.js.map +1 -1
  165. package/dist/study-analysis-provider.d.ts +4 -2
  166. package/dist/study-analysis-provider.js +1 -1
  167. package/dist/study-analysis-provider.js.map +1 -1
  168. package/dist/study-analysis-service.d.ts +3 -0
  169. package/dist/study-analysis-service.js +24 -6
  170. package/dist/study-analysis-service.js.map +1 -1
  171. package/dist/study-analysis-validation.d.ts +43 -19
  172. package/dist/study-analysis-validation.js +23 -10
  173. package/dist/study-analysis-validation.js.map +1 -1
  174. package/dist/study-analysis.d.ts +27 -2
  175. package/dist/study-costs.js +6 -0
  176. package/dist/study-costs.js.map +1 -1
  177. package/dist/tui-app.js +102 -102
  178. package/docs/architecture/browser-control.md +117 -0
  179. package/docs/architecture/desktop-sessions.md +80 -0
  180. package/docs/architecture/guest-desktop.md +87 -0
  181. package/docs/architecture/local-browser-runtime.md +100 -0
  182. package/docs/architecture/restricted-codex-analysis.md +91 -0
  183. package/docs/architecture/runtime-broker-core.md +30 -0
  184. package/docs/contracts/schemas.md +1 -1
  185. package/docs/contracts/study-analysis.md +44 -2
  186. package/docs/goals/current.md +25 -9
  187. package/docs/product/automatic-analysis.md +24 -3
  188. package/docs/product/open-source-install-experience.md +7 -0
  189. package/docs/ramp/README.md +30 -13
  190. package/docs/release/0.97.0-codex-account-analysis.md +45 -0
  191. package/docs/release/0.98.0-local-browser-studies.md +31 -0
  192. package/package.json +4 -2
  193. package/skills/humanish/SKILL.md +39 -0
@@ -0,0 +1,1213 @@
1
+ // Hosted desktop setup primitives. The lane adapter owns their lifecycle.
2
+ import { readFile } from "node:fs/promises";
3
+ import path from "node:path";
4
+ import { CHROMIUM_EVIDENCE_HYGIENE_FLAGS, chromiumEvidenceProfilePreferencesJson } from "./browser-evidence-hygiene.js";
5
+ import { chromeCdpProbeCommand, parseChromeCdpProbeOutput } from "./chrome-cdp-probe.js";
6
+ import { runDesktopCommandOrThrow, toErrorMessage } from "./command-failure.js";
7
+ import {} from "./device-presets.js";
8
+ import { withOneRetryOnTransientE2BError } from "./e2b-desktop-launch.js";
9
+ import { probeUrl, readDetachedLog, runDetachedStep, startDetachedProcess } from "./e2b-detached.js";
10
+ import { isHttpUrl } from "./lab-config.js";
11
+ import { digestText, redactText, redactedTail } from "./redaction.js";
12
+ import {} from "./run.js";
13
+ import { corepackCommandFor, needsNodeRuntime, nodeBootstrapCommand } from "./subject-runtime.js";
14
+ import { TERMINAL_NODE_BOOTSTRAP_COMMAND } from "./terminal-node-bootstrap.js";
15
+ export const CUA_ACTOR_LAB_PROVIDER_METADATA = {
16
+ mode: "cua-actor-lab",
17
+ tool: "humanish"
18
+ };
19
+ // Settle after opening the browser, before the first screenshot — long enough for a cold
20
+ // browser + page load to paint (2s captured a blank desktop; the render empirically needs ~6-9s).
21
+ export const BROWSER_SETTLE_MS = 8_000;
22
+ /** Where a lane's synthetic camera feed lives inside the sandbox: a tmpfs the sandbox user can
23
+ * write, and a path that contains neither /tmp/ nor /home/, which the public-safety scan reads
24
+ * as an operator's local path (this one is the harness's own and belongs in the bundle). */
25
+ export const SANDBOX_MEDIA_DIR = "/dev/shm/humanish-media";
26
+ export const SANDBOX_CAMERA_PATH = `${SANDBOX_MEDIA_DIR}/camera.y4m`;
27
+ /** The synthetic feed: ffmpeg's test pattern, 640x480 at 10 fps, six seconds (about 28 MB of
28
+ * raw Y4M on the tmpfs), looped by Chrome's fake capture device. */
29
+ export const SYNTHETIC_CAMERA_COMMAND = `mkdir -p ${SANDBOX_MEDIA_DIR} && ffmpeg -y -loglevel error -f lavfi -i testsrc=size=640x480:rate=10 -t 6 -pix_fmt yuv420p ${SANDBOX_CAMERA_PATH}`;
30
+ /**
31
+ * Put the declared camera feed in the sandbox and return the Chromium flags that present it as a
32
+ * capture device (#509). Fails CLOSED: a feed that cannot be produced (no ffmpeg on the image, an
33
+ * unreadable host file) is named before the browser launches, because a participant told it has
34
+ * a camera and finds none reports the instrument's gap as the product's.
35
+ */
36
+ export async function prepareDesktopMedia(desktop, media, permission, cwd, requestTimeoutMs, readHostFile = (absolutePath) => readFile(absolutePath)) {
37
+ if (media.microphone !== undefined) {
38
+ throw new Error("execution.desktop.media.microphone.source injection is unsupported; the declared microphone file cannot be delivered, including on custom templates.");
39
+ }
40
+ const flags = [];
41
+ let camera;
42
+ if (media.camera !== undefined) {
43
+ if (media.camera.source === "synthetic") {
44
+ const made = await desktop.commands.run(SYNTHETIC_CAMERA_COMMAND, { requestTimeoutMs, timeoutMs: 60_000 });
45
+ if (made.exitCode !== undefined && made.exitCode !== 0) {
46
+ throw new Error(`the synthetic camera feed could not be generated on this desktop image (ffmpeg exited ${made.exitCode}: ${tailOf(made.stderr ?? made.stdout ?? "")}); give execution.desktop.media.camera.source a .y4m file instead`);
47
+ }
48
+ camera = { source: "synthetic", file: SANDBOX_CAMERA_PATH };
49
+ }
50
+ else {
51
+ const absolutePath = path.resolve(cwd, media.camera.source);
52
+ let bytes;
53
+ try {
54
+ bytes = await readHostFile(absolutePath);
55
+ }
56
+ catch (error) {
57
+ throw new Error(`execution.desktop.media.camera.source could not be read (${toErrorMessage(error)})`);
58
+ }
59
+ if (bytes.length > 64 * 1024 * 1024) {
60
+ throw new Error(`execution.desktop.media.camera.source is ${bytes.length} bytes; the camera feed is capped at 64 MiB`);
61
+ }
62
+ await desktop.commands.run(`mkdir -p ${SANDBOX_MEDIA_DIR}`, { requestTimeoutMs, timeoutMs: 15_000 });
63
+ const payload = bytes.buffer.slice(bytes.byteOffset, bytes.byteOffset + bytes.byteLength);
64
+ await desktop.files.write(SANDBOX_CAMERA_PATH, payload, { requestTimeoutMs, useOctetStream: true });
65
+ camera = { source: "file", file: SANDBOX_CAMERA_PATH };
66
+ }
67
+ flags.push("--use-fake-device-for-media-stream", `--use-file-for-fake-video-capture=${SANDBOX_CAMERA_PATH}`);
68
+ }
69
+ if (permission === "granted")
70
+ flags.push("--use-fake-ui-for-media-stream");
71
+ return { ...(camera === undefined ? {} : { camera }), permission, flags };
72
+ }
73
+ export const SUBJECT_DIR = "/home/user/subject";
74
+ // Remote path for the once-per-run packed local-tree archive; removed by the extract step
75
+ // after it unpacks into SUBJECT_DIR.
76
+ const LOCAL_TREE_REMOTE_ARCHIVE_PATH = "/home/user/.humanish-source.tar.gz";
77
+ const CLONE_TIMEOUT_MS = 5 * 60_000;
78
+ const INSTALL_TIMEOUT_MS = 10 * 60_000;
79
+ const BUILD_TIMEOUT_MS = 10 * 60_000;
80
+ const DEFAULT_READY_TIMEOUT_MS = 180_000;
81
+ // Per-step budget for subject.state seed steps; each step's declared (or default) budget is
82
+ // also summed into the default sandbox deadline so seeding never eats the session's room.
83
+ export const DEFAULT_STATE_STEP_TIMEOUT_MS = 5 * 60_000;
84
+ // How much of a failing step's log tail rides the (redacted) error message.
85
+ const ERROR_TAIL_CHARS = 2000;
86
+ /** Mid-run inbox-surface render cadence (ms). Coarse enough that the per-tick `cat` + file writes stay
87
+ * cheap; fine enough that a verification email is visible seconds after the app sends it. */
88
+ export const INBOX_SURFACE_CADENCE_MS = 2500;
89
+ /**
90
+ * The DECLARED preset to record alongside the rendered screen, or undefined when the preset
91
+ * rendered faithfully.
92
+ *
93
+ * `desktopGeometry.screen.verified` compares the FLOORED number with itself, so on its own a
94
+ * floored run is indistinguishable from a faithful one: a reader sees requested 500 / verified 500
95
+ * and concludes a 500-wide screen was asked for. Recording the declared preset is what makes
96
+ * "the preset width did not render" legible in the bundle.
97
+ */
98
+ export function declaredScreenForRender(preset, presetName, rendered) {
99
+ if (preset.width === rendered[0] && preset.height === rendered[1])
100
+ return undefined;
101
+ return { width: preset.width, height: preset.height, preset: presetName };
102
+ }
103
+ /** ISO timestamp from an injectable clock (tests freeze `now` for deterministic durationMs). */
104
+ function isoNow(now) {
105
+ return new Date(now()).toISOString();
106
+ }
107
+ /** Emit a phase-started event (no ok/durationMs: those belong to the matching completed event). */
108
+ /**
109
+ * Run a provisioning step and, when it fails with an EXIT CODE, run it once more (#602). A cold
110
+ * install of 0.74.0 lost its whole first live study to one transient TLS error inside the
111
+ * sandbox's `npm install`; the parallel install twenty seconds later passed, as had the ten
112
+ * before it. One retry clears that class. A TIMEOUT is not retried: its budget is already spent,
113
+ * and a second wait would double it. The retry runs under its own step name so both logs stay.
114
+ */
115
+ async function runProvisioningStepWithOneRetry(desktop, args) {
116
+ const first = await runDetachedStep(desktop, {
117
+ name: args.name,
118
+ command: args.command,
119
+ cwd: args.cwd,
120
+ timeoutMs: args.timeoutMs,
121
+ requestTimeoutMs: args.requestTimeoutMs,
122
+ ...args.timers
123
+ });
124
+ if (first.ok || first.timedOut)
125
+ return { ...first, attempts: 1 };
126
+ const retryStartedAt = args.now();
127
+ emitPhaseStarted(args.onPhase, args.now, args.retryPhase, `${args.retryMessage} (first attempt exited ${first.exitCode ?? "null"}; retrying once)`);
128
+ const second = await runDetachedStep(desktop, {
129
+ name: `${args.name}-retry`,
130
+ command: args.command,
131
+ cwd: args.cwd,
132
+ timeoutMs: args.timeoutMs,
133
+ requestTimeoutMs: args.requestTimeoutMs,
134
+ ...args.timers
135
+ });
136
+ emitPhaseCompleted(args.onPhase, args.now, retryStartedAt, args.retryPhase, second.ok, second.ok ? `${args.retryMessage}: succeeded on the second attempt` : `${args.retryMessage}: failed twice`);
137
+ return { ...second, attempts: 2, ...(first.exitCode === undefined ? {} : { firstExitCode: first.exitCode }) };
138
+ }
139
+ function emitPhaseStarted(onPhase, now, phase, message) {
140
+ onPhase?.({ at: isoNow(now), type: `cua-lab.subject.${phase}.started`, message });
141
+ }
142
+ /** Emit the matching phase-completed event: always carries ok and durationMs (>= 0). */
143
+ function emitPhaseCompleted(onPhase, now, startedAt, phase, ok, message) {
144
+ onPhase?.({
145
+ at: isoNow(now),
146
+ type: `cua-lab.subject.${phase}.completed`,
147
+ ok,
148
+ durationMs: Math.max(0, now() - startedAt),
149
+ message
150
+ });
151
+ }
152
+ /** Default phase-boundary sink (stderr): one line per event, prefixed with the lane id ONLY
153
+ * when laneCount > 1. Single-lane emission is unconditional: total single-lane silence for the
154
+ * whole clone/install/build/ready boot is the bug this event stream exists to close.
155
+ * Overridable via CuaActorLabHooks.onPhase so deterministic tests capture instead of writing to
156
+ * the real stderr. */
157
+ export function defaultSubjectPhaseSink(event, ctx) {
158
+ const durationSuffix = event.durationMs === undefined ? "" : ` (${event.durationMs}ms)`;
159
+ const prefix = ctx.laneCount > 1 ? `humanish cua [${ctx.laneId}]` : "humanish cua";
160
+ process.stderr.write(`${prefix}: ${event.message}${durationSuffix}\n`);
161
+ }
162
+ /**
163
+ * Verify the desktop screen geometry IN-SANDBOX (the per-lane device claim is checked, never
164
+ * assumed). A parseable mismatch fails closed. Unavailable/unparseable evidence is returned as
165
+ * an explicit warning: the lane may still run, but its bundle records only the requested screen
166
+ * and never upgrades that request into a verified measurement.
167
+ */
168
+ export async function inspectDesktopScreenGeometry(args) {
169
+ let out = "";
170
+ try {
171
+ const result = await args.desktop.commands.run("xdpyinfo 2>/dev/null | grep -i dimensions || true", { requestTimeoutMs: args.requestTimeoutMs });
172
+ out = (result.stdout ?? "").trim();
173
+ }
174
+ catch {
175
+ return { warning: `Desktop screen geometry could not be measured for lane ${args.laneId}; requested geometry remains unverified.` };
176
+ }
177
+ const match = out.match(/(\d+)\s*x\s*(\d+)\s*pixels/i);
178
+ if (!match) {
179
+ return { warning: `Desktop screen geometry could not be parsed for lane ${args.laneId}; requested geometry remains unverified.` };
180
+ }
181
+ const width = Number(match[1]);
182
+ const height = Number(match[2]);
183
+ const [expectedWidth, expectedHeight] = args.requestedScreen;
184
+ if (width === expectedWidth && height === expectedHeight) {
185
+ return { verified: { width, height, source: "xdpyinfo" } };
186
+ }
187
+ return {
188
+ verified: { width, height, source: "xdpyinfo" },
189
+ error: `HUMANISH_CUA_LAB_DEVICE_GEOMETRY: lane ${args.laneId} requested a ${expectedWidth}x${expectedHeight} desktop but xdpyinfo reports ${width}x${height} in-sandbox; the per-lane device geometry is unverified (fail-closed).`
190
+ };
191
+ }
192
+ async function findVisibleBrowserWindowId(desktop, requestTimeoutMs, browserFamily, launchIdentity) {
193
+ if (browserFamily === "unknown")
194
+ return undefined;
195
+ // The candidate loop keeps the LAST identity match: with a launch identity the match is
196
+ // unique anyway, and without one every family candidate matches, so the newest visible
197
+ // window of the launched family wins (the window this lane just opened).
198
+ const finder = browserFamily === "firefox"
199
+ ? [
200
+ "find_firefox_window() {",
201
+ " timeout 2s xdotool search --onlyvisible --class 'firefox|Firefox' 2>/dev/null || true",
202
+ "}",
203
+ "window_id=",
204
+ "for _ in $(seq 1 10); do",
205
+ " for candidate in $(find_firefox_window); do",
206
+ " window_pid=\"$(xdotool getwindowpid \"$candidate\" 2>/dev/null || true)\"",
207
+ " if matches_launch_identity \"$window_pid\"; then window_id=\"$candidate\"; fi",
208
+ " done",
209
+ " if [ -n \"$window_id\" ]; then break; fi",
210
+ " sleep 0.5",
211
+ "done"
212
+ ]
213
+ : [
214
+ "find_chrome_window() {",
215
+ " timeout 2s xdotool search --onlyvisible --class 'google-chrome|Google-chrome|chromium|Chromium|chrome|Chrome' 2>/dev/null || true",
216
+ "}",
217
+ "window_id=",
218
+ "for _ in $(seq 1 10); do",
219
+ " for candidate in $(find_chrome_window); do",
220
+ " window_pid=\"$(xdotool getwindowpid \"$candidate\" 2>/dev/null || true)\"",
221
+ " if matches_launch_identity \"$window_pid\"; then window_id=\"$candidate\"; fi",
222
+ " done",
223
+ " if [ -n \"$window_id\" ]; then break; fi",
224
+ " sleep 0.5",
225
+ "done"
226
+ ];
227
+ const result = await desktop.commands.run([
228
+ "set -euo pipefail",
229
+ "export DISPLAY=\"${DISPLAY:-:0}\"",
230
+ `launch_pid=${shellSingleQuote(launchIdentity?.processId ?? "")}`,
231
+ `profile_dir=${shellSingleQuote(launchIdentity?.profileDir ?? "")}`,
232
+ "matches_launch_identity() {",
233
+ " if [ -z \"$launch_pid\" ] && [ -z \"$profile_dir\" ]; then return 0; fi",
234
+ " local current=\"${1:-}\"",
235
+ " while [[ \"$current\" =~ ^[0-9]+$ ]] && [ \"$current\" -gt 1 ]; do",
236
+ " cmdline=\"$(tr '\\0' ' ' < \"/proc/$current/cmdline\" 2>/dev/null || true)\"",
237
+ " if [ -n \"$profile_dir\" ] && [[ \"$cmdline\" == *\"$profile_dir\"* ]]; then return 0; fi",
238
+ " if [ \"$current\" = \"$launch_pid\" ]; then return 0; fi",
239
+ " current=\"$(ps -o ppid= -p \"$current\" 2>/dev/null | tr -d ' ' || true)\"",
240
+ " done",
241
+ " return 1",
242
+ "}",
243
+ ...finder,
244
+ "if [ -n \"$window_id\" ]; then printf 'WINDOW_ID=%s\\n' \"$window_id\"; fi"
245
+ ].join("\n"), {
246
+ requestTimeoutMs,
247
+ timeoutMs: 15_000
248
+ });
249
+ return (result.stdout ?? "").match(/^WINDOW_ID=(\S+)$/m)?.[1];
250
+ }
251
+ /**
252
+ * Build the xdotool command that makes a browser window fill the desktop.
253
+ * Exported (pure) for contract tests. A window manager can ignore Chrome's
254
+ * --window-size, so xdotool is the robust path: move the window to the origin,
255
+ * then size it to the exact desktop resolution so Observer screenshots carry no
256
+ * dead margin around the browser.
257
+ */
258
+ export function buildFillDesktopWindowCommand(windowId, width, height) {
259
+ return [
260
+ "set -euo pipefail",
261
+ `win=${shellSingleQuote(windowId)}`,
262
+ `xdotool windowactivate "$win" >/dev/null 2>&1 || true`,
263
+ `xdotool windowmove "$win" 0 0 >/dev/null 2>&1 || true`,
264
+ `xdotool windowsize "$win" ${width} ${height} >/dev/null 2>&1 || true`,
265
+ ].join("\n");
266
+ }
267
+ /**
268
+ * Best-effort initial fill. A contained smaller window remains usable; the capture
269
+ * below checks for clipping and refuses an uncorrectable window before the actor runs.
270
+ */
271
+ async function fillDesktopBrowserWindow(desktop, windowId, resolution, requestTimeoutMs) {
272
+ const [width, height] = resolution;
273
+ await desktop.commands
274
+ .run(buildFillDesktopWindowCommand(windowId, width, height), {
275
+ requestTimeoutMs,
276
+ timeoutMs: 10_000,
277
+ })
278
+ .catch(() => undefined);
279
+ }
280
+ export async function openDesktopBrowserTarget(desktop, targetUrl, requestTimeoutMs, browserPreference,
281
+ /** Launch-time flags that make mobile fidelity (#221) hold across every tab: the user agent and
282
+ * touch events are browser-wide here, where the CDP holder covers only the launch page. */
283
+ extraChromiumFlags = []) {
284
+ const requestedBrowser = browserPreference ?? "default";
285
+ if (isHttpUrl(targetUrl)) {
286
+ const chromiumFlags = [...CHROMIUM_EVIDENCE_HYGIENE_FLAGS, ...extraChromiumFlags].map(shellSingleQuote).join(" ");
287
+ const browserLaunchCommand = [
288
+ "set -euo pipefail",
289
+ `target_url=${shellSingleQuote(targetUrl)}`,
290
+ `browser_preference=${shellSingleQuote(requestedBrowser)}`,
291
+ "chrome_profile_dir=",
292
+ `chrome_preferences_json=${shellSingleQuote(chromiumEvidenceProfilePreferencesJson())}`,
293
+ "prepare_chrome_profile() {",
294
+ " chrome_profile_dir=\"$(mktemp -d /tmp/humanish-chrome-profile.XXXXXX)\"",
295
+ " mkdir -p \"$chrome_profile_dir/Default\"",
296
+ " printf '%s\\n' \"$chrome_preferences_json\" > \"$chrome_profile_dir/Default/Preferences\"",
297
+ "}",
298
+ "launch_browser() {",
299
+ " local label=\"$1\"",
300
+ " local binary=\"$2\"",
301
+ " shift 2",
302
+ " if command -v \"$binary\" >/dev/null 2>&1; then",
303
+ " nohup \"$binary\" \"$@\" \"$target_url\" >/tmp/humanish-browser-open.log 2>&1 &",
304
+ " local launch_pid=$!",
305
+ " printf 'HUMANISH_BROWSER_RESOLVED=%s\\n' \"$label\"",
306
+ " printf 'HUMANISH_BROWSER_PID=%s\\n' \"$launch_pid\"",
307
+ " printf 'HUMANISH_BROWSER_PROFILE_DIR=%s\\n' \"$chrome_profile_dir\"",
308
+ " if [[ \"$label\" =~ ^(google-chrome|google-chrome-stable|chromium|chromium-browser)$ ]]; then",
309
+ " for _ in $(seq 1 30); do",
310
+ " if [ -s \"$chrome_profile_dir/DevToolsActivePort\" ]; then",
311
+ " head -n 1 \"$chrome_profile_dir/DevToolsActivePort\" | sed 's/^/HUMANISH_BROWSER_CDP_PORT=/'",
312
+ " break",
313
+ " fi",
314
+ " sleep 0.1",
315
+ " done",
316
+ " fi",
317
+ " return 0",
318
+ " fi",
319
+ " return 1",
320
+ "}",
321
+ // Fixed CDP port (not :0/random): each seat has its OWN desktop sandbox, so a known port
322
+ // cannot conflict, and it makes the observer's port resolution deterministic. With :0 the
323
+ // real port lives only in DevToolsActivePort; when the launch-time capture misses on a cold
324
+ // start the observer falls back to 9222 and — being wrong — every CDP read fails for the
325
+ // whole run (the lobby-code handoff then never sees the host's /lobby URL). 9222 is already
326
+ // the fallback, so making it the actual port aligns launch, capture, and fallback.
327
+ `chrome_debug_flags=(--remote-debugging-address=127.0.0.1 --remote-debugging-port=9222 ${chromiumFlags})`,
328
+ "open_target() {",
329
+ " case \"$browser_preference\" in",
330
+ " chrome)",
331
+ " prepare_chrome_profile",
332
+ " launch_browser google-chrome google-chrome --new-window \"--user-data-dir=$chrome_profile_dir\" \"${chrome_debug_flags[@]}\" && return 0",
333
+ " launch_browser google-chrome-stable google-chrome-stable --new-window \"--user-data-dir=$chrome_profile_dir\" \"${chrome_debug_flags[@]}\" && return 0",
334
+ " echo 'requested browser chrome was not found' >&2",
335
+ " return 127",
336
+ " ;;",
337
+ " chromium)",
338
+ " prepare_chrome_profile",
339
+ " launch_browser chromium chromium --new-window \"--user-data-dir=$chrome_profile_dir\" \"${chrome_debug_flags[@]}\" && return 0",
340
+ " launch_browser chromium-browser chromium-browser --new-window \"--user-data-dir=$chrome_profile_dir\" \"${chrome_debug_flags[@]}\" && return 0",
341
+ " echo 'requested browser chromium was not found' >&2",
342
+ " return 127",
343
+ " ;;",
344
+ " firefox)",
345
+ " prepare_chrome_profile",
346
+ " launch_browser firefox firefox --new-instance --no-remote --new-window --profile \"$chrome_profile_dir\" && return 0",
347
+ " echo 'requested browser firefox was not found' >&2",
348
+ " return 127",
349
+ " ;;",
350
+ " default)",
351
+ " prepare_chrome_profile",
352
+ " launch_browser google-chrome google-chrome --new-window \"--user-data-dir=$chrome_profile_dir\" \"${chrome_debug_flags[@]}\" && return 0",
353
+ " launch_browser google-chrome-stable google-chrome-stable --new-window \"--user-data-dir=$chrome_profile_dir\" \"${chrome_debug_flags[@]}\" && return 0",
354
+ " launch_browser chromium chromium --new-window \"--user-data-dir=$chrome_profile_dir\" \"${chrome_debug_flags[@]}\" && return 0",
355
+ " launch_browser chromium-browser chromium-browser --new-window \"--user-data-dir=$chrome_profile_dir\" \"${chrome_debug_flags[@]}\" && return 0",
356
+ " launch_browser firefox firefox --new-instance --no-remote --new-window --profile \"$chrome_profile_dir\" && return 0",
357
+ " launch_browser xdg-open xdg-open && return 0",
358
+ " echo 'no browser opener found' >&2",
359
+ " return 127",
360
+ " ;;",
361
+ " esac",
362
+ "}",
363
+ "open_target"
364
+ ].join("\n");
365
+ const result = await runDesktopCommandOrThrow(() => desktop.commands.run(browserLaunchCommand, {
366
+ requestTimeoutMs,
367
+ timeoutMs: 15_000,
368
+ }), ({ exitCode, stderrTail }) => new Error(`browser launch failed${exitCode === undefined ? "" : ` with exit ${exitCode}`}: ${stderrTail}`));
369
+ if (result.exitCode !== undefined && result.exitCode !== 0) {
370
+ throw new Error(`browser launch failed with exit ${result.exitCode}: ${tailOf(result.stderr ?? result.stdout ?? "")}`);
371
+ }
372
+ const resolved = (result.stdout ?? "").match(/^HUMANISH_BROWSER_RESOLVED=(\S+)$/m)?.[1];
373
+ const processId = (result.stdout ?? "").match(/^HUMANISH_BROWSER_PID=(\d+)$/m)?.[1];
374
+ const profileDir = (result.stdout ?? "").match(/^HUMANISH_BROWSER_PROFILE_DIR=(\S+)$/m)?.[1];
375
+ const cdpPortRaw = (result.stdout ?? "").match(/^HUMANISH_BROWSER_CDP_PORT=(\d+)$/m)?.[1];
376
+ const cdpPort = cdpPortRaw === undefined ? undefined : Number(cdpPortRaw);
377
+ return {
378
+ family: desktopBrowserFamily(resolved ?? requestedBrowser),
379
+ ...(processId === undefined || profileDir === undefined
380
+ ? {}
381
+ : { identity: { processId, profileDir, targetUrl, ...(cdpPort === undefined ? {} : { cdpPort }) } }),
382
+ ...(browserPreference === undefined
383
+ ? {}
384
+ : { evidence: { requested: requestedBrowser, ...(resolved === undefined ? {} : { resolved }) } })
385
+ };
386
+ }
387
+ if (browserPreference === undefined || browserPreference === "default") {
388
+ if (desktop.open) {
389
+ await desktop.open(targetUrl);
390
+ }
391
+ else {
392
+ await desktop.launch("google-chrome", targetUrl);
393
+ }
394
+ return {
395
+ family: desktop.open ? "unknown" : "chromium",
396
+ ...(browserPreference === undefined ? {} : { evidence: { requested: requestedBrowser } })
397
+ };
398
+ }
399
+ const launchTarget = requestedBrowser === "chrome" ? "google-chrome"
400
+ : requestedBrowser === "chromium" ? "chromium"
401
+ : requestedBrowser === "firefox" ? "firefox"
402
+ : "google-chrome";
403
+ await desktop.launch(launchTarget, targetUrl);
404
+ return {
405
+ family: desktopBrowserFamily(launchTarget),
406
+ evidence: { requested: requestedBrowser, resolved: launchTarget }
407
+ };
408
+ }
409
+ export function desktopBrowserFamily(value) {
410
+ if (value === "firefox")
411
+ return "firefox";
412
+ if (value === "chrome" || value === "chromium" || value === "google-chrome" || value === "google-chrome-stable" || value === "chromium-browser") {
413
+ return "chromium";
414
+ }
415
+ return "unknown";
416
+ }
417
+ /**
418
+ * The URL / title / page-text / scroll observer behind stopWhen and task criteria. One probe per
419
+ * observation, run on the sandbox's python3 (see chrome-cdp-probe.ts for why not node: #514).
420
+ *
421
+ * "active": follow the participant to whatever tab they are driving now — never pin the state
422
+ * observer to the launch tab (a verification link that opened in a NEW tab left a pinned observer
423
+ * reading the old tab forever).
424
+ *
425
+ * `onUnavailable` fires ONCE, on the first probe that could not read the page, with the reason.
426
+ * The observer still degrades to `{}` for the loop; the callback is how a lane says out loud that
427
+ * url/text criteria are not being measured, instead of letting the funnel report 0/N (#514).
428
+ */
429
+ export function makeChromeBrowserStateObserver(desktop, requestTimeoutMs, endpoint, targetId, onUnavailable,
430
+ /**
431
+ * Mobile emulation on later tabs (#623): the holder attaches to every page target Chrome opens
432
+ * after the launch page, so a tab the participant opens later should lay out at the phone width
433
+ * too. The first observation on each new target reads that page's OWN report; a target that
434
+ * reports the requested width is recorded through `onCovered`, and one that does not (or cannot
435
+ * be read) fires `onDrift` once, so a phone-labelled lane that spent part of its session at
436
+ * desktop layout says so with the number the page gave.
437
+ */
438
+ drift) {
439
+ let reported = false;
440
+ let drifted = false;
441
+ const checkedTargets = new Set(drift === undefined ? [] : [drift.emulatedTargetId]);
442
+ const unavailable = (reason) => {
443
+ if (!reported) {
444
+ reported = true;
445
+ onUnavailable?.(reason);
446
+ }
447
+ return {};
448
+ };
449
+ const checkLaterTarget = async (newTargetId) => {
450
+ if (drift === undefined || checkedTargets.has(newTargetId))
451
+ return;
452
+ checkedTargets.add(newTargetId);
453
+ const read = await desktop.commands.run(chromeCdpProbeCommand({ ...endpoint, targetId: newTargetId, prefer: "pinned", mode: "fidelity" }), { requestTimeoutMs, timeoutMs: 5_000 });
454
+ const fidelity = read.exitCode !== undefined && read.exitCode !== 0 ? undefined : parseChromeCdpProbeOutput(read.stdout).fidelity;
455
+ if (fidelity !== undefined && fidelity.innerWidth === drift.expectedWidth) {
456
+ drift.onCovered?.(newTargetId, { innerWidth: fidelity.innerWidth, devicePixelRatio: fidelity.devicePixelRatio, maxTouchPoints: fidelity.maxTouchPoints });
457
+ if (drift.expectTouch === true && fidelity.maxTouchPoints === 0 && !drifted) {
458
+ // The viewport followed; touch did not (yet): the holder reloads a later tab once after its
459
+ // first navigation commits, and this observation may have landed before that reload.
460
+ drifted = true;
461
+ drift.onDrift(`a later page target reports the ${fidelity.innerWidth} px viewport but navigator.maxTouchPoints 0 on its first observation; touch reaches a document only when it loads under the override`);
462
+ }
463
+ return;
464
+ }
465
+ if (drifted)
466
+ return;
467
+ drifted = true;
468
+ drift.onDrift(fidelity === undefined
469
+ ? "the participant drove a page target other than the emulated launch tab and that page's own read-back could not be taken; whether it laid out at the phone width is not known"
470
+ : `the participant drove a page target other than the emulated launch tab and that page reports a ${fidelity.innerWidth} px viewport where ${drift.expectedWidth} px was requested (DPR ${fidelity.devicePixelRatio}); the mobile user agent and touch events are browser-wide, the viewport override was not re-applied to it`);
471
+ };
472
+ return async () => {
473
+ const result = await desktop.commands.run(chromeCdpProbeCommand({ ...endpoint, ...(targetId === undefined ? {} : { targetId }), prefer: "active", mode: "state" }), { requestTimeoutMs, timeoutMs: 5_000 });
474
+ if (result.exitCode !== undefined && result.exitCode !== 0) {
475
+ return unavailable(`probe exited ${result.exitCode}: ${tailOf(result.stderr ?? result.stdout ?? "")}`);
476
+ }
477
+ const parsed = parseChromeCdpProbeOutput(result.stdout);
478
+ if (parsed.unavailable !== undefined)
479
+ return unavailable(parsed.unavailable);
480
+ if (parsed.targetId !== undefined)
481
+ await checkLaterTarget(parsed.targetId);
482
+ return {
483
+ ...(parsed.url === undefined ? {} : { url: parsed.url }),
484
+ ...(parsed.title === undefined ? {} : { title: parsed.title }),
485
+ ...(parsed.text === undefined ? {} : { text: parsed.text }),
486
+ ...(parsed.scrollY === undefined ? {} : { scrollY: parsed.scrollY })
487
+ };
488
+ };
489
+ }
490
+ /**
491
+ * Read the running browser's actual outer-window bounds and CSS layout viewport through the
492
+ * already-enabled local Chrome DevTools endpoint. The returned values come from `window.*` in
493
+ * the target page; requested E2B resolution is deliberately not an input to this function.
494
+ * Missing channels report their reason via `onUnavailable`, so the geometry warning can name
495
+ * the cause (a dead CDP endpoint, no python3) instead of only the symptom. Returns `undefined`
496
+ * only when neither channel could be measured.
497
+ * Outer bounds and CSS dimensions are independent channels: a background page can report zero
498
+ * outer dimensions while still reporting a CSS viewport. Final captures follow the active tab;
499
+ * launch captures and emulation attribution keep the pinned target.
500
+ */
501
+ export function makeChromeDesktopGeometryObserver(desktop, requestTimeoutMs, endpoint, targetId, onUnavailable, prefer = "pinned") {
502
+ return async () => {
503
+ const result = await desktop.commands.run(chromeCdpProbeCommand({ ...endpoint, ...(targetId === undefined ? {} : { targetId }), prefer, mode: "geometry" }), { requestTimeoutMs, timeoutMs: 5_000 });
504
+ if (result.exitCode !== undefined && result.exitCode !== 0) {
505
+ onUnavailable?.(`probe exited ${result.exitCode}: ${tailOf(result.stderr ?? result.stdout ?? "")}`);
506
+ return undefined;
507
+ }
508
+ const parsed = parseChromeCdpProbeOutput(result.stdout);
509
+ if (parsed.unavailable !== undefined) {
510
+ onUnavailable?.(parsed.unavailable);
511
+ return undefined;
512
+ }
513
+ const browserWindow = isMeasuredRect(parsed.browserWindow) ? { ...parsed.browserWindow, source: "cdp" } : undefined;
514
+ const viewport = isMeasuredViewport(parsed.viewport) ? { ...parsed.viewport, source: "cdp" } : undefined;
515
+ if (browserWindow === undefined && viewport === undefined) {
516
+ onUnavailable?.("the page reported no usable window or viewport dimensions");
517
+ return undefined;
518
+ }
519
+ if (browserWindow === undefined)
520
+ onUnavailable?.("the page reported no usable outer-window dimensions");
521
+ if (viewport === undefined)
522
+ onUnavailable?.("the page reported no usable CSS viewport dimensions");
523
+ return {
524
+ ...(browserWindow === undefined ? {} : { browserWindow }),
525
+ ...(viewport === undefined ? {} : { viewport }),
526
+ ...(parsed.targetId === undefined ? {} : { targetId: parsed.targetId })
527
+ };
528
+ };
529
+ }
530
+ /** The user agent a mobile-emulated lane presents unless the lab sets its own. */
531
+ export const DEFAULT_MOBILE_USER_AGENT = "Mozilla/5.0 (iPhone; CPU iPhone OS 17_0 like Mac OS X) AppleWebKit/605.1.15 (KHTML, like Gecko) Version/17.0 Mobile/15E148 Safari/604.1";
532
+ /**
533
+ * Apply mobile emulation (#221) to the lane's launch page and read back what the page reports.
534
+ * Fails CLOSED: a request that cannot be applied throws, because a desktop run labelled mobile is
535
+ * the over-trust this feature exists to prevent. A read-back that cannot be taken is a warning
536
+ * (the emulation was applied; only the proof is missing).
537
+ */
538
+ export async function applyMobileEmulation(desktop, requestTimeoutMs, endpoint, targetId, request) {
539
+ const command = (mode) => chromeCdpProbeCommand({ ...endpoint, ...(targetId === undefined ? {} : { targetId }), prefer: "pinned", mode, emulation: request });
540
+ const read = async () => {
541
+ const result = await desktop.commands.run(command("fidelity"), { requestTimeoutMs, timeoutMs: 15_000 });
542
+ if (result.exitCode !== undefined && result.exitCode !== 0) {
543
+ return { unavailable: `probe exited ${result.exitCode}: ${tailOf(result.stderr ?? result.stdout ?? "")}` };
544
+ }
545
+ return parseChromeCdpProbeOutput(result.stdout);
546
+ };
547
+ // The UA / touch / DPR overrides are bound to the DevTools session that set them and lapse the
548
+ // moment its socket closes (measured: only the viewport width survived a one-shot apply). So the
549
+ // applier stays attached for the lane's whole life as a detached process; the sandbox teardown
550
+ // ends it. Its first stdout line says what was applied.
551
+ const holderName = `mobile-emulation-${Date.now().toString(36)}`;
552
+ await startDetachedProcess(desktop, { name: holderName, command: command("hold"), requestTimeoutMs });
553
+ let announced;
554
+ for (let attempt = 0; attempt < 30 && announced === undefined; attempt += 1) {
555
+ await new Promise((resolve) => setTimeout(resolve, 500));
556
+ const log = await readDetachedLog(desktop, holderName, requestTimeoutMs).catch(() => "");
557
+ const line = log.split("\n").find((candidate) => candidate.trim().startsWith("{"));
558
+ if (line !== undefined)
559
+ announced = parseChromeCdpProbeOutput(line);
560
+ }
561
+ if (announced === undefined) {
562
+ throw new Error("mobile emulation could not be applied: the in-sandbox applier printed nothing within 15 s");
563
+ }
564
+ if (announced.unavailable !== undefined) {
565
+ throw new Error(`mobile emulation could not be applied (${announced.unavailable}); applied before failing: ${(announced.applied ?? []).join(", ") || "nothing"}`);
566
+ }
567
+ const applied = announced;
568
+ // Viewport/touch read-back proves context settings, not gesture equivalence. Two hosted
569
+ // replicas and a native-X conversion-toggle control reproduced reset click counts (#676).
570
+ const warnings = request.touch
571
+ ? ["Mobile emulation uses desktop pointer-to-touch conversion, which can differ for repeated taps. Confirm gesture failures with direct or native touch input before attributing them to the app."]
572
+ : [];
573
+ // The reload inside the applier takes a moment; the read-back is retried until the page reports
574
+ // the requested viewport and user agent, so a slow page does not read as "no proof".
575
+ let readBack = await read();
576
+ for (let attempt = 0; attempt < 20 && (readBack.fidelity === undefined || readBack.fidelity.innerWidth !== request.width || !readBack.fidelity.userAgent.includes(request.userAgent.slice(0, 24))); attempt += 1) {
577
+ await new Promise((resolve) => setTimeout(resolve, 500));
578
+ readBack = await read();
579
+ }
580
+ const fidelityRead = readBack;
581
+ const requested = {
582
+ width: request.width,
583
+ height: request.height,
584
+ deviceScaleFactor: request.deviceScaleFactor,
585
+ touch: request.touch,
586
+ userAgent: request.userAgent
587
+ };
588
+ const emulatedTargetId = applied.targetId ?? fidelityRead.targetId;
589
+ if (fidelityRead.fidelity === undefined) {
590
+ warnings.push(`Mobile emulation was applied but the page's own report could not be read (${fidelityRead.unavailable ?? "no fidelity read"}); desktopGeometry.fidelity carries the request without a resolved block.`);
591
+ return { fidelity: { tier: "mobile-emulated", requested, applied: applied.applied ?? [] }, warnings, holderName, ...(emulatedTargetId === undefined ? {} : { targetId: emulatedTargetId }) };
592
+ }
593
+ const resolved = { ...fidelityRead.fidelity, source: "cdp" };
594
+ if (resolved.innerWidth !== request.width) {
595
+ warnings.push(`Mobile emulation requested a ${request.width} px viewport; the page reports ${resolved.innerWidth} px.`);
596
+ }
597
+ if (resolved.devicePixelRatio !== request.deviceScaleFactor) {
598
+ warnings.push(`Mobile emulation requested devicePixelRatio ${request.deviceScaleFactor}; the page reports ${resolved.devicePixelRatio}.`);
599
+ }
600
+ if (request.touch && resolved.maxTouchPoints === 0) {
601
+ warnings.push("Mobile emulation requested touch; the page reports navigator.maxTouchPoints 0.");
602
+ }
603
+ if (!resolved.userAgent.includes("Mobile") && !resolved.userAgent.includes("Android") && !resolved.userAgent.includes("iPhone")) {
604
+ warnings.push("Mobile emulation requested a mobile user agent; the page reports a desktop one.");
605
+ }
606
+ return { fidelity: { tier: "mobile-emulated", requested, applied: applied.applied ?? [], resolved }, warnings, holderName, ...(emulatedTargetId === undefined ? {} : { targetId: emulatedTargetId }) };
607
+ }
608
+ function isMeasuredRect(value) {
609
+ if (!value || typeof value !== "object")
610
+ return false;
611
+ const record = value;
612
+ return Number.isFinite(record.x)
613
+ && Number.isFinite(record.y)
614
+ && isPositiveMeasurement(record.width)
615
+ && isPositiveMeasurement(record.height);
616
+ }
617
+ function isMeasuredViewport(value) {
618
+ if (!value || typeof value !== "object")
619
+ return false;
620
+ const record = value;
621
+ return isPositiveMeasurement(record.width)
622
+ && isPositiveMeasurement(record.height)
623
+ && isPositiveMeasurement(record.deviceScaleFactor);
624
+ }
625
+ function isPositiveMeasurement(value) {
626
+ return typeof value === "number" && Number.isFinite(value) && value > 0;
627
+ }
628
+ async function measureBrowserWindowWithXwininfo(desktop, windowId, requestTimeoutMs) {
629
+ const result = await desktop.commands.run([
630
+ "set -euo pipefail",
631
+ `win=${shellSingleQuote(windowId)}`,
632
+ // Older xdotool builds translate parent-relative offsets twice. With window
633
+ // decorations that falsely reports a visible client as clipped, triggering
634
+ // fullscreen and hiding the participant's address bar. Read root-relative
635
+ // client coordinates directly; never substitute emulated CDP outer bounds.
636
+ "LC_ALL=C xwininfo -id \"$win\" -stats 2>/dev/null"
637
+ ].join("\n"), { requestTimeoutMs, timeoutMs: 5_000 });
638
+ if (result.exitCode !== undefined && result.exitCode !== 0)
639
+ return undefined;
640
+ return parseXwininfoGeometry(result.stdout ?? "");
641
+ }
642
+ /** Root-relative physical client bounds from xwininfo's C-locale stats. */
643
+ export function parseXwininfoGeometry(output) {
644
+ const read = (label) => {
645
+ const matches = [...output.matchAll(new RegExp(`^\\s*${label}:\\s*(-?\\d+)\\s*$`, "gm"))];
646
+ if (matches.length !== 1)
647
+ return undefined;
648
+ const value = Number(matches[0][1]);
649
+ return Number.isSafeInteger(value) ? value : undefined;
650
+ };
651
+ const x = read("Absolute upper-left X");
652
+ const y = read("Absolute upper-left Y");
653
+ const width = read("Width");
654
+ const height = read("Height");
655
+ const mapStates = [...output.matchAll(/^\s*Map State:\s*(\S+)\s*$/gm)];
656
+ if (mapStates.length !== 1 || mapStates[0][1] !== "IsViewable")
657
+ return undefined;
658
+ if (x === undefined || y === undefined || width === undefined || height === undefined || width <= 0 || height <= 0) {
659
+ return undefined;
660
+ }
661
+ return { x, y, width, height, source: "xwininfo" };
662
+ }
663
+ /** Physical X client bounds, never the page's emulated window.outerWidth/Height. */
664
+ function isBrowserWindowContained(bounds, [width, height]) {
665
+ return bounds.x >= 0 && bounds.y >= 0
666
+ && bounds.x + bounds.width <= width && bounds.y + bounds.height <= height;
667
+ }
668
+ /** Bounded repair. Resizing can clear a window-manager maximize state and move the
669
+ * client origin as decorations return, so remeasure before a second adjustment. */
670
+ async function fitBrowserWindowWithinDesktop(desktop, windowId, resolution, requestTimeoutMs) {
671
+ const run = (command) => desktop.commands.run([
672
+ "set -euo pipefail",
673
+ `win=${shellSingleQuote(windowId)}`,
674
+ command
675
+ ].join("\n"), { requestTimeoutMs, timeoutMs: 5_000 }).catch(() => undefined);
676
+ await run('xdotool windowmove "$win" 0 0');
677
+ await desktop.wait(250).catch(() => undefined);
678
+ const moved = await measureBrowserWindowWithXwininfo(desktop, windowId, requestTimeoutMs).catch(() => undefined);
679
+ if (moved === undefined)
680
+ return moved;
681
+ let resized = moved;
682
+ // Resizing alone cannot fix an offscreen client origin. The window manager
683
+ // can also center a minimum-width client at a negative x on a narrow screen.
684
+ for (let attempt = 0; attempt < 2; attempt += 1) {
685
+ if (isBrowserWindowContained(resized, resolution))
686
+ return resized;
687
+ const width = resolution[0] - resized.x;
688
+ const height = resolution[1] - resized.y;
689
+ if (resized.x < 0 || resized.y < 0 || width <= 0 || height <= 0)
690
+ break;
691
+ await run(`xdotool windowsize "$win" ${width} ${height}`);
692
+ await desktop.wait(250).catch(() => undefined);
693
+ const measured = await measureBrowserWindowWithXwininfo(desktop, windowId, requestTimeoutMs).catch(() => undefined);
694
+ if (measured === undefined)
695
+ return measured;
696
+ resized = measured;
697
+ }
698
+ if (resized === undefined || isBrowserWindowContained(resized, resolution))
699
+ return resized;
700
+ // Chrome's minimum client width can equal the whole desktop. Window-manager
701
+ // borders then make a decorated window impossible to contain, even after a
702
+ // successful move/resize. Request fullscreen once and prove the physical result.
703
+ // xprop/xdotool ship with the desktop template; wmctrl is not required.
704
+ // Check state first so the fullscreen shortcut cannot toggle an existing state off.
705
+ await run([
706
+ 'state=$(xprop -id "$win" _NET_WM_STATE)',
707
+ 'case "$state" in',
708
+ ' *_NET_WM_STATE_FULLSCREEN*) ;;',
709
+ ' *) xdotool windowactivate --sync "$win"; xdotool key --clearmodifiers F11 ;;',
710
+ 'esac'
711
+ ].join("\n"));
712
+ // The fullscreen animation may report its new origin before its final width.
713
+ // Give the window manager a bounded settling window, keeping missing reads unverified.
714
+ for (let attempt = 0; attempt < 4; attempt += 1) {
715
+ await desktop.wait(250).catch(() => undefined);
716
+ const measured = await measureBrowserWindowWithXwininfo(desktop, windowId, requestTimeoutMs).catch(() => undefined);
717
+ if (measured === undefined || isBrowserWindowContained(measured, resolution))
718
+ return measured;
719
+ resized = measured;
720
+ }
721
+ return resized;
722
+ }
723
+ /** Shared hosted-browser geometry capture used by per-lane and sequential shared-world routes. */
724
+ export async function captureDesktopBrowserGeometry(args) {
725
+ const warnings = [];
726
+ let browserWindowId = args.browserWindowId;
727
+ if (browserWindowId === undefined && args.browserFamily !== "unknown") {
728
+ browserWindowId = await findVisibleBrowserWindowId(args.desktop, args.requestTimeoutMs, args.browserFamily, args.launchIdentity).catch((error) => {
729
+ warnings.push(`Browser window lookup failed for lane ${args.laneId}: ${redactText(toErrorMessage(error))}`);
730
+ return undefined;
731
+ });
732
+ }
733
+ let physicalWindow;
734
+ if (browserWindowId !== undefined) {
735
+ if (args.resize !== false) {
736
+ await fillDesktopBrowserWindow(args.desktop, browserWindowId, args.requestedScreen, args.requestTimeoutMs);
737
+ // Let the window manager apply the resize before querying both X and page layout geometry.
738
+ await args.desktop.wait(250).catch(() => undefined);
739
+ }
740
+ physicalWindow = await measureBrowserWindowWithXwininfo(args.desktop, browserWindowId, args.requestTimeoutMs)
741
+ .catch(() => undefined);
742
+ }
743
+ else {
744
+ warnings.push(`Browser window bounds could not be measured for lane ${args.laneId}; the live stream will use the full desktop.`);
745
+ }
746
+ let unusable;
747
+ if (physicalWindow !== undefined && !isBrowserWindowContained(physicalWindow, args.requestedScreen)) {
748
+ const before = physicalWindow;
749
+ if (args.resize !== false && browserWindowId !== undefined) {
750
+ physicalWindow = await fitBrowserWindowWithinDesktop(args.desktop, browserWindowId, args.requestedScreen, args.requestTimeoutMs);
751
+ if (physicalWindow === undefined) {
752
+ // Keep the last measured bad state; a missing observation cannot prove a successful fix.
753
+ physicalWindow = before;
754
+ unusable = `Physical browser containment could not be verified after correction for lane ${args.laneId}; the last measured window was clipped.`;
755
+ }
756
+ else if (isBrowserWindowContained(physicalWindow, args.requestedScreen)) {
757
+ warnings.push(`Browser window clipping corrected for lane ${args.laneId}; physical bounds are ${physicalWindow.width}x${physicalWindow.height} at (${physicalWindow.x}, ${physicalWindow.y}).`);
758
+ }
759
+ }
760
+ if (unusable === undefined && !isBrowserWindowContained(physicalWindow, args.requestedScreen)) {
761
+ unusable = `Browser window is outside the captured ${args.requestedScreen[0]}x${args.requestedScreen[1]} desktop for lane ${args.laneId}: physical bounds ${physicalWindow.width}x${physicalWindow.height} at (${physicalWindow.x}, ${physicalWindow.y}), right=${physicalWindow.x + physicalWindow.width}, bottom=${physicalWindow.y + physicalWindow.height}.`;
762
+ }
763
+ if (unusable !== undefined)
764
+ warnings.push(unusable);
765
+ }
766
+ if (physicalWindow === undefined) {
767
+ warnings.push(`Physical browser containment is unverified for lane ${args.laneId}; X window bounds could not be measured. Page-reported outer dimensions can be emulated and do not prove physical visibility.`);
768
+ }
769
+ let cdpUnavailable;
770
+ const chromeGeometry = args.browserFamily === "chromium"
771
+ ? await makeChromeDesktopGeometryObserver(args.desktop, args.requestTimeoutMs, {
772
+ ...(args.launchIdentity?.cdpPort === undefined ? {} : { cdpPort: args.launchIdentity.cdpPort }),
773
+ ...(args.launchIdentity?.profileDir === undefined ? {} : { profileDir: args.launchIdentity.profileDir }),
774
+ targetUrl: args.targetUrl
775
+ }, args.browserTargetId, (reason) => {
776
+ cdpUnavailable = reason;
777
+ }, args.pagePreference ?? "pinned")().catch((error) => {
778
+ cdpUnavailable = toErrorMessage(error);
779
+ return undefined;
780
+ })
781
+ : undefined;
782
+ const browserWindow = physicalWindow ?? chromeGeometry?.browserWindow;
783
+ const viewport = chromeGeometry?.viewport;
784
+ // The fill check reads the X window when it was measured: under mobile emulation (#221) the
785
+ // page's window.outerWidth reports the EMULATED screen (414), which is not a fill failure.
786
+ const fillBounds = physicalWindow;
787
+ if (!browserWindow) {
788
+ warnings.push(`Browser outer bounds could not be measured for lane ${args.laneId}.`);
789
+ }
790
+ else if (unusable === undefined && fillBounds !== undefined && (fillBounds.x !== 0 || fillBounds.y !== 0 || fillBounds.width !== args.requestedScreen[0] || fillBounds.height !== args.requestedScreen[1])) {
791
+ warnings.push(`Browser window fill did not reach the requested ${args.requestedScreen[0]}x${args.requestedScreen[1]} screen for lane ${args.laneId}; measured physical bounds are ${fillBounds.width}x${fillBounds.height} at (${fillBounds.x}, ${fillBounds.y}).`);
792
+ }
793
+ if (!viewport) {
794
+ // Name the cause, not only the symptom: the same dead DevTools channel that loses the viewport
795
+ // loses every url/text observation, and a reader of the bundle should learn that here (#514).
796
+ const cause = cdpUnavailable === undefined ? "" : ` DevTools probe: ${redactText(cdpUnavailable)}.`;
797
+ warnings.push(args.browserFamily === "firefox"
798
+ ? `Browser CSS viewport measurement is unavailable for Firefox on lane ${args.laneId}; stream.viewport is omitted instead of reading a different browser's CDP endpoint.`
799
+ : `Browser CSS viewport could not be measured for lane ${args.laneId}; stream.viewport is omitted instead of copying the requested screen resolution.${cause}`);
800
+ }
801
+ return {
802
+ ...(unusable === undefined ? {} : { unusable }),
803
+ ...(browserWindowId === undefined ? {} : { browserWindowId }),
804
+ ...(chromeGeometry?.targetId === undefined ? {} : { browserTargetId: chromeGeometry.targetId }),
805
+ ...(browserWindow === undefined ? {} : { browserWindow }),
806
+ ...(viewport === undefined ? {} : { viewport }),
807
+ warnings
808
+ };
809
+ }
810
+ function shellSingleQuote(value) {
811
+ return `'${value.replace(/'/g, "'\\''")}'`;
812
+ }
813
+ /**
814
+ * Prepare a CLI study's runtime and, only when declared, its product (#495, #515).
815
+ *
816
+ * The install runs UNKEYED and before the session starts, for the same reason the clone route
817
+ * provisions its subject first: what is being studied begins when the participant looks at the
818
+ * screen. Omitting install deliberately studies product installation; Node/npm remain a
819
+ * harness prerequisite so the participant can follow the product's public npm instructions.
820
+ */
821
+ export async function provisionDesktopCli(desktop, args) {
822
+ const install = args.install;
823
+ const now = () => Date.now();
824
+ if (install === undefined || needsNodeRuntime([install])) {
825
+ const startedAt = now();
826
+ emitPhaseStarted(args.onPhase, now, "runtime", "providing Node/npm for the desktop CLI study");
827
+ const bootstrap = await runDetachedStep(desktop, {
828
+ name: "desktop-cli-runtime-node",
829
+ command: TERMINAL_NODE_BOOTSTRAP_COMMAND,
830
+ cwd: "/home/user",
831
+ timeoutMs: INSTALL_TIMEOUT_MS,
832
+ requestTimeoutMs: args.requestTimeoutMs
833
+ });
834
+ emitPhaseCompleted(args.onPhase, now, startedAt, "runtime", bootstrap.ok, bootstrap.ok
835
+ ? "Node runtime ready"
836
+ : "Node runtime bootstrap failed");
837
+ if (!bootstrap.ok) {
838
+ throw new Error(`desktop-cli runtime bootstrap failed for "${args.product}"`);
839
+ }
840
+ }
841
+ if (install === undefined)
842
+ return;
843
+ const startedAt = now();
844
+ emitPhaseStarted(args.onPhase, now, "install", `installing ${args.product} on the desktop`);
845
+ const result = await runDetachedStep(desktop, {
846
+ name: "desktop-cli-install",
847
+ command: install,
848
+ cwd: "/home/user",
849
+ timeoutMs: INSTALL_TIMEOUT_MS,
850
+ requestTimeoutMs: args.requestTimeoutMs
851
+ });
852
+ emitPhaseCompleted(args.onPhase, now, startedAt, "install", result.ok, result.ok
853
+ ? `${args.product} installed`
854
+ : `installing ${args.product} failed`);
855
+ if (!result.ok) {
856
+ // Fail closed: a participant handed a desktop where the product is not installed would produce
857
+ // a transcript about a missing command, and that finding belongs to the harness, not the tool.
858
+ // The tail rides along, scrubbed before truncation like every other provisioning failure — a
859
+ // bare "install failed" is unactionable to whoever wrote the command.
860
+ throw new Error(args.scrub(`desktop-cli install failed for "${args.product}" (${result.timedOut ? "timed out" : `exit ${result.exitCode ?? "?"}`}): ${tailOf(args.scrub(result.logTail))}`));
861
+ }
862
+ }
863
+ /**
864
+ * Open a terminal window on the desktop.
865
+ *
866
+ * The stock template is XFCE and ships xfce4-terminal (also aliased x-terminal-emulator), verified
867
+ * live before this route was built. `x-terminal-emulator` is tried first so a template that swaps
868
+ * the emulator still works; a desktop with neither is a template problem and fails closed rather
869
+ * than handing a participant an empty screen and calling it a study.
870
+ */
871
+ export async function openDesktopTerminal(desktop, requestTimeoutMs, workdir) {
872
+ const dir = workdir ?? "/home/user";
873
+ const result = await runDetachedStep(desktop, {
874
+ name: "desktop-cli-terminal",
875
+ command: [
876
+ "for candidate in x-terminal-emulator xfce4-terminal gnome-terminal konsole xterm; do",
877
+ ' if command -v "$candidate" >/dev/null 2>&1; then',
878
+ // LANG is set on the terminal we open, not globally: the stock image declares no locale, and
879
+ // a study that measures our own mojibake against an unconfigured template would be measuring
880
+ // the template. The PRODUCT-side fix (an ASCII fallback when the locale is not UTF-8) is in
881
+ // src/terminal-encoding.ts, and it is the one that matters for real users.
882
+ ` (cd ${shellSingleQuote(dir)} 2>/dev/null || cd /home/user; DISPLAY=:0 LANG=C.UTF-8 LC_ALL=C.UTF-8 HUMANISH_STUDY_PARTICIPANT=1 nohup "$candidate" >/dev/null 2>&1 &)`,
883
+ " sleep 3",
884
+ ' echo "humanish: opened $candidate"',
885
+ " exit 0",
886
+ " fi",
887
+ "done",
888
+ "echo 'humanish: no terminal emulator on this desktop template' >&2",
889
+ "exit 1"
890
+ ].join("\n"),
891
+ cwd: "/home/user",
892
+ timeoutMs: 60_000,
893
+ requestTimeoutMs
894
+ });
895
+ if (!result.ok) {
896
+ throw new Error("desktop-cli lane could not open a terminal on this desktop template");
897
+ }
898
+ }
899
+ export async function startDesktopStream(desktop, browserWindowId) {
900
+ if (!browserWindowId) {
901
+ await desktop.stream.start({ requireAuth: true });
902
+ return;
903
+ }
904
+ try {
905
+ await desktop.stream.start({ requireAuth: true, windowId: browserWindowId });
906
+ }
907
+ catch {
908
+ await desktop.stream.start({ requireAuth: true });
909
+ }
910
+ }
911
+ /**
912
+ * Shared post-populate provisioning pipeline (clone AND local-tree routes): (install) ->
913
+ * state(before-build) -> (build) -> state(before-start) -> detached start -> readiness probe ->
914
+ * state(after-ready). Both provisioning routes populate SUBJECT_DIR by different means (git
915
+ * clone vs. upload+extract) and then run this identical pipeline unchanged.
916
+ *
917
+ * State steps run through the same detached primitive as serve steps (author-trusted, the
918
+ * "serve commands are author-trusted" corollary) under the reserved `subject-state-<name>`
919
+ * label prefix, so a step name can never collide with subject-clone/subject-extract/install/
920
+ * build/start. after-ready steps complete BEFORE the caller opens the browser: the actor never
921
+ * drives a half-seeded subject and seeding never eats the session budget.
922
+ */
923
+ async function runSubjectServePipeline(desktop, args) {
924
+ const timers = {
925
+ ...(args.now === undefined ? {} : { now: args.now }),
926
+ ...(args.sleep === undefined ? {} : { sleep: args.sleep })
927
+ };
928
+ const now = args.now ?? Date.now;
929
+ const refresh = args.onPhaseComplete ?? (() => Promise.resolve());
930
+ const stateSteps = args.state?.seed ?? [];
931
+ const runStateSteps = async (when) => {
932
+ const steps = stateSteps.filter((step) => (step.when ?? "before-start") === when);
933
+ if (steps.length === 0) {
934
+ // No declared steps for this group: no boundary to report (avoids empty-group noise on
935
+ // every run, since before-build/before-start/after-ready are always called).
936
+ return;
937
+ }
938
+ const groupStartedAt = now();
939
+ emitPhaseStarted(args.onPhase, now, `state.${when}`, `running subject state seed steps (${when})`);
940
+ for (const step of steps) {
941
+ const stepTimeoutMs = step.timeoutMs ?? DEFAULT_STATE_STEP_TIMEOUT_MS;
942
+ const startedAt = now();
943
+ const result = await runDetachedStep(desktop, {
944
+ name: `subject-state-${step.name}`,
945
+ command: step.command,
946
+ cwd: SUBJECT_DIR,
947
+ timeoutMs: stepTimeoutMs,
948
+ requestTimeoutMs: args.requestTimeoutMs,
949
+ ...timers
950
+ });
951
+ args.onStateStep?.({
952
+ name: step.name,
953
+ when,
954
+ // Digest only (sha256-16): the command text never persists: the lab YAML in the
955
+ // consumer's repo is the plaintext source of truth.
956
+ commandDigest: commandDigestOf(step.command),
957
+ ok: result.ok,
958
+ ...(result.exitCode === undefined ? {} : { exitCode: result.exitCode }),
959
+ ...(result.timedOut ? { timedOut: true } : {}),
960
+ durationMs: Math.max(0, now() - startedAt)
961
+ });
962
+ if (!result.ok) {
963
+ emitPhaseCompleted(args.onPhase, now, groupStartedAt, `state.${when}`, false, `subject state seed steps failed (${when})`);
964
+ // Fail closed with the existing scrub-before-truncate tail chain: literal scrub of
965
+ // every provisioned value PRE-truncation, then pattern redaction + cap in tailOf.
966
+ throw new Error(`subject state step "${step.name}" ${result.timedOut ? `timed out after ${stepTimeoutMs}ms` : `failed (exit ${result.exitCode})`}: ${tailOf(args.scrub(result.logTail))}`);
967
+ }
968
+ }
969
+ emitPhaseCompleted(args.onPhase, now, groupStartedAt, `state.${when}`, true, `subject state seed steps complete (${when})`);
970
+ };
971
+ // Provide the runtime the pipeline needs before running it (#371). The stock desktop template
972
+ // ships python3 and curl but no Node, so an `npm install` here used to die at exit 127 after the
973
+ // sandbox was already paid for. Probe-first, so a template that ships its own Node pays nothing.
974
+ const serveCommands = [args.serve.install, args.serve.build, args.serve.start];
975
+ if (needsNodeRuntime(serveCommands)) {
976
+ const runtimeStartedAt = now();
977
+ emitPhaseStarted(args.onPhase, now, "runtime", "providing the Node runtime the serve pipeline needs");
978
+ const bootstrap = await runProvisioningStepWithOneRetry(desktop, {
979
+ name: "subject-runtime-node",
980
+ command: nodeBootstrapCommand(),
981
+ cwd: SUBJECT_DIR,
982
+ timeoutMs: args.serve.installTimeoutMs ?? INSTALL_TIMEOUT_MS,
983
+ requestTimeoutMs: args.requestTimeoutMs,
984
+ timers,
985
+ retryPhase: "runtime-retry",
986
+ retryMessage: "Node runtime bootstrap",
987
+ onPhase: args.onPhase,
988
+ now
989
+ });
990
+ let ok = bootstrap.ok;
991
+ const corepack = ok ? corepackCommandFor(serveCommands) : undefined;
992
+ if (corepack) {
993
+ const pm = await runDetachedStep(desktop, {
994
+ name: "subject-runtime-pm",
995
+ command: corepack,
996
+ cwd: SUBJECT_DIR,
997
+ timeoutMs: args.serve.installTimeoutMs ?? INSTALL_TIMEOUT_MS,
998
+ requestTimeoutMs: args.requestTimeoutMs,
999
+ ...timers
1000
+ });
1001
+ ok = pm.ok;
1002
+ }
1003
+ emitPhaseCompleted(args.onPhase, now, runtimeStartedAt, "runtime", ok, ok ? "Node runtime ready" : "could not provide a Node runtime");
1004
+ if (!ok) {
1005
+ throw new Error(`the subject's serve pipeline needs a Node runtime and this desktop template has none, and bootstrapping one failed${bootstrap.attempts === 2 ? " twice" : ""}: ${tailOf(args.scrub(bootstrap.logTail))}. Use execution.desktop.template with an image that ships Node, or change serve.install to a runtime the template provides.`);
1006
+ }
1007
+ }
1008
+ if (args.serve.install) {
1009
+ const installStartedAt = now();
1010
+ emitPhaseStarted(args.onPhase, now, "install", "installing subject dependencies");
1011
+ const install = await runProvisioningStepWithOneRetry(desktop, {
1012
+ name: "subject-install",
1013
+ command: args.serve.install,
1014
+ cwd: SUBJECT_DIR,
1015
+ timeoutMs: args.serve.installTimeoutMs ?? INSTALL_TIMEOUT_MS,
1016
+ requestTimeoutMs: args.requestTimeoutMs,
1017
+ timers,
1018
+ retryPhase: "install-retry",
1019
+ retryMessage: "subject install",
1020
+ onPhase: args.onPhase,
1021
+ now
1022
+ });
1023
+ emitPhaseCompleted(args.onPhase, now, installStartedAt, "install", install.ok, install.ok
1024
+ ? install.attempts === 2
1025
+ ? "subject dependencies installed (on the second attempt)"
1026
+ : "subject dependencies installed"
1027
+ : install.attempts === 2
1028
+ ? "subject install failed twice"
1029
+ : "subject install failed");
1030
+ if (!install.ok) {
1031
+ // Lead with the line a person can act on; npm's own trace follows it (#602).
1032
+ const headline = install.timedOut
1033
+ ? `subject install timed out after ${args.serve.installTimeoutMs ?? INSTALL_TIMEOUT_MS}ms`
1034
+ : install.attempts === 2
1035
+ ? `subject install failed twice (exit ${install.firstExitCode ?? "null"}, then exit ${install.exitCode ?? "null"}); the sandbox could not complete serve.install`
1036
+ : `subject install failed (exit ${install.exitCode ?? "null"})`;
1037
+ throw new Error(`${headline}: ${tailOf(args.scrub(install.logTail))}`);
1038
+ }
1039
+ await refresh();
1040
+ }
1041
+ // before-build: after install, before build (builds that read seeded state, e.g. SSG).
1042
+ // When no build is declared this simply precedes start: equivalent to before-start.
1043
+ await runStateSteps("before-build");
1044
+ await refresh();
1045
+ if (args.serve.build) {
1046
+ const buildStartedAt = now();
1047
+ emitPhaseStarted(args.onPhase, now, "build", "building subject");
1048
+ const build = await runDetachedStep(desktop, {
1049
+ name: "subject-build",
1050
+ command: args.serve.build,
1051
+ cwd: SUBJECT_DIR,
1052
+ timeoutMs: args.serve.buildTimeoutMs ?? BUILD_TIMEOUT_MS,
1053
+ requestTimeoutMs: args.requestTimeoutMs,
1054
+ ...timers
1055
+ });
1056
+ emitPhaseCompleted(args.onPhase, now, buildStartedAt, "build", build.ok, build.ok ? "subject build complete" : "subject build failed");
1057
+ if (!build.ok) {
1058
+ throw new Error(`subject build ${build.timedOut ? "timed out" : `failed (exit ${build.exitCode})`}: ${tailOf(args.scrub(build.logTail))}`);
1059
+ }
1060
+ await refresh();
1061
+ }
1062
+ // before-start (the default phase): migrations, SQL/file fixtures, an in-sandbox DB server
1063
+ // (`sudo service postgresql start && pg_isready` is a bounded step; the daemon it forks is
1064
+ // reclaimed by the sandbox lifecycle like everything else).
1065
+ await runStateSteps("before-start");
1066
+ await refresh();
1067
+ await startDetachedProcess(desktop, {
1068
+ name: "subject-start",
1069
+ command: args.serve.start,
1070
+ cwd: SUBJECT_DIR,
1071
+ requestTimeoutMs: args.requestTimeoutMs
1072
+ });
1073
+ // Fire-and-forget: startDetachedProcess never waits for the long-lived server to exit, so
1074
+ // there is no matching completed event here (no ok/durationMs to report yet); readiness is
1075
+ // the next boundary.
1076
+ args.onPhase?.({ at: isoNow(now), type: "cua-lab.subject.serve.started", message: "subject server launched (detached)" });
1077
+ const readyStartedAt = now();
1078
+ emitPhaseStarted(args.onPhase, now, "ready", "waiting for subject to become ready");
1079
+ const ready = await probeUrl(desktop, args.serve.url, {
1080
+ timeoutMs: args.serve.readyTimeoutMs ?? DEFAULT_READY_TIMEOUT_MS,
1081
+ requestTimeoutMs: args.requestTimeoutMs,
1082
+ ...timers
1083
+ });
1084
+ emitPhaseCompleted(args.onPhase, now, readyStartedAt, "ready", ready, ready ? "subject is ready" : "subject did not become ready in time");
1085
+ if (!ready) {
1086
+ const startLog = await readDetachedLog(desktop, "subject-start", args.requestTimeoutMs).catch(() => "");
1087
+ throw new Error(`subject did not answer at ${args.serve.url} within ${args.serve.readyTimeoutMs ?? DEFAULT_READY_TIMEOUT_MS}ms; server log tail: ${tailOf(args.scrub(startLog))}`);
1088
+ }
1089
+ // after-ready: fixture loading through the RUNNING app (loopback curl from in-sandbox:
1090
+ // steps are author-trusted provisioning, not actors, so no new URL policy surface). These
1091
+ // complete before the caller opens the browser and the session timer starts.
1092
+ await runStateSteps("after-ready");
1093
+ await refresh();
1094
+ }
1095
+ /**
1096
+ * Provision a clone subject inside the sandbox: clone → the shared serve pipeline
1097
+ * (install → state(before-build) → build → state(before-start) → start → readiness
1098
+ * probe → state(after-ready)). Returns the latest subject HEAD after successful
1099
+ * provisioning. Throws (with a capped log tail for the caller to redact) on any failing step:
1100
+ * the lab persists that as a failed-evidence bundle.
1101
+ *
1102
+ * Auth: when GITHUB_TOKEN is among the declared subject env names, the clone authenticates
1103
+ * via an Authorization header computed IN-SANDBOX from the provisioned env: the token never
1104
+ * appears in the script text, the process argv beyond the transient git call, the clone URL,
1105
+ * or .git/config.
1106
+ */
1107
+ export async function provisionCloneSubject(desktop, args) {
1108
+ const timers = {
1109
+ ...(args.now === undefined ? {} : { now: args.now }),
1110
+ ...(args.sleep === undefined ? {} : { sleep: args.sleep })
1111
+ };
1112
+ const now = args.now ?? Date.now;
1113
+ let latestCommit;
1114
+ const refreshCommit = async () => {
1115
+ const head = await desktop.commands.run(`git -C ${SUBJECT_DIR} rev-parse HEAD 2>/dev/null || true`, { requestTimeoutMs: args.requestTimeoutMs });
1116
+ const commit = (head.stdout ?? "").trim() || undefined;
1117
+ if (commit) {
1118
+ latestCommit = commit;
1119
+ args.onCommit?.(commit);
1120
+ }
1121
+ };
1122
+ const cloneCommand = args.hasGithubToken
1123
+ ? `auth=$(printf 'x-access-token:%s' "$GITHUB_TOKEN" | base64 -w0) && git -c http.extraHeader="Authorization: Basic $auth" clone --depth ${args.depth} https://github.com/${args.repo}.git ${SUBJECT_DIR}`
1124
+ : `git clone --depth ${args.depth} https://github.com/${args.repo}.git ${SUBJECT_DIR}`;
1125
+ const cloneStartedAt = now();
1126
+ emitPhaseStarted(args.onPhase, now, "clone", "cloning subject repository");
1127
+ const clone = await runDetachedStep(desktop, {
1128
+ name: "subject-clone",
1129
+ command: cloneCommand,
1130
+ timeoutMs: CLONE_TIMEOUT_MS,
1131
+ requestTimeoutMs: args.requestTimeoutMs,
1132
+ ...timers
1133
+ });
1134
+ emitPhaseCompleted(args.onPhase, now, cloneStartedAt, "clone", clone.ok, clone.ok ? "subject repository cloned" : "subject clone failed");
1135
+ if (!clone.ok) {
1136
+ throw new Error(`subject clone ${clone.timedOut ? "timed out" : `failed (exit ${clone.exitCode})`}: ${tailOf(args.scrub(clone.logTail))}`);
1137
+ }
1138
+ await refreshCommit();
1139
+ await runSubjectServePipeline(desktop, {
1140
+ serve: args.serve,
1141
+ ...(args.state === undefined ? {} : { state: args.state }),
1142
+ requestTimeoutMs: args.requestTimeoutMs,
1143
+ scrub: args.scrub,
1144
+ ...(args.onStateStep === undefined ? {} : { onStateStep: args.onStateStep }),
1145
+ ...(args.onPhase === undefined ? {} : { onPhase: args.onPhase }),
1146
+ onPhaseComplete: refreshCommit,
1147
+ ...timers
1148
+ });
1149
+ return latestCommit;
1150
+ }
1151
+ /**
1152
+ * Provision a local-tree subject inside the sandbox: upload the once-per-run packed archive
1153
+ * (identical bytes across every fan-out lane) → extract it into SUBJECT_DIR → the
1154
+ * same shared serve pipeline provisionCloneSubject uses. Unlike the clone route there is no
1155
+ * in-sandbox git refresh: the archive excludes .git entirely (see source-archive.ts), so
1156
+ * subject identity is the host-side LocalTreeArchive captured at pack time, never anything
1157
+ * resolved in-sandbox.
1158
+ */
1159
+ export async function provisionLocalTreeSubject(desktop, args) {
1160
+ const timers = {
1161
+ ...(args.now === undefined ? {} : { now: args.now }),
1162
+ ...(args.sleep === undefined ? {} : { sleep: args.sleep })
1163
+ };
1164
+ const now = args.now ?? Date.now;
1165
+ const uploadStartedAt = now();
1166
+ emitPhaseStarted(args.onPhase, now, "upload", "uploading packed local-tree archive");
1167
+ try {
1168
+ await withOneRetryOnTransientE2BError(() => desktop.files.write(LOCAL_TREE_REMOTE_ARCHIVE_PATH, args.archiveBuffer, {
1169
+ requestTimeoutMs: args.requestTimeoutMs,
1170
+ useOctetStream: true
1171
+ }), {
1172
+ onRetry: (reason) => emitPhaseStarted(args.onPhase, now, "upload-retry", `local-tree archive upload retried once (${tailOf(args.scrub(reason))})`),
1173
+ ...(args.sleep === undefined ? {} : { sleep: args.sleep })
1174
+ });
1175
+ }
1176
+ catch (error) {
1177
+ emitPhaseCompleted(args.onPhase, now, uploadStartedAt, "upload", false, "local-tree archive upload failed");
1178
+ throw new Error(`subject-upload failed: ${tailOf(args.scrub(toErrorMessage(error)))}`);
1179
+ }
1180
+ emitPhaseCompleted(args.onPhase, now, uploadStartedAt, "upload", true, "local-tree archive uploaded");
1181
+ const extractCommand = `rm -rf ${SUBJECT_DIR} && mkdir -p ${SUBJECT_DIR} && tar -xzf ${LOCAL_TREE_REMOTE_ARCHIVE_PATH} -C ${SUBJECT_DIR} && rm -f ${LOCAL_TREE_REMOTE_ARCHIVE_PATH}`;
1182
+ const extractStartedAt = now();
1183
+ emitPhaseStarted(args.onPhase, now, "extract", "extracting local-tree archive");
1184
+ const extract = await runDetachedStep(desktop, {
1185
+ name: "subject-extract",
1186
+ command: extractCommand,
1187
+ timeoutMs: CLONE_TIMEOUT_MS,
1188
+ requestTimeoutMs: args.requestTimeoutMs,
1189
+ ...timers
1190
+ });
1191
+ emitPhaseCompleted(args.onPhase, now, extractStartedAt, "extract", extract.ok, extract.ok ? "local-tree archive extracted" : "local-tree archive extraction failed");
1192
+ if (!extract.ok) {
1193
+ throw new Error(`subject extract ${extract.timedOut ? "timed out" : `failed (exit ${extract.exitCode})`}: ${tailOf(args.scrub(extract.logTail))}`);
1194
+ }
1195
+ await runSubjectServePipeline(desktop, {
1196
+ serve: args.serve,
1197
+ ...(args.state === undefined ? {} : { state: args.state }),
1198
+ requestTimeoutMs: args.requestTimeoutMs,
1199
+ scrub: args.scrub,
1200
+ ...(args.onStateStep === undefined ? {} : { onStateStep: args.onStateStep }),
1201
+ ...(args.onPhase === undefined ? {} : { onPhase: args.onPhase }),
1202
+ ...timers
1203
+ });
1204
+ }
1205
+ /** sha256 hex of the exact command string, first 16 chars (the promptDigest convention). */
1206
+ export function commandDigestOf(command) {
1207
+ return digestText(command, 16);
1208
+ }
1209
+ // The in-sandbox `tail -c` upstream is a fundamental log-tail limit we cannot redact past.
1210
+ function tailOf(log) {
1211
+ return redactedTail(log, ERROR_TAIL_CHARS);
1212
+ }
1213
+ //# sourceMappingURL=e2b-cua-provisioning.js.map