humanish 0.96.0 → 0.97.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +4 -1
- package/dist/automatic-analysis-config.d.ts +15 -5
- package/dist/automatic-analysis-config.js +25 -4
- package/dist/automatic-analysis-config.js.map +1 -1
- package/dist/automatic-study-analysis.js +4 -2
- package/dist/automatic-study-analysis.js.map +1 -1
- package/dist/cua-actor-lab.d.ts +18 -286
- package/dist/cua-actor-lab.js +151 -1955
- package/dist/cua-actor-lab.js.map +1 -1
- package/dist/cua-desktop-lane.d.ts +35 -0
- package/dist/cua-desktop-lane.js +13 -0
- package/dist/cua-desktop-lane.js.map +1 -0
- package/dist/desktop-session.d.ts +41 -0
- package/dist/desktop-session.js +44 -0
- package/dist/desktop-session.js.map +1 -0
- package/dist/doctor-lab.d.ts +6 -1
- package/dist/doctor-lab.js +17 -3
- package/dist/doctor-lab.js.map +1 -1
- package/dist/e2b-cua-desktop.d.ts +3 -0
- package/dist/e2b-cua-desktop.js +675 -0
- package/dist/e2b-cua-desktop.js.map +1 -0
- package/dist/e2b-cua-provisioning.d.ts +311 -0
- package/dist/e2b-cua-provisioning.js +1213 -0
- package/dist/e2b-cua-provisioning.js.map +1 -0
- package/dist/e2b-desktop-session.d.ts +7 -0
- package/dist/e2b-desktop-session.js +29 -0
- package/dist/e2b-desktop-session.js.map +1 -0
- package/dist/index.d.ts +1 -1
- package/dist/lab-summary.d.ts +3 -1
- package/dist/local-agent-cli.js +1 -1
- package/dist/local-agent-cli.js.map +1 -1
- package/dist/observer-app.html +2 -2
- package/dist/program.js +24 -9
- package/dist/program.js.map +1 -1
- package/dist/restricted-codex-analysis.d.ts +15 -0
- package/dist/restricted-codex-analysis.js +13 -0
- package/dist/restricted-codex-analysis.js.map +1 -0
- package/dist/restricted-codex-policy.d.ts +56 -0
- package/dist/restricted-codex-policy.js +151 -0
- package/dist/restricted-codex-policy.js.map +1 -0
- package/dist/restricted-codex-session.d.ts +19 -0
- package/dist/restricted-codex-session.js +419 -0
- package/dist/restricted-codex-session.js.map +1 -0
- package/dist/restricted-codex-transport.d.ts +58 -0
- package/dist/restricted-codex-transport.js +233 -0
- package/dist/restricted-codex-transport.js.map +1 -0
- package/dist/run.d.ts +1 -1
- package/dist/run.js +1 -1
- package/dist/run.js.map +1 -1
- package/dist/study-analysis-codex-config.d.ts +11 -0
- package/dist/study-analysis-codex-config.js +34 -0
- package/dist/study-analysis-codex-config.js.map +1 -0
- package/dist/study-analysis-engine.d.ts +6 -3
- package/dist/study-analysis-engine.js +19 -10
- package/dist/study-analysis-engine.js.map +1 -1
- package/dist/study-analysis-job.d.ts +1 -0
- package/dist/study-analysis-job.js +1 -1
- package/dist/study-analysis-job.js.map +1 -1
- package/dist/study-analysis-provider.d.ts +4 -2
- package/dist/study-analysis-provider.js +1 -1
- package/dist/study-analysis-provider.js.map +1 -1
- package/dist/study-analysis-service.d.ts +3 -0
- package/dist/study-analysis-service.js +24 -6
- package/dist/study-analysis-service.js.map +1 -1
- package/dist/study-analysis-validation.d.ts +29 -5
- package/dist/study-analysis-validation.js +23 -10
- package/dist/study-analysis-validation.js.map +1 -1
- package/dist/study-analysis.d.ts +27 -2
- package/dist/tui-app.js +102 -102
- package/docs/architecture/desktop-sessions.md +80 -0
- package/docs/architecture/restricted-codex-analysis.md +89 -0
- package/docs/contracts/run-bundle.md +7 -4
- package/docs/contracts/schemas.md +1 -1
- package/docs/contracts/study-analysis.md +44 -2
- package/docs/goals/current.md +4 -4
- package/docs/product/automatic-analysis.md +24 -3
- package/docs/ramp/README.md +11 -1
- package/docs/release/0.96.1-browser-navigation.md +22 -0
- package/docs/release/0.97.0-codex-account-analysis.md +45 -0
- package/package.json +1 -1
- package/skills/humanish/SKILL.md +19 -0
package/dist/cua-actor-lab.js
CHANGED
|
@@ -1,7 +1,11 @@
|
|
|
1
|
-
|
|
1
|
+
export { inboxRecipientFor, laneHasInboxRecipient } from "./cua-desktop-lane.js";
|
|
2
|
+
export { CUA_ACTOR_LAB_PROVIDER_METADATA, DEFAULT_MOBILE_USER_AGENT, SANDBOX_CAMERA_PATH, SANDBOX_MEDIA_DIR, SUBJECT_DIR, SYNTHETIC_CAMERA_COMMAND, applyMobileEmulation, buildFillDesktopWindowCommand, captureDesktopBrowserGeometry, commandDigestOf, declaredScreenForRender, desktopBrowserFamily, inspectDesktopScreenGeometry, makeChromeBrowserStateObserver, makeChromeDesktopGeometryObserver, parseXwininfoGeometry, prepareDesktopMedia, provisionCloneSubject, provisionLocalTreeSubject } from "./e2b-cua-provisioning.js";
|
|
2
3
|
import { prepareReceivingRun, receivingPublication } from "./comms-receiving-runtime.js";
|
|
3
|
-
import {
|
|
4
|
+
import { laneHasInboxRecipient } from "./cua-desktop-lane.js";
|
|
5
|
+
import { createE2BCuaDesktopLane } from "./e2b-cua-desktop.js";
|
|
6
|
+
import { DEFAULT_STATE_STEP_TIMEOUT_MS, commandDigestOf, declaredScreenForRender } from "./e2b-cua-provisioning.js";
|
|
4
7
|
import { receivingEmailValidationReason } from "./lab-config.js";
|
|
8
|
+
import { withTransientCommsSecrets } from "./run-narration-secrets.js";
|
|
5
9
|
// The computer-use lab backend: a subject (an app-url the caller provisioned, or a repo the
|
|
6
10
|
// lab clones AND serves in-sandbox) driven by a REGISTRY-RESOLVED computer-use actor inside a
|
|
7
11
|
// hosted E2B desktop. This is the path that makes `actors[].type` load-bearing — the
|
|
@@ -24,52 +28,47 @@ import { receivingEmailValidationReason } from "./lab-config.js";
|
|
|
24
28
|
// harness errors are redacted at THIS boundary; the bundle's `stream.actor` carries the
|
|
25
29
|
// conformant humanish.actor-trace.v1 projection, whose `redaction.screenshots` records the
|
|
26
30
|
// run's actual mode ("raw" | "blurred" | "n/a") — every label downstream derives from it.
|
|
27
|
-
import { resolveAutomaticAnalysis } from "./automatic-analysis-config.js";
|
|
28
|
-
import { completeAutomaticAnalysis, markFinalizedStudyResult } from "./automatic-analysis-completion.js";
|
|
29
|
-
import { desktopMediaValidationReason, taskProtocolValidationReason } from "./lab-config.js";
|
|
30
31
|
import { randomBytes } from "node:crypto";
|
|
31
|
-
import { describeMissingKeys } from "./key-resolution.js";
|
|
32
32
|
import { readFile, realpath, rm } from "node:fs/promises";
|
|
33
33
|
import path from "node:path";
|
|
34
|
+
import { completeAutomaticAnalysis, markFinalizedStudyResult } from "./automatic-analysis-completion.js";
|
|
35
|
+
import { resolveAutomaticAnalysis } from "./automatic-analysis-config.js";
|
|
36
|
+
import { describeMissingKeys } from "./key-resolution.js";
|
|
37
|
+
import { desktopMediaValidationReason, taskProtocolValidationReason } from "./lab-config.js";
|
|
38
|
+
import { pathToFileURL } from "node:url";
|
|
39
|
+
import { toErrorMessage } from "./command-failure.js";
|
|
34
40
|
import { cuaLaneDiagnostics, summarizeCuaDiagnostics } from "./cua-diagnostics.js";
|
|
35
41
|
import { feedbackProofCommands } from "./feedback-proof.js";
|
|
36
|
-
import { runDesktopCommandOrThrow, toErrorMessage } from "./command-failure.js";
|
|
37
|
-
import { pathToFileURL } from "node:url";
|
|
38
|
-
import { beginRunStatus, withRunStatusScope } from "./run-status.js";
|
|
39
|
-
import { adapterScoreFailureMessage, applyBrowserAdapterHooks } from "./adapter-extension.js";
|
|
40
42
|
import { actorRegistry, isCuaActorDescriptor } from "./actor-registry.js";
|
|
41
|
-
import {
|
|
42
|
-
import {
|
|
43
|
-
import { createLocalAgentProvider, detectLocalAgents } from "./local-agent-cli.js";
|
|
44
|
-
import { startAppServerSession } from "./local-agent-appserver.js";
|
|
45
|
-
import { startClaudeSession } from "./local-agent-claude-session.js";
|
|
46
|
-
import { createDesktopSandbox, withOneRetryOnTransientE2BError, loadE2BDesktopModule } from "./e2b-desktop-launch.js";
|
|
47
|
-
import { probeUrl, readDetachedLog, runDetachedStep, startDetachedProcess } from "./e2b-detached.js";
|
|
48
|
-
import { DEFAULT_SANDBOX_CATCH_PORT, collectCommsThread, collectExternalCommsThread, deployCommsCatch, externalCatchHealthy, externalInboxUrl, refreshInboxSurface, writeInboxSurface } from "./comms-sandbox-catch.js";
|
|
43
|
+
import { actorEnding } from "./actor-stop-cause.js";
|
|
44
|
+
import { adapterScoreFailureMessage, applyBrowserAdapterHooks } from "./adapter-extension.js";
|
|
49
45
|
import { FakeInbox } from "./comms-fake-inbox.js";
|
|
50
|
-
import {
|
|
51
|
-
import {
|
|
52
|
-
import { cuaLaneValidationReason, outputTokenLimitValidationReason, isHttpUrl, isLoopbackUrl, MAX_CUA_LANES, subjectStateInvalidReason } from "./lab-config.js";
|
|
46
|
+
import { recipientInboxUrl } from "./comms-inbox.js";
|
|
47
|
+
import { collectExternalCommsThread, externalCatchHealthy, externalInboxUrl } from "./comms-sandbox-catch.js";
|
|
53
48
|
import { mapWithConcurrency } from "./concurrency.js";
|
|
54
|
-
import {
|
|
49
|
+
import { DEFAULT_DEVICE_PRESET, isDevicePresetName, resolveDevicePreset } from "./device-presets.js";
|
|
50
|
+
import {} from "./e2b-desktop-launch.js";
|
|
51
|
+
import {} from "./e2b-desktop-resources.js";
|
|
52
|
+
import {} from "./e2b-detached.js";
|
|
55
53
|
import { assertScreenshotEvidence } from "./image-evidence.js";
|
|
54
|
+
import { MAX_CUA_LANES, cuaLaneValidationReason, isHttpUrl, isLoopbackUrl, outputTokenLimitValidationReason, subjectStateInvalidReason } from "./lab-config.js";
|
|
55
|
+
import { startAppServerSession } from "./local-agent-appserver.js";
|
|
56
|
+
import { startClaudeSession } from "./local-agent-claude-session.js";
|
|
57
|
+
import { createLocalAgentProvider, detectLocalAgents } from "./local-agent-cli.js";
|
|
56
58
|
import { buildObserverData } from "./observer-data.js";
|
|
57
|
-
import { corepackCommandFor, needsNodeRuntime, nodeBootstrapCommand } from "./subject-runtime.js";
|
|
58
|
-
import { TERMINAL_NODE_BOOTSTRAP_COMMAND } from "./terminal-node-bootstrap.js";
|
|
59
|
-
import { chromeCdpProbeCommand, parseChromeCdpProbeOutput } from "./chrome-cdp-probe.js";
|
|
60
|
-
import { personaToDirectives, renderPersonaPromptSection } from "./persona.js";
|
|
61
|
-
import { labPersonaIds, resolveCommittedPersonas } from "./persona-resolve.js";
|
|
62
|
-
import { renderTaskPrompt } from "./tasks.js";
|
|
63
59
|
import { attachObserverRuntimeStreamUrls, renderObserver } from "./observer.js";
|
|
64
|
-
import {
|
|
60
|
+
import { DEFAULT_OPENAI_CU_MODEL } from "./openai-responses-cu.js";
|
|
65
61
|
import { participantAssignment } from "./participant-assignment.js";
|
|
66
|
-
import {
|
|
67
|
-
import {
|
|
62
|
+
import { labPersonaIds, resolveCommittedPersonas } from "./persona-resolve.js";
|
|
63
|
+
import { personaToDirectives, renderPersonaPromptSection } from "./persona.js";
|
|
64
|
+
import { MODEL_RATES, estimateActorCost, estimateAllocatedDesktopCost, estimateDesktopCost, round6 } from "./pricing.js";
|
|
65
|
+
import { containsSensitive, digestText, redactText } from "./redaction.js";
|
|
68
66
|
import { prepareRunArtifactPaths, validatePreparedRunArtifactPaths } from "./run-paths.js";
|
|
67
|
+
import { beginRunStatus, withRunStatusScope } from "./run-status.js";
|
|
68
|
+
import { PUBLIC_TARGET_CWD, REVIEW_SCHEMA, RUN_BUNDLE_SCHEMA, aggregateTaskFunnels, buildRunSource, formatParticipantOutcomes, formatStudyTaskFunnel, loadRunBundle, tallyParticipantOutcomes, withCuaReviewProvenance } from "./run.js";
|
|
69
|
+
import { assertPreparedSelectedOutputDirectory, assertSafeOutputPathSegment, prepareContainedOutputDirectory, prepareSelectedOutputDirectory, writeContainedOutputFile, writePreparedRunLatestPointer } from "./selected-output-paths.js";
|
|
69
70
|
import { createLocalTreeArchive } from "./source-archive.js";
|
|
70
|
-
import {
|
|
71
|
-
import { estimateActorCost, estimateDesktopCost, estimateAllocatedDesktopCost, MODEL_RATES, round6 } from "./pricing.js";
|
|
72
|
-
import { observeDesktopResources } from "./e2b-desktop-resources.js";
|
|
71
|
+
import { renderTaskPrompt } from "./tasks.js";
|
|
73
72
|
export const CUA_ACTOR_LAB_SCHEMA = "humanish.cua-lab-result.v2";
|
|
74
73
|
// The only fan-out topology this slice ships: N lanes = N independent E2B desktop sandboxes,
|
|
75
74
|
// each its own world (clone/serve + subject.state per lane). Shared-world is layer 7 (#164).
|
|
@@ -77,10 +76,6 @@ export const CUA_FANOUT_STRATEGY = "per-lane-worlds";
|
|
|
77
76
|
// Env override that may only LOWER the effective concurrency (never raise concurrent paid
|
|
78
77
|
// desktops — invariant 3). Read names-only into a local; the value never persists.
|
|
79
78
|
const CUA_MAX_CONCURRENCY_ENV = "HUMANISH_CUA_MAX_CONCURRENCY";
|
|
80
|
-
export const CUA_ACTOR_LAB_PROVIDER_METADATA = {
|
|
81
|
-
mode: "cua-actor-lab",
|
|
82
|
-
tool: "humanish"
|
|
83
|
-
};
|
|
84
79
|
// The DEFAULT session budget, sized so a study can FINISH (docs/principles/three-roles.md: a
|
|
85
80
|
// session ends because the participant is done, not because a timer fired — the time-box is a
|
|
86
81
|
// session-level cap a researcher sets generously; spend protection is the dollar caps' job).
|
|
@@ -103,60 +98,6 @@ function defaultSessionTimeoutMs(config) {
|
|
|
103
98
|
const room = MAX_SANDBOX_MS - SUBJECT_PROVISION_BUDGET_MS - stateBudgetMs - SANDBOX_TIMEOUT_BUFFER_MS;
|
|
104
99
|
return Math.max(MIN_DERIVED_SESSION_TIMEOUT_MS, Math.min(DEFAULT_APP_URL_SESSION_TIMEOUT_MS, room));
|
|
105
100
|
}
|
|
106
|
-
// Settle after opening the browser, before the first screenshot — long enough for a cold
|
|
107
|
-
// browser + page load to paint (2s captured a blank desktop; the render empirically needs ~6-9s).
|
|
108
|
-
const BROWSER_SETTLE_MS = 8_000;
|
|
109
|
-
/** Where a lane's synthetic camera feed lives inside the sandbox: a tmpfs the sandbox user can
|
|
110
|
-
* write, and a path that contains neither /tmp/ nor /home/, which the public-safety scan reads
|
|
111
|
-
* as an operator's local path (this one is the harness's own and belongs in the bundle). */
|
|
112
|
-
export const SANDBOX_MEDIA_DIR = "/dev/shm/humanish-media";
|
|
113
|
-
export const SANDBOX_CAMERA_PATH = `${SANDBOX_MEDIA_DIR}/camera.y4m`;
|
|
114
|
-
/** The synthetic feed: ffmpeg's test pattern, 640x480 at 10 fps, six seconds (about 28 MB of
|
|
115
|
-
* raw Y4M on the tmpfs), looped by Chrome's fake capture device. */
|
|
116
|
-
export const SYNTHETIC_CAMERA_COMMAND = `mkdir -p ${SANDBOX_MEDIA_DIR} && ffmpeg -y -loglevel error -f lavfi -i testsrc=size=640x480:rate=10 -t 6 -pix_fmt yuv420p ${SANDBOX_CAMERA_PATH}`;
|
|
117
|
-
/**
|
|
118
|
-
* Put the declared camera feed in the sandbox and return the Chromium flags that present it as a
|
|
119
|
-
* capture device (#509). Fails CLOSED: a feed that cannot be produced (no ffmpeg on the image, an
|
|
120
|
-
* unreadable host file) is named before the browser launches, because a participant told it has
|
|
121
|
-
* a camera and finds none reports the instrument's gap as the product's.
|
|
122
|
-
*/
|
|
123
|
-
export async function prepareDesktopMedia(desktop, media, permission, cwd, requestTimeoutMs, readHostFile = (absolutePath) => readFile(absolutePath)) {
|
|
124
|
-
if (media.microphone !== undefined) {
|
|
125
|
-
throw new Error("execution.desktop.media.microphone.source injection is unsupported; the declared microphone file cannot be delivered, including on custom templates.");
|
|
126
|
-
}
|
|
127
|
-
const flags = [];
|
|
128
|
-
let camera;
|
|
129
|
-
if (media.camera !== undefined) {
|
|
130
|
-
if (media.camera.source === "synthetic") {
|
|
131
|
-
const made = await desktop.commands.run(SYNTHETIC_CAMERA_COMMAND, { requestTimeoutMs, timeoutMs: 60_000 });
|
|
132
|
-
if (made.exitCode !== undefined && made.exitCode !== 0) {
|
|
133
|
-
throw new Error(`the synthetic camera feed could not be generated on this desktop image (ffmpeg exited ${made.exitCode}: ${tailOf(made.stderr ?? made.stdout ?? "")}); give execution.desktop.media.camera.source a .y4m file instead`);
|
|
134
|
-
}
|
|
135
|
-
camera = { source: "synthetic", file: SANDBOX_CAMERA_PATH };
|
|
136
|
-
}
|
|
137
|
-
else {
|
|
138
|
-
const absolutePath = path.resolve(cwd, media.camera.source);
|
|
139
|
-
let bytes;
|
|
140
|
-
try {
|
|
141
|
-
bytes = await readHostFile(absolutePath);
|
|
142
|
-
}
|
|
143
|
-
catch (error) {
|
|
144
|
-
throw new Error(`execution.desktop.media.camera.source could not be read (${toErrorMessage(error)})`);
|
|
145
|
-
}
|
|
146
|
-
if (bytes.length > 64 * 1024 * 1024) {
|
|
147
|
-
throw new Error(`execution.desktop.media.camera.source is ${bytes.length} bytes; the camera feed is capped at 64 MiB`);
|
|
148
|
-
}
|
|
149
|
-
await desktop.commands.run(`mkdir -p ${SANDBOX_MEDIA_DIR}`, { requestTimeoutMs, timeoutMs: 15_000 });
|
|
150
|
-
const payload = bytes.buffer.slice(bytes.byteOffset, bytes.byteOffset + bytes.byteLength);
|
|
151
|
-
await desktop.files.write(SANDBOX_CAMERA_PATH, payload, { requestTimeoutMs, useOctetStream: true });
|
|
152
|
-
camera = { source: "file", file: SANDBOX_CAMERA_PATH };
|
|
153
|
-
}
|
|
154
|
-
flags.push("--use-fake-device-for-media-stream", `--use-file-for-fake-video-capture=${SANDBOX_CAMERA_PATH}`);
|
|
155
|
-
}
|
|
156
|
-
if (permission === "granted")
|
|
157
|
-
flags.push("--use-fake-ui-for-media-stream");
|
|
158
|
-
return { ...(camera === undefined ? {} : { camera }), permission, flags };
|
|
159
|
-
}
|
|
160
101
|
// Device/screen size comes from the named-preset registry (device-presets.ts), selectable per run
|
|
161
102
|
// via execution.desktop.device (default `desktop`=1440x950). NOTE: this is run-wide for now; a
|
|
162
103
|
// per-PERSONA device dimension (N personas × devices, as the bespoke sims author) lands with
|
|
@@ -172,19 +113,6 @@ const SUBJECT_PROVISION_BUDGET_MS = 30 * 60_000;
|
|
|
172
113
|
* The derived per-lane deadline has to stay under it, and saying so at plan time beats discovering
|
|
173
114
|
* it from a raw provider 400 after a plan has already printed. */
|
|
174
115
|
const MAX_SANDBOX_MS = 60 * 60_000;
|
|
175
|
-
export const SUBJECT_DIR = "/home/user/subject";
|
|
176
|
-
// Remote path for the once-per-run packed local-tree archive; removed by the extract step
|
|
177
|
-
// after it unpacks into SUBJECT_DIR.
|
|
178
|
-
const LOCAL_TREE_REMOTE_ARCHIVE_PATH = "/home/user/.humanish-source.tar.gz";
|
|
179
|
-
const CLONE_TIMEOUT_MS = 5 * 60_000;
|
|
180
|
-
const INSTALL_TIMEOUT_MS = 10 * 60_000;
|
|
181
|
-
const BUILD_TIMEOUT_MS = 10 * 60_000;
|
|
182
|
-
const DEFAULT_READY_TIMEOUT_MS = 180_000;
|
|
183
|
-
// Per-step budget for subject.state seed steps; each step's declared (or default) budget is
|
|
184
|
-
// also summed into the default sandbox deadline so seeding never eats the session's room.
|
|
185
|
-
const DEFAULT_STATE_STEP_TIMEOUT_MS = 5 * 60_000;
|
|
186
|
-
// How much of a failing step's log tail rides the (redacted) error message.
|
|
187
|
-
const ERROR_TAIL_CHARS = 2000;
|
|
188
116
|
const DEFAULT_MISSION = "You are testing a web application. The browser is already open at the subject URL. Explore it, accomplish what the scenario asks, and stop when done.";
|
|
189
117
|
/**
|
|
190
118
|
* The participant's outcome as ONE fixed first line of its last message (#570, second half). The
|
|
@@ -262,21 +190,6 @@ export function withInboxMission(spec, inboxUrl, address, receiving = false) {
|
|
|
262
190
|
instructions: `${spec.instructions}\n\nEmail inbox:${identity} When the app tells you it has emailed you (a verification link, confirmation code, or magic link), open ${recipientInboxUrl(inboxUrl, address)} in the browser to read that email and follow its link or enter its code. All email the app sends you arrives there. Waiting for an email is normal, not a blocker — do not end your session while waiting; open the inbox and refresh it until the email appears.`
|
|
263
191
|
};
|
|
264
192
|
}
|
|
265
|
-
/** The lane's addressed comms recipient, when one exists — the gate AND the address source for the
|
|
266
|
-
* inbox instruction (#351). A lane told to check an inbox it can never receive into would stall,
|
|
267
|
-
* so no addressed recipient means no instruction. */
|
|
268
|
-
export function inboxRecipientFor(commsEmail, laneId) {
|
|
269
|
-
return (commsEmail.recipients ?? []).find((recipient) => recipient.lane === laneId && recipient.address !== undefined);
|
|
270
|
-
}
|
|
271
|
-
/** True when a lane has a declared comms recipient WITH an address, so the drain can actually match the
|
|
272
|
-
* mail the persona will be told to read. Gates the inbox instruction to lanes that can receive mail —
|
|
273
|
-
* a lane told to check an inbox it can never receive into would just stall. */
|
|
274
|
-
export function laneHasInboxRecipient(commsEmail, laneId) {
|
|
275
|
-
return inboxRecipientFor(commsEmail, laneId) !== undefined;
|
|
276
|
-
}
|
|
277
|
-
/** Mid-run inbox-surface render cadence (ms). Coarse enough that the per-tick `cat` + file writes stay
|
|
278
|
-
* cheap; fine enough that a verification email is visible seconds after the app sends it. */
|
|
279
|
-
const INBOX_SURFACE_CADENCE_MS = 2500;
|
|
280
193
|
/**
|
|
281
194
|
* The narrowest browser WINDOW Chrome/Chromium will render on the E2B desktop. Chrome refuses to
|
|
282
195
|
* make its window narrower than this (~500 CSS px observed: a 414-wide X screen produced a 500-wide
|
|
@@ -292,20 +205,6 @@ export const MIN_DESKTOP_RENDER_WIDTH = 500;
|
|
|
292
205
|
export function floorRenderResolution(resolution) {
|
|
293
206
|
return [Math.max(resolution[0], MIN_DESKTOP_RENDER_WIDTH), resolution[1]];
|
|
294
207
|
}
|
|
295
|
-
/**
|
|
296
|
-
* The DECLARED preset to record alongside the rendered screen, or undefined when the preset
|
|
297
|
-
* rendered faithfully.
|
|
298
|
-
*
|
|
299
|
-
* `desktopGeometry.screen.verified` compares the FLOORED number with itself, so on its own a
|
|
300
|
-
* floored run is indistinguishable from a faithful one: a reader sees requested 500 / verified 500
|
|
301
|
-
* and concludes a 500-wide screen was asked for. Recording the declared preset is what makes
|
|
302
|
-
* "the preset width did not render" legible in the bundle.
|
|
303
|
-
*/
|
|
304
|
-
export function declaredScreenForRender(preset, presetName, rendered) {
|
|
305
|
-
if (preset.width === rendered[0] && preset.height === rendered[1])
|
|
306
|
-
return undefined;
|
|
307
|
-
return { width: preset.width, height: preset.height, preset: presetName };
|
|
308
|
-
}
|
|
309
208
|
/**
|
|
310
209
|
* Resolve a lane's device + rendered resolution (most-specific wins, exactly as the single-lane
|
|
311
210
|
* path always has): a raw execution.desktop.resolution escape hatch (only legal when no lane
|
|
@@ -582,65 +481,6 @@ function formatLanePlanEntry(lane) {
|
|
|
582
481
|
].filter((part) => part !== undefined);
|
|
583
482
|
return `${lane.id}: persona=${lane.persona}${taxonomy.length > 0 ? ` ${taxonomy.join(" ")}` : ""} device=${lane.device} ${lane.resolution[0]}x${lane.resolution[1]} prompt#${lane.instructionDigest}${lane.targetDigest ? ` target#${lane.targetDigest}` : ""}`;
|
|
584
483
|
}
|
|
585
|
-
/** ISO timestamp from an injectable clock (tests freeze `now` for deterministic durationMs). */
|
|
586
|
-
function isoNow(now) {
|
|
587
|
-
return new Date(now()).toISOString();
|
|
588
|
-
}
|
|
589
|
-
/** Emit a phase-started event (no ok/durationMs: those belong to the matching completed event). */
|
|
590
|
-
/**
|
|
591
|
-
* Run a provisioning step and, when it fails with an EXIT CODE, run it once more (#602). A cold
|
|
592
|
-
* install of 0.74.0 lost its whole first live study to one transient TLS error inside the
|
|
593
|
-
* sandbox's `npm install`; the parallel install twenty seconds later passed, as had the ten
|
|
594
|
-
* before it. One retry clears that class. A TIMEOUT is not retried: its budget is already spent,
|
|
595
|
-
* and a second wait would double it. The retry runs under its own step name so both logs stay.
|
|
596
|
-
*/
|
|
597
|
-
async function runProvisioningStepWithOneRetry(desktop, args) {
|
|
598
|
-
const first = await runDetachedStep(desktop, {
|
|
599
|
-
name: args.name,
|
|
600
|
-
command: args.command,
|
|
601
|
-
cwd: args.cwd,
|
|
602
|
-
timeoutMs: args.timeoutMs,
|
|
603
|
-
requestTimeoutMs: args.requestTimeoutMs,
|
|
604
|
-
...args.timers
|
|
605
|
-
});
|
|
606
|
-
if (first.ok || first.timedOut)
|
|
607
|
-
return { ...first, attempts: 1 };
|
|
608
|
-
const retryStartedAt = args.now();
|
|
609
|
-
emitPhaseStarted(args.onPhase, args.now, args.retryPhase, `${args.retryMessage} (first attempt exited ${first.exitCode ?? "null"}; retrying once)`);
|
|
610
|
-
const second = await runDetachedStep(desktop, {
|
|
611
|
-
name: `${args.name}-retry`,
|
|
612
|
-
command: args.command,
|
|
613
|
-
cwd: args.cwd,
|
|
614
|
-
timeoutMs: args.timeoutMs,
|
|
615
|
-
requestTimeoutMs: args.requestTimeoutMs,
|
|
616
|
-
...args.timers
|
|
617
|
-
});
|
|
618
|
-
emitPhaseCompleted(args.onPhase, args.now, retryStartedAt, args.retryPhase, second.ok, second.ok ? `${args.retryMessage}: succeeded on the second attempt` : `${args.retryMessage}: failed twice`);
|
|
619
|
-
return { ...second, attempts: 2, ...(first.exitCode === undefined ? {} : { firstExitCode: first.exitCode }) };
|
|
620
|
-
}
|
|
621
|
-
function emitPhaseStarted(onPhase, now, phase, message) {
|
|
622
|
-
onPhase?.({ at: isoNow(now), type: `cua-lab.subject.${phase}.started`, message });
|
|
623
|
-
}
|
|
624
|
-
/** Emit the matching phase-completed event: always carries ok and durationMs (>= 0). */
|
|
625
|
-
function emitPhaseCompleted(onPhase, now, startedAt, phase, ok, message) {
|
|
626
|
-
onPhase?.({
|
|
627
|
-
at: isoNow(now),
|
|
628
|
-
type: `cua-lab.subject.${phase}.completed`,
|
|
629
|
-
ok,
|
|
630
|
-
durationMs: Math.max(0, now() - startedAt),
|
|
631
|
-
message
|
|
632
|
-
});
|
|
633
|
-
}
|
|
634
|
-
/** Default phase-boundary sink (stderr): one line per event, prefixed with the lane id ONLY
|
|
635
|
-
* when laneCount > 1. Single-lane emission is unconditional: total single-lane silence for the
|
|
636
|
-
* whole clone/install/build/ready boot is the bug this event stream exists to close.
|
|
637
|
-
* Overridable via CuaActorLabHooks.onPhase so deterministic tests capture instead of writing to
|
|
638
|
-
* the real stderr. */
|
|
639
|
-
function defaultSubjectPhaseSink(event, ctx) {
|
|
640
|
-
const durationSuffix = event.durationMs === undefined ? "" : ` (${event.durationMs}ms)`;
|
|
641
|
-
const prefix = ctx.laneCount > 1 ? `humanish cua [${ctx.laneId}]` : "humanish cua";
|
|
642
|
-
process.stderr.write(`${prefix}: ${event.message}${durationSuffix}\n`);
|
|
643
|
-
}
|
|
644
484
|
/** Short id-safe suffix for a subject-phase RunEvent: drops the shared prefix/suffix so each
|
|
645
485
|
* phase gets a distinct bundle event id (e.g. "clone", "state-before-build"). */
|
|
646
486
|
function phaseEventIdSuffix(type) {
|
|
@@ -681,36 +521,6 @@ export function makeLaneWriteScreenshot(artifactRoot, spec, screenshots) {
|
|
|
681
521
|
return rel;
|
|
682
522
|
};
|
|
683
523
|
}
|
|
684
|
-
/**
|
|
685
|
-
* Verify the desktop screen geometry IN-SANDBOX (the per-lane device claim is checked, never
|
|
686
|
-
* assumed). A parseable mismatch fails closed. Unavailable/unparseable evidence is returned as
|
|
687
|
-
* an explicit warning: the lane may still run, but its bundle records only the requested screen
|
|
688
|
-
* and never upgrades that request into a verified measurement.
|
|
689
|
-
*/
|
|
690
|
-
export async function inspectDesktopScreenGeometry(args) {
|
|
691
|
-
let out = "";
|
|
692
|
-
try {
|
|
693
|
-
const result = await args.desktop.commands.run("xdpyinfo 2>/dev/null | grep -i dimensions || true", { requestTimeoutMs: args.requestTimeoutMs });
|
|
694
|
-
out = (result.stdout ?? "").trim();
|
|
695
|
-
}
|
|
696
|
-
catch {
|
|
697
|
-
return { warning: `Desktop screen geometry could not be measured for lane ${args.laneId}; requested geometry remains unverified.` };
|
|
698
|
-
}
|
|
699
|
-
const match = out.match(/(\d+)\s*x\s*(\d+)\s*pixels/i);
|
|
700
|
-
if (!match) {
|
|
701
|
-
return { warning: `Desktop screen geometry could not be parsed for lane ${args.laneId}; requested geometry remains unverified.` };
|
|
702
|
-
}
|
|
703
|
-
const width = Number(match[1]);
|
|
704
|
-
const height = Number(match[2]);
|
|
705
|
-
const [expectedWidth, expectedHeight] = args.requestedScreen;
|
|
706
|
-
if (width === expectedWidth && height === expectedHeight) {
|
|
707
|
-
return { verified: { width, height, source: "xdpyinfo" } };
|
|
708
|
-
}
|
|
709
|
-
return {
|
|
710
|
-
verified: { width, height, source: "xdpyinfo" },
|
|
711
|
-
error: `HUMANISH_CUA_LAB_DEVICE_GEOMETRY: lane ${args.laneId} requested a ${expectedWidth}x${expectedHeight} desktop but xdpyinfo reports ${width}x${height} in-sandbox; the per-lane device geometry is unverified (fail-closed).`
|
|
712
|
-
};
|
|
713
|
-
}
|
|
714
524
|
/** A blocked lane outcome (pipeline gate / fail-fast skipped it before it ran). */
|
|
715
525
|
function blockedLaneOutcome(spec, reason) {
|
|
716
526
|
return {
|
|
@@ -728,709 +538,6 @@ function blockedLaneOutcome(spec, reason) {
|
|
|
728
538
|
harnessError: false
|
|
729
539
|
};
|
|
730
540
|
}
|
|
731
|
-
async function findVisibleBrowserWindowId(desktop, requestTimeoutMs, browserFamily, launchIdentity) {
|
|
732
|
-
if (browserFamily === "unknown")
|
|
733
|
-
return undefined;
|
|
734
|
-
// The candidate loop keeps the LAST identity match: with a launch identity the match is
|
|
735
|
-
// unique anyway, and without one every family candidate matches, so the newest visible
|
|
736
|
-
// window of the launched family wins (the window this lane just opened).
|
|
737
|
-
const finder = browserFamily === "firefox"
|
|
738
|
-
? [
|
|
739
|
-
"find_firefox_window() {",
|
|
740
|
-
" timeout 2s xdotool search --onlyvisible --class 'firefox|Firefox' 2>/dev/null || true",
|
|
741
|
-
"}",
|
|
742
|
-
"window_id=",
|
|
743
|
-
"for _ in $(seq 1 10); do",
|
|
744
|
-
" for candidate in $(find_firefox_window); do",
|
|
745
|
-
" window_pid=\"$(xdotool getwindowpid \"$candidate\" 2>/dev/null || true)\"",
|
|
746
|
-
" if matches_launch_identity \"$window_pid\"; then window_id=\"$candidate\"; fi",
|
|
747
|
-
" done",
|
|
748
|
-
" if [ -n \"$window_id\" ]; then break; fi",
|
|
749
|
-
" sleep 0.5",
|
|
750
|
-
"done"
|
|
751
|
-
]
|
|
752
|
-
: [
|
|
753
|
-
"find_chrome_window() {",
|
|
754
|
-
" timeout 2s xdotool search --onlyvisible --class 'google-chrome|Google-chrome|chromium|Chromium|chrome|Chrome' 2>/dev/null || true",
|
|
755
|
-
"}",
|
|
756
|
-
"window_id=",
|
|
757
|
-
"for _ in $(seq 1 10); do",
|
|
758
|
-
" for candidate in $(find_chrome_window); do",
|
|
759
|
-
" window_pid=\"$(xdotool getwindowpid \"$candidate\" 2>/dev/null || true)\"",
|
|
760
|
-
" if matches_launch_identity \"$window_pid\"; then window_id=\"$candidate\"; fi",
|
|
761
|
-
" done",
|
|
762
|
-
" if [ -n \"$window_id\" ]; then break; fi",
|
|
763
|
-
" sleep 0.5",
|
|
764
|
-
"done"
|
|
765
|
-
];
|
|
766
|
-
const result = await desktop.commands.run([
|
|
767
|
-
"set -euo pipefail",
|
|
768
|
-
"export DISPLAY=\"${DISPLAY:-:0}\"",
|
|
769
|
-
`launch_pid=${shellSingleQuote(launchIdentity?.processId ?? "")}`,
|
|
770
|
-
`profile_dir=${shellSingleQuote(launchIdentity?.profileDir ?? "")}`,
|
|
771
|
-
"matches_launch_identity() {",
|
|
772
|
-
" if [ -z \"$launch_pid\" ] && [ -z \"$profile_dir\" ]; then return 0; fi",
|
|
773
|
-
" local current=\"${1:-}\"",
|
|
774
|
-
" while [[ \"$current\" =~ ^[0-9]+$ ]] && [ \"$current\" -gt 1 ]; do",
|
|
775
|
-
" cmdline=\"$(tr '\\0' ' ' < \"/proc/$current/cmdline\" 2>/dev/null || true)\"",
|
|
776
|
-
" if [ -n \"$profile_dir\" ] && [[ \"$cmdline\" == *\"$profile_dir\"* ]]; then return 0; fi",
|
|
777
|
-
" if [ \"$current\" = \"$launch_pid\" ]; then return 0; fi",
|
|
778
|
-
" current=\"$(ps -o ppid= -p \"$current\" 2>/dev/null | tr -d ' ' || true)\"",
|
|
779
|
-
" done",
|
|
780
|
-
" return 1",
|
|
781
|
-
"}",
|
|
782
|
-
...finder,
|
|
783
|
-
"if [ -n \"$window_id\" ]; then printf 'WINDOW_ID=%s\\n' \"$window_id\"; fi"
|
|
784
|
-
].join("\n"), {
|
|
785
|
-
requestTimeoutMs,
|
|
786
|
-
timeoutMs: 15_000
|
|
787
|
-
});
|
|
788
|
-
return (result.stdout ?? "").match(/^WINDOW_ID=(\S+)$/m)?.[1];
|
|
789
|
-
}
|
|
790
|
-
/**
|
|
791
|
-
* Build the xdotool command that makes a browser window fill the desktop.
|
|
792
|
-
* Exported (pure) for contract tests. A window manager can ignore Chrome's
|
|
793
|
-
* --window-size, so xdotool is the robust path: move the window to the origin,
|
|
794
|
-
* then size it to the exact desktop resolution so Observer screenshots carry no
|
|
795
|
-
* dead margin around the browser.
|
|
796
|
-
*/
|
|
797
|
-
export function buildFillDesktopWindowCommand(windowId, width, height) {
|
|
798
|
-
return [
|
|
799
|
-
"set -euo pipefail",
|
|
800
|
-
`win=${shellSingleQuote(windowId)}`,
|
|
801
|
-
`xdotool windowactivate "$win" >/dev/null 2>&1 || true`,
|
|
802
|
-
`xdotool windowmove "$win" 0 0 >/dev/null 2>&1 || true`,
|
|
803
|
-
`xdotool windowsize "$win" ${width} ${height} >/dev/null 2>&1 || true`,
|
|
804
|
-
].join("\n");
|
|
805
|
-
}
|
|
806
|
-
/**
|
|
807
|
-
* Best-effort initial fill. A contained smaller window remains usable; the capture
|
|
808
|
-
* below checks for clipping and refuses an uncorrectable window before the actor runs.
|
|
809
|
-
*/
|
|
810
|
-
async function fillDesktopBrowserWindow(desktop, windowId, resolution, requestTimeoutMs) {
|
|
811
|
-
const [width, height] = resolution;
|
|
812
|
-
await desktop.commands
|
|
813
|
-
.run(buildFillDesktopWindowCommand(windowId, width, height), {
|
|
814
|
-
requestTimeoutMs,
|
|
815
|
-
timeoutMs: 10_000,
|
|
816
|
-
})
|
|
817
|
-
.catch(() => undefined);
|
|
818
|
-
}
|
|
819
|
-
async function openDesktopBrowserTarget(desktop, targetUrl, requestTimeoutMs, browserPreference,
|
|
820
|
-
/** Launch-time flags that make mobile fidelity (#221) hold across every tab: the user agent and
|
|
821
|
-
* touch events are browser-wide here, where the CDP holder covers only the launch page. */
|
|
822
|
-
extraChromiumFlags = []) {
|
|
823
|
-
const requestedBrowser = browserPreference ?? "default";
|
|
824
|
-
if (isHttpUrl(targetUrl)) {
|
|
825
|
-
const chromiumFlags = [...CHROMIUM_EVIDENCE_HYGIENE_FLAGS, ...extraChromiumFlags].map(shellSingleQuote).join(" ");
|
|
826
|
-
const browserLaunchCommand = [
|
|
827
|
-
"set -euo pipefail",
|
|
828
|
-
`target_url=${shellSingleQuote(targetUrl)}`,
|
|
829
|
-
`browser_preference=${shellSingleQuote(requestedBrowser)}`,
|
|
830
|
-
"chrome_profile_dir=",
|
|
831
|
-
`chrome_preferences_json=${shellSingleQuote(chromiumEvidenceProfilePreferencesJson())}`,
|
|
832
|
-
"prepare_chrome_profile() {",
|
|
833
|
-
" chrome_profile_dir=\"$(mktemp -d /tmp/humanish-chrome-profile.XXXXXX)\"",
|
|
834
|
-
" mkdir -p \"$chrome_profile_dir/Default\"",
|
|
835
|
-
" printf '%s\\n' \"$chrome_preferences_json\" > \"$chrome_profile_dir/Default/Preferences\"",
|
|
836
|
-
"}",
|
|
837
|
-
"launch_browser() {",
|
|
838
|
-
" local label=\"$1\"",
|
|
839
|
-
" local binary=\"$2\"",
|
|
840
|
-
" shift 2",
|
|
841
|
-
" if command -v \"$binary\" >/dev/null 2>&1; then",
|
|
842
|
-
" nohup \"$binary\" \"$@\" \"$target_url\" >/tmp/humanish-browser-open.log 2>&1 &",
|
|
843
|
-
" local launch_pid=$!",
|
|
844
|
-
" printf 'HUMANISH_BROWSER_RESOLVED=%s\\n' \"$label\"",
|
|
845
|
-
" printf 'HUMANISH_BROWSER_PID=%s\\n' \"$launch_pid\"",
|
|
846
|
-
" printf 'HUMANISH_BROWSER_PROFILE_DIR=%s\\n' \"$chrome_profile_dir\"",
|
|
847
|
-
" if [[ \"$label\" =~ ^(google-chrome|google-chrome-stable|chromium|chromium-browser)$ ]]; then",
|
|
848
|
-
" for _ in $(seq 1 30); do",
|
|
849
|
-
" if [ -s \"$chrome_profile_dir/DevToolsActivePort\" ]; then",
|
|
850
|
-
" head -n 1 \"$chrome_profile_dir/DevToolsActivePort\" | sed 's/^/HUMANISH_BROWSER_CDP_PORT=/'",
|
|
851
|
-
" break",
|
|
852
|
-
" fi",
|
|
853
|
-
" sleep 0.1",
|
|
854
|
-
" done",
|
|
855
|
-
" fi",
|
|
856
|
-
" return 0",
|
|
857
|
-
" fi",
|
|
858
|
-
" return 1",
|
|
859
|
-
"}",
|
|
860
|
-
// Fixed CDP port (not :0/random): each seat has its OWN desktop sandbox, so a known port
|
|
861
|
-
// cannot conflict, and it makes the observer's port resolution deterministic. With :0 the
|
|
862
|
-
// real port lives only in DevToolsActivePort; when the launch-time capture misses on a cold
|
|
863
|
-
// start the observer falls back to 9222 and — being wrong — every CDP read fails for the
|
|
864
|
-
// whole run (the lobby-code handoff then never sees the host's /lobby URL). 9222 is already
|
|
865
|
-
// the fallback, so making it the actual port aligns launch, capture, and fallback.
|
|
866
|
-
`chrome_debug_flags=(--remote-debugging-address=127.0.0.1 --remote-debugging-port=9222 ${chromiumFlags})`,
|
|
867
|
-
"open_target() {",
|
|
868
|
-
" case \"$browser_preference\" in",
|
|
869
|
-
" chrome)",
|
|
870
|
-
" prepare_chrome_profile",
|
|
871
|
-
" launch_browser google-chrome google-chrome --new-window \"--user-data-dir=$chrome_profile_dir\" \"${chrome_debug_flags[@]}\" && return 0",
|
|
872
|
-
" launch_browser google-chrome-stable google-chrome-stable --new-window \"--user-data-dir=$chrome_profile_dir\" \"${chrome_debug_flags[@]}\" && return 0",
|
|
873
|
-
" echo 'requested browser chrome was not found' >&2",
|
|
874
|
-
" return 127",
|
|
875
|
-
" ;;",
|
|
876
|
-
" chromium)",
|
|
877
|
-
" prepare_chrome_profile",
|
|
878
|
-
" launch_browser chromium chromium --new-window \"--user-data-dir=$chrome_profile_dir\" \"${chrome_debug_flags[@]}\" && return 0",
|
|
879
|
-
" launch_browser chromium-browser chromium-browser --new-window \"--user-data-dir=$chrome_profile_dir\" \"${chrome_debug_flags[@]}\" && return 0",
|
|
880
|
-
" echo 'requested browser chromium was not found' >&2",
|
|
881
|
-
" return 127",
|
|
882
|
-
" ;;",
|
|
883
|
-
" firefox)",
|
|
884
|
-
" prepare_chrome_profile",
|
|
885
|
-
" launch_browser firefox firefox --new-instance --no-remote --new-window --profile \"$chrome_profile_dir\" && return 0",
|
|
886
|
-
" echo 'requested browser firefox was not found' >&2",
|
|
887
|
-
" return 127",
|
|
888
|
-
" ;;",
|
|
889
|
-
" default)",
|
|
890
|
-
" prepare_chrome_profile",
|
|
891
|
-
" launch_browser google-chrome google-chrome --new-window \"--user-data-dir=$chrome_profile_dir\" \"${chrome_debug_flags[@]}\" && return 0",
|
|
892
|
-
" launch_browser google-chrome-stable google-chrome-stable --new-window \"--user-data-dir=$chrome_profile_dir\" \"${chrome_debug_flags[@]}\" && return 0",
|
|
893
|
-
" launch_browser chromium chromium --new-window \"--user-data-dir=$chrome_profile_dir\" \"${chrome_debug_flags[@]}\" && return 0",
|
|
894
|
-
" launch_browser chromium-browser chromium-browser --new-window \"--user-data-dir=$chrome_profile_dir\" \"${chrome_debug_flags[@]}\" && return 0",
|
|
895
|
-
" launch_browser firefox firefox --new-instance --no-remote --new-window --profile \"$chrome_profile_dir\" && return 0",
|
|
896
|
-
" launch_browser xdg-open xdg-open && return 0",
|
|
897
|
-
" echo 'no browser opener found' >&2",
|
|
898
|
-
" return 127",
|
|
899
|
-
" ;;",
|
|
900
|
-
" esac",
|
|
901
|
-
"}",
|
|
902
|
-
"open_target"
|
|
903
|
-
].join("\n");
|
|
904
|
-
const result = await runDesktopCommandOrThrow(() => desktop.commands.run(browserLaunchCommand, {
|
|
905
|
-
requestTimeoutMs,
|
|
906
|
-
timeoutMs: 15_000,
|
|
907
|
-
}), ({ exitCode, stderrTail }) => new Error(`browser launch failed${exitCode === undefined ? "" : ` with exit ${exitCode}`}: ${stderrTail}`));
|
|
908
|
-
if (result.exitCode !== undefined && result.exitCode !== 0) {
|
|
909
|
-
throw new Error(`browser launch failed with exit ${result.exitCode}: ${tailOf(result.stderr ?? result.stdout ?? "")}`);
|
|
910
|
-
}
|
|
911
|
-
const resolved = (result.stdout ?? "").match(/^HUMANISH_BROWSER_RESOLVED=(\S+)$/m)?.[1];
|
|
912
|
-
const processId = (result.stdout ?? "").match(/^HUMANISH_BROWSER_PID=(\d+)$/m)?.[1];
|
|
913
|
-
const profileDir = (result.stdout ?? "").match(/^HUMANISH_BROWSER_PROFILE_DIR=(\S+)$/m)?.[1];
|
|
914
|
-
const cdpPortRaw = (result.stdout ?? "").match(/^HUMANISH_BROWSER_CDP_PORT=(\d+)$/m)?.[1];
|
|
915
|
-
const cdpPort = cdpPortRaw === undefined ? undefined : Number(cdpPortRaw);
|
|
916
|
-
return {
|
|
917
|
-
family: desktopBrowserFamily(resolved ?? requestedBrowser),
|
|
918
|
-
...(processId === undefined || profileDir === undefined
|
|
919
|
-
? {}
|
|
920
|
-
: { identity: { processId, profileDir, targetUrl, ...(cdpPort === undefined ? {} : { cdpPort }) } }),
|
|
921
|
-
...(browserPreference === undefined
|
|
922
|
-
? {}
|
|
923
|
-
: { evidence: { requested: requestedBrowser, ...(resolved === undefined ? {} : { resolved }) } })
|
|
924
|
-
};
|
|
925
|
-
}
|
|
926
|
-
if (browserPreference === undefined || browserPreference === "default") {
|
|
927
|
-
if (desktop.open) {
|
|
928
|
-
await desktop.open(targetUrl);
|
|
929
|
-
}
|
|
930
|
-
else {
|
|
931
|
-
await desktop.launch("google-chrome", targetUrl);
|
|
932
|
-
}
|
|
933
|
-
return {
|
|
934
|
-
family: desktop.open ? "unknown" : "chromium",
|
|
935
|
-
...(browserPreference === undefined ? {} : { evidence: { requested: requestedBrowser } })
|
|
936
|
-
};
|
|
937
|
-
}
|
|
938
|
-
const launchTarget = requestedBrowser === "chrome" ? "google-chrome"
|
|
939
|
-
: requestedBrowser === "chromium" ? "chromium"
|
|
940
|
-
: requestedBrowser === "firefox" ? "firefox"
|
|
941
|
-
: "google-chrome";
|
|
942
|
-
await desktop.launch(launchTarget, targetUrl);
|
|
943
|
-
return {
|
|
944
|
-
family: desktopBrowserFamily(launchTarget),
|
|
945
|
-
evidence: { requested: requestedBrowser, resolved: launchTarget }
|
|
946
|
-
};
|
|
947
|
-
}
|
|
948
|
-
export function desktopBrowserFamily(value) {
|
|
949
|
-
if (value === "firefox")
|
|
950
|
-
return "firefox";
|
|
951
|
-
if (value === "chrome" || value === "chromium" || value === "google-chrome" || value === "google-chrome-stable" || value === "chromium-browser") {
|
|
952
|
-
return "chromium";
|
|
953
|
-
}
|
|
954
|
-
return "unknown";
|
|
955
|
-
}
|
|
956
|
-
/**
|
|
957
|
-
* The URL / title / page-text / scroll observer behind stopWhen and task criteria. One probe per
|
|
958
|
-
* observation, run on the sandbox's python3 (see chrome-cdp-probe.ts for why not node: #514).
|
|
959
|
-
*
|
|
960
|
-
* "active": follow the participant to whatever tab they are driving now — never pin the state
|
|
961
|
-
* observer to the launch tab (a verification link that opened in a NEW tab left a pinned observer
|
|
962
|
-
* reading the old tab forever).
|
|
963
|
-
*
|
|
964
|
-
* `onUnavailable` fires ONCE, on the first probe that could not read the page, with the reason.
|
|
965
|
-
* The observer still degrades to `{}` for the loop; the callback is how a lane says out loud that
|
|
966
|
-
* url/text criteria are not being measured, instead of letting the funnel report 0/N (#514).
|
|
967
|
-
*/
|
|
968
|
-
export function makeChromeBrowserStateObserver(desktop, requestTimeoutMs, endpoint, targetId, onUnavailable,
|
|
969
|
-
/**
|
|
970
|
-
* Mobile emulation on later tabs (#623): the holder attaches to every page target Chrome opens
|
|
971
|
-
* after the launch page, so a tab the participant opens later should lay out at the phone width
|
|
972
|
-
* too. The first observation on each new target reads that page's OWN report; a target that
|
|
973
|
-
* reports the requested width is recorded through `onCovered`, and one that does not (or cannot
|
|
974
|
-
* be read) fires `onDrift` once, so a phone-labelled lane that spent part of its session at
|
|
975
|
-
* desktop layout says so with the number the page gave.
|
|
976
|
-
*/
|
|
977
|
-
drift) {
|
|
978
|
-
let reported = false;
|
|
979
|
-
let drifted = false;
|
|
980
|
-
const checkedTargets = new Set(drift === undefined ? [] : [drift.emulatedTargetId]);
|
|
981
|
-
const unavailable = (reason) => {
|
|
982
|
-
if (!reported) {
|
|
983
|
-
reported = true;
|
|
984
|
-
onUnavailable?.(reason);
|
|
985
|
-
}
|
|
986
|
-
return {};
|
|
987
|
-
};
|
|
988
|
-
const checkLaterTarget = async (newTargetId) => {
|
|
989
|
-
if (drift === undefined || checkedTargets.has(newTargetId))
|
|
990
|
-
return;
|
|
991
|
-
checkedTargets.add(newTargetId);
|
|
992
|
-
const read = await desktop.commands.run(chromeCdpProbeCommand({ ...endpoint, targetId: newTargetId, prefer: "pinned", mode: "fidelity" }), { requestTimeoutMs, timeoutMs: 5_000 });
|
|
993
|
-
const fidelity = read.exitCode !== undefined && read.exitCode !== 0 ? undefined : parseChromeCdpProbeOutput(read.stdout).fidelity;
|
|
994
|
-
if (fidelity !== undefined && fidelity.innerWidth === drift.expectedWidth) {
|
|
995
|
-
drift.onCovered?.(newTargetId, { innerWidth: fidelity.innerWidth, devicePixelRatio: fidelity.devicePixelRatio, maxTouchPoints: fidelity.maxTouchPoints });
|
|
996
|
-
if (drift.expectTouch === true && fidelity.maxTouchPoints === 0 && !drifted) {
|
|
997
|
-
// The viewport followed; touch did not (yet): the holder reloads a later tab once after its
|
|
998
|
-
// first navigation commits, and this observation may have landed before that reload.
|
|
999
|
-
drifted = true;
|
|
1000
|
-
drift.onDrift(`a later page target reports the ${fidelity.innerWidth} px viewport but navigator.maxTouchPoints 0 on its first observation; touch reaches a document only when it loads under the override`);
|
|
1001
|
-
}
|
|
1002
|
-
return;
|
|
1003
|
-
}
|
|
1004
|
-
if (drifted)
|
|
1005
|
-
return;
|
|
1006
|
-
drifted = true;
|
|
1007
|
-
drift.onDrift(fidelity === undefined
|
|
1008
|
-
? "the participant drove a page target other than the emulated launch tab and that page's own read-back could not be taken; whether it laid out at the phone width is not known"
|
|
1009
|
-
: `the participant drove a page target other than the emulated launch tab and that page reports a ${fidelity.innerWidth} px viewport where ${drift.expectedWidth} px was requested (DPR ${fidelity.devicePixelRatio}); the mobile user agent and touch events are browser-wide, the viewport override was not re-applied to it`);
|
|
1010
|
-
};
|
|
1011
|
-
return async () => {
|
|
1012
|
-
const result = await desktop.commands.run(chromeCdpProbeCommand({ ...endpoint, ...(targetId === undefined ? {} : { targetId }), prefer: "active", mode: "state" }), { requestTimeoutMs, timeoutMs: 5_000 });
|
|
1013
|
-
if (result.exitCode !== undefined && result.exitCode !== 0) {
|
|
1014
|
-
return unavailable(`probe exited ${result.exitCode}: ${tailOf(result.stderr ?? result.stdout ?? "")}`);
|
|
1015
|
-
}
|
|
1016
|
-
const parsed = parseChromeCdpProbeOutput(result.stdout);
|
|
1017
|
-
if (parsed.unavailable !== undefined)
|
|
1018
|
-
return unavailable(parsed.unavailable);
|
|
1019
|
-
if (parsed.targetId !== undefined)
|
|
1020
|
-
await checkLaterTarget(parsed.targetId);
|
|
1021
|
-
return {
|
|
1022
|
-
...(parsed.url === undefined ? {} : { url: parsed.url }),
|
|
1023
|
-
...(parsed.title === undefined ? {} : { title: parsed.title }),
|
|
1024
|
-
...(parsed.text === undefined ? {} : { text: parsed.text }),
|
|
1025
|
-
...(parsed.scrollY === undefined ? {} : { scrollY: parsed.scrollY })
|
|
1026
|
-
};
|
|
1027
|
-
};
|
|
1028
|
-
}
|
|
1029
|
-
/**
|
|
1030
|
-
* Read the running browser's actual outer-window bounds and CSS layout viewport through the
|
|
1031
|
-
* already-enabled local Chrome DevTools endpoint. The returned values come from `window.*` in
|
|
1032
|
-
* the target page; requested E2B resolution is deliberately not an input to this function.
|
|
1033
|
-
* Missing channels report their reason via `onUnavailable`, so the geometry warning can name
|
|
1034
|
-
* the cause (a dead CDP endpoint, no python3) instead of only the symptom. Returns `undefined`
|
|
1035
|
-
* only when neither channel could be measured.
|
|
1036
|
-
* Outer bounds and CSS dimensions are independent channels: a background page can report zero
|
|
1037
|
-
* outer dimensions while still reporting a CSS viewport. Final captures follow the active tab;
|
|
1038
|
-
* launch captures and emulation attribution keep the pinned target.
|
|
1039
|
-
*/
|
|
1040
|
-
export function makeChromeDesktopGeometryObserver(desktop, requestTimeoutMs, endpoint, targetId, onUnavailable, prefer = "pinned") {
|
|
1041
|
-
return async () => {
|
|
1042
|
-
const result = await desktop.commands.run(chromeCdpProbeCommand({ ...endpoint, ...(targetId === undefined ? {} : { targetId }), prefer, mode: "geometry" }), { requestTimeoutMs, timeoutMs: 5_000 });
|
|
1043
|
-
if (result.exitCode !== undefined && result.exitCode !== 0) {
|
|
1044
|
-
onUnavailable?.(`probe exited ${result.exitCode}: ${tailOf(result.stderr ?? result.stdout ?? "")}`);
|
|
1045
|
-
return undefined;
|
|
1046
|
-
}
|
|
1047
|
-
const parsed = parseChromeCdpProbeOutput(result.stdout);
|
|
1048
|
-
if (parsed.unavailable !== undefined) {
|
|
1049
|
-
onUnavailable?.(parsed.unavailable);
|
|
1050
|
-
return undefined;
|
|
1051
|
-
}
|
|
1052
|
-
const browserWindow = isMeasuredRect(parsed.browserWindow) ? { ...parsed.browserWindow, source: "cdp" } : undefined;
|
|
1053
|
-
const viewport = isMeasuredViewport(parsed.viewport) ? { ...parsed.viewport, source: "cdp" } : undefined;
|
|
1054
|
-
if (browserWindow === undefined && viewport === undefined) {
|
|
1055
|
-
onUnavailable?.("the page reported no usable window or viewport dimensions");
|
|
1056
|
-
return undefined;
|
|
1057
|
-
}
|
|
1058
|
-
if (browserWindow === undefined)
|
|
1059
|
-
onUnavailable?.("the page reported no usable outer-window dimensions");
|
|
1060
|
-
if (viewport === undefined)
|
|
1061
|
-
onUnavailable?.("the page reported no usable CSS viewport dimensions");
|
|
1062
|
-
return {
|
|
1063
|
-
...(browserWindow === undefined ? {} : { browserWindow }),
|
|
1064
|
-
...(viewport === undefined ? {} : { viewport }),
|
|
1065
|
-
...(parsed.targetId === undefined ? {} : { targetId: parsed.targetId })
|
|
1066
|
-
};
|
|
1067
|
-
};
|
|
1068
|
-
}
|
|
1069
|
-
/** The user agent a mobile-emulated lane presents unless the lab sets its own. */
|
|
1070
|
-
export const DEFAULT_MOBILE_USER_AGENT = "Mozilla/5.0 (iPhone; CPU iPhone OS 17_0 like Mac OS X) AppleWebKit/605.1.15 (KHTML, like Gecko) Version/17.0 Mobile/15E148 Safari/604.1";
|
|
1071
|
-
/**
|
|
1072
|
-
* Apply mobile emulation (#221) to the lane's launch page and read back what the page reports.
|
|
1073
|
-
* Fails CLOSED: a request that cannot be applied throws, because a desktop run labelled mobile is
|
|
1074
|
-
* the over-trust this feature exists to prevent. A read-back that cannot be taken is a warning
|
|
1075
|
-
* (the emulation was applied; only the proof is missing).
|
|
1076
|
-
*/
|
|
1077
|
-
export async function applyMobileEmulation(desktop, requestTimeoutMs, endpoint, targetId, request) {
|
|
1078
|
-
const command = (mode) => chromeCdpProbeCommand({ ...endpoint, ...(targetId === undefined ? {} : { targetId }), prefer: "pinned", mode, emulation: request });
|
|
1079
|
-
const read = async () => {
|
|
1080
|
-
const result = await desktop.commands.run(command("fidelity"), { requestTimeoutMs, timeoutMs: 15_000 });
|
|
1081
|
-
if (result.exitCode !== undefined && result.exitCode !== 0) {
|
|
1082
|
-
return { unavailable: `probe exited ${result.exitCode}: ${tailOf(result.stderr ?? result.stdout ?? "")}` };
|
|
1083
|
-
}
|
|
1084
|
-
return parseChromeCdpProbeOutput(result.stdout);
|
|
1085
|
-
};
|
|
1086
|
-
// The UA / touch / DPR overrides are bound to the DevTools session that set them and lapse the
|
|
1087
|
-
// moment its socket closes (measured: only the viewport width survived a one-shot apply). So the
|
|
1088
|
-
// applier stays attached for the lane's whole life as a detached process; the sandbox teardown
|
|
1089
|
-
// ends it. Its first stdout line says what was applied.
|
|
1090
|
-
const holderName = `mobile-emulation-${Date.now().toString(36)}`;
|
|
1091
|
-
await startDetachedProcess(desktop, { name: holderName, command: command("hold"), requestTimeoutMs });
|
|
1092
|
-
let announced;
|
|
1093
|
-
for (let attempt = 0; attempt < 30 && announced === undefined; attempt += 1) {
|
|
1094
|
-
await new Promise((resolve) => setTimeout(resolve, 500));
|
|
1095
|
-
const log = await readDetachedLog(desktop, holderName, requestTimeoutMs).catch(() => "");
|
|
1096
|
-
const line = log.split("\n").find((candidate) => candidate.trim().startsWith("{"));
|
|
1097
|
-
if (line !== undefined)
|
|
1098
|
-
announced = parseChromeCdpProbeOutput(line);
|
|
1099
|
-
}
|
|
1100
|
-
if (announced === undefined) {
|
|
1101
|
-
throw new Error("mobile emulation could not be applied: the in-sandbox applier printed nothing within 15 s");
|
|
1102
|
-
}
|
|
1103
|
-
if (announced.unavailable !== undefined) {
|
|
1104
|
-
throw new Error(`mobile emulation could not be applied (${announced.unavailable}); applied before failing: ${(announced.applied ?? []).join(", ") || "nothing"}`);
|
|
1105
|
-
}
|
|
1106
|
-
const applied = announced;
|
|
1107
|
-
// Viewport/touch read-back proves context settings, not gesture equivalence. Two hosted
|
|
1108
|
-
// replicas and a native-X conversion-toggle control reproduced reset click counts (#676).
|
|
1109
|
-
const warnings = request.touch
|
|
1110
|
-
? ["Mobile emulation uses desktop pointer-to-touch conversion, which can differ for repeated taps. Confirm gesture failures with direct or native touch input before attributing them to the app."]
|
|
1111
|
-
: [];
|
|
1112
|
-
// The reload inside the applier takes a moment; the read-back is retried until the page reports
|
|
1113
|
-
// the requested viewport and user agent, so a slow page does not read as "no proof".
|
|
1114
|
-
let readBack = await read();
|
|
1115
|
-
for (let attempt = 0; attempt < 20 && (readBack.fidelity === undefined || readBack.fidelity.innerWidth !== request.width || !readBack.fidelity.userAgent.includes(request.userAgent.slice(0, 24))); attempt += 1) {
|
|
1116
|
-
await new Promise((resolve) => setTimeout(resolve, 500));
|
|
1117
|
-
readBack = await read();
|
|
1118
|
-
}
|
|
1119
|
-
const fidelityRead = readBack;
|
|
1120
|
-
const requested = {
|
|
1121
|
-
width: request.width,
|
|
1122
|
-
height: request.height,
|
|
1123
|
-
deviceScaleFactor: request.deviceScaleFactor,
|
|
1124
|
-
touch: request.touch,
|
|
1125
|
-
userAgent: request.userAgent
|
|
1126
|
-
};
|
|
1127
|
-
const emulatedTargetId = applied.targetId ?? fidelityRead.targetId;
|
|
1128
|
-
if (fidelityRead.fidelity === undefined) {
|
|
1129
|
-
warnings.push(`Mobile emulation was applied but the page's own report could not be read (${fidelityRead.unavailable ?? "no fidelity read"}); desktopGeometry.fidelity carries the request without a resolved block.`);
|
|
1130
|
-
return { fidelity: { tier: "mobile-emulated", requested, applied: applied.applied ?? [] }, warnings, holderName, ...(emulatedTargetId === undefined ? {} : { targetId: emulatedTargetId }) };
|
|
1131
|
-
}
|
|
1132
|
-
const resolved = { ...fidelityRead.fidelity, source: "cdp" };
|
|
1133
|
-
if (resolved.innerWidth !== request.width) {
|
|
1134
|
-
warnings.push(`Mobile emulation requested a ${request.width} px viewport; the page reports ${resolved.innerWidth} px.`);
|
|
1135
|
-
}
|
|
1136
|
-
if (resolved.devicePixelRatio !== request.deviceScaleFactor) {
|
|
1137
|
-
warnings.push(`Mobile emulation requested devicePixelRatio ${request.deviceScaleFactor}; the page reports ${resolved.devicePixelRatio}.`);
|
|
1138
|
-
}
|
|
1139
|
-
if (request.touch && resolved.maxTouchPoints === 0) {
|
|
1140
|
-
warnings.push("Mobile emulation requested touch; the page reports navigator.maxTouchPoints 0.");
|
|
1141
|
-
}
|
|
1142
|
-
if (!resolved.userAgent.includes("Mobile") && !resolved.userAgent.includes("Android") && !resolved.userAgent.includes("iPhone")) {
|
|
1143
|
-
warnings.push("Mobile emulation requested a mobile user agent; the page reports a desktop one.");
|
|
1144
|
-
}
|
|
1145
|
-
return { fidelity: { tier: "mobile-emulated", requested, applied: applied.applied ?? [], resolved }, warnings, holderName, ...(emulatedTargetId === undefined ? {} : { targetId: emulatedTargetId }) };
|
|
1146
|
-
}
|
|
1147
|
-
function isMeasuredRect(value) {
|
|
1148
|
-
if (!value || typeof value !== "object")
|
|
1149
|
-
return false;
|
|
1150
|
-
const record = value;
|
|
1151
|
-
return Number.isFinite(record.x)
|
|
1152
|
-
&& Number.isFinite(record.y)
|
|
1153
|
-
&& isPositiveMeasurement(record.width)
|
|
1154
|
-
&& isPositiveMeasurement(record.height);
|
|
1155
|
-
}
|
|
1156
|
-
function isMeasuredViewport(value) {
|
|
1157
|
-
if (!value || typeof value !== "object")
|
|
1158
|
-
return false;
|
|
1159
|
-
const record = value;
|
|
1160
|
-
return isPositiveMeasurement(record.width)
|
|
1161
|
-
&& isPositiveMeasurement(record.height)
|
|
1162
|
-
&& isPositiveMeasurement(record.deviceScaleFactor);
|
|
1163
|
-
}
|
|
1164
|
-
function isPositiveMeasurement(value) {
|
|
1165
|
-
return typeof value === "number" && Number.isFinite(value) && value > 0;
|
|
1166
|
-
}
|
|
1167
|
-
async function measureBrowserWindowWithXdotool(desktop, windowId, requestTimeoutMs) {
|
|
1168
|
-
const result = await desktop.commands.run([
|
|
1169
|
-
"set -euo pipefail",
|
|
1170
|
-
`win=${shellSingleQuote(windowId)}`,
|
|
1171
|
-
"xdotool getwindowgeometry --shell \"$win\" 2>/dev/null || true"
|
|
1172
|
-
].join("\n"), { requestTimeoutMs, timeoutMs: 5_000 });
|
|
1173
|
-
const output = result.stdout ?? "";
|
|
1174
|
-
const read = (name) => {
|
|
1175
|
-
const raw = output.match(new RegExp(`^${name}=(-?\\d+)$`, "m"))?.[1];
|
|
1176
|
-
if (raw === undefined)
|
|
1177
|
-
return undefined;
|
|
1178
|
-
const value = Number(raw);
|
|
1179
|
-
return Number.isFinite(value) ? value : undefined;
|
|
1180
|
-
};
|
|
1181
|
-
const x = read("X");
|
|
1182
|
-
const y = read("Y");
|
|
1183
|
-
const width = read("WIDTH");
|
|
1184
|
-
const height = read("HEIGHT");
|
|
1185
|
-
if (x === undefined || y === undefined || width === undefined || height === undefined || width <= 0 || height <= 0) {
|
|
1186
|
-
return undefined;
|
|
1187
|
-
}
|
|
1188
|
-
return { x, y, width, height, source: "xdotool" };
|
|
1189
|
-
}
|
|
1190
|
-
/** Physical X client bounds, never the page's emulated window.outerWidth/Height. */
|
|
1191
|
-
function isBrowserWindowContained(bounds, [width, height]) {
|
|
1192
|
-
return bounds.x >= 0 && bounds.y >= 0
|
|
1193
|
-
&& bounds.x + bounds.width <= width && bounds.y + bounds.height <= height;
|
|
1194
|
-
}
|
|
1195
|
-
/** One bounded repair. Window-manager decorations may keep the client origin below (0, 0),
|
|
1196
|
-
* so a full-screen client height can clip the bottom even after windowmove succeeds. */
|
|
1197
|
-
async function fitBrowserWindowWithinDesktop(desktop, windowId, resolution, requestTimeoutMs) {
|
|
1198
|
-
const run = (command) => desktop.commands.run([
|
|
1199
|
-
"set -euo pipefail",
|
|
1200
|
-
`win=${shellSingleQuote(windowId)}`,
|
|
1201
|
-
command
|
|
1202
|
-
].join("\n"), { requestTimeoutMs, timeoutMs: 5_000 }).catch(() => undefined);
|
|
1203
|
-
await run('xdotool windowmove "$win" 0 0');
|
|
1204
|
-
await desktop.wait(250).catch(() => undefined);
|
|
1205
|
-
const moved = await measureBrowserWindowWithXdotool(desktop, windowId, requestTimeoutMs).catch(() => undefined);
|
|
1206
|
-
if (moved === undefined)
|
|
1207
|
-
return moved;
|
|
1208
|
-
let resized = moved;
|
|
1209
|
-
const width = resolution[0] - moved.x;
|
|
1210
|
-
const height = resolution[1] - moved.y;
|
|
1211
|
-
// Resizing alone cannot fix an offscreen client origin. The window manager
|
|
1212
|
-
// can also center a minimum-width client at a negative x on a narrow screen.
|
|
1213
|
-
if (moved.x >= 0 && moved.y >= 0 && width > 0 && height > 0) {
|
|
1214
|
-
await run(`xdotool windowsize "$win" ${width} ${height}`);
|
|
1215
|
-
await desktop.wait(250).catch(() => undefined);
|
|
1216
|
-
const measured = await measureBrowserWindowWithXdotool(desktop, windowId, requestTimeoutMs).catch(() => undefined);
|
|
1217
|
-
if (measured === undefined)
|
|
1218
|
-
return measured;
|
|
1219
|
-
resized = measured;
|
|
1220
|
-
}
|
|
1221
|
-
if (resized === undefined || isBrowserWindowContained(resized, resolution))
|
|
1222
|
-
return resized;
|
|
1223
|
-
// Chrome's minimum client width can equal the whole desktop. Window-manager
|
|
1224
|
-
// borders then make a decorated window impossible to contain, even after a
|
|
1225
|
-
// successful move/resize. Request fullscreen once and prove the physical result.
|
|
1226
|
-
// xprop/xdotool ship with the desktop template; wmctrl is not required.
|
|
1227
|
-
// Check state first so the fullscreen shortcut cannot toggle an existing state off.
|
|
1228
|
-
await run([
|
|
1229
|
-
'state=$(xprop -id "$win" _NET_WM_STATE)',
|
|
1230
|
-
'case "$state" in',
|
|
1231
|
-
' *_NET_WM_STATE_FULLSCREEN*) ;;',
|
|
1232
|
-
' *) xdotool windowactivate --sync "$win"; xdotool key --clearmodifiers F11 ;;',
|
|
1233
|
-
'esac'
|
|
1234
|
-
].join("\n"));
|
|
1235
|
-
// The fullscreen animation may report its new origin before its final width.
|
|
1236
|
-
// Give the window manager a bounded settling window, keeping missing reads unverified.
|
|
1237
|
-
for (let attempt = 0; attempt < 4; attempt += 1) {
|
|
1238
|
-
await desktop.wait(250).catch(() => undefined);
|
|
1239
|
-
const measured = await measureBrowserWindowWithXdotool(desktop, windowId, requestTimeoutMs).catch(() => undefined);
|
|
1240
|
-
if (measured === undefined || isBrowserWindowContained(measured, resolution))
|
|
1241
|
-
return measured;
|
|
1242
|
-
resized = measured;
|
|
1243
|
-
}
|
|
1244
|
-
return resized;
|
|
1245
|
-
}
|
|
1246
|
-
/** Shared hosted-browser geometry capture used by per-lane and sequential shared-world routes. */
|
|
1247
|
-
export async function captureDesktopBrowserGeometry(args) {
|
|
1248
|
-
const warnings = [];
|
|
1249
|
-
let browserWindowId = args.browserWindowId;
|
|
1250
|
-
if (browserWindowId === undefined && args.browserFamily !== "unknown") {
|
|
1251
|
-
browserWindowId = await findVisibleBrowserWindowId(args.desktop, args.requestTimeoutMs, args.browserFamily, args.launchIdentity).catch((error) => {
|
|
1252
|
-
warnings.push(`Browser window lookup failed for lane ${args.laneId}: ${redactText(toErrorMessage(error))}`);
|
|
1253
|
-
return undefined;
|
|
1254
|
-
});
|
|
1255
|
-
}
|
|
1256
|
-
let xdotoolWindow;
|
|
1257
|
-
if (browserWindowId !== undefined) {
|
|
1258
|
-
if (args.resize !== false) {
|
|
1259
|
-
await fillDesktopBrowserWindow(args.desktop, browserWindowId, args.requestedScreen, args.requestTimeoutMs);
|
|
1260
|
-
// Let the window manager apply the resize before querying both X and page layout geometry.
|
|
1261
|
-
await args.desktop.wait(250).catch(() => undefined);
|
|
1262
|
-
}
|
|
1263
|
-
xdotoolWindow = await measureBrowserWindowWithXdotool(args.desktop, browserWindowId, args.requestTimeoutMs)
|
|
1264
|
-
.catch(() => undefined);
|
|
1265
|
-
}
|
|
1266
|
-
else {
|
|
1267
|
-
warnings.push(`Browser window bounds could not be measured for lane ${args.laneId}; the live stream will use the full desktop.`);
|
|
1268
|
-
}
|
|
1269
|
-
let unusable;
|
|
1270
|
-
if (xdotoolWindow !== undefined && !isBrowserWindowContained(xdotoolWindow, args.requestedScreen)) {
|
|
1271
|
-
const before = xdotoolWindow;
|
|
1272
|
-
if (args.resize !== false && browserWindowId !== undefined) {
|
|
1273
|
-
xdotoolWindow = await fitBrowserWindowWithinDesktop(args.desktop, browserWindowId, args.requestedScreen, args.requestTimeoutMs);
|
|
1274
|
-
if (xdotoolWindow === undefined) {
|
|
1275
|
-
// Keep the last measured bad state; a missing observation cannot prove a successful fix.
|
|
1276
|
-
xdotoolWindow = before;
|
|
1277
|
-
unusable = `Physical browser containment could not be verified after correction for lane ${args.laneId}; the last measured window was clipped.`;
|
|
1278
|
-
}
|
|
1279
|
-
else if (isBrowserWindowContained(xdotoolWindow, args.requestedScreen)) {
|
|
1280
|
-
warnings.push(`Browser window clipping corrected for lane ${args.laneId}; physical bounds are ${xdotoolWindow.width}x${xdotoolWindow.height} at (${xdotoolWindow.x}, ${xdotoolWindow.y}).`);
|
|
1281
|
-
}
|
|
1282
|
-
}
|
|
1283
|
-
if (unusable === undefined && !isBrowserWindowContained(xdotoolWindow, args.requestedScreen)) {
|
|
1284
|
-
unusable = `Browser window is outside the captured ${args.requestedScreen[0]}x${args.requestedScreen[1]} desktop for lane ${args.laneId}: physical bounds ${xdotoolWindow.width}x${xdotoolWindow.height} at (${xdotoolWindow.x}, ${xdotoolWindow.y}), right=${xdotoolWindow.x + xdotoolWindow.width}, bottom=${xdotoolWindow.y + xdotoolWindow.height}.`;
|
|
1285
|
-
}
|
|
1286
|
-
if (unusable !== undefined)
|
|
1287
|
-
warnings.push(unusable);
|
|
1288
|
-
}
|
|
1289
|
-
if (xdotoolWindow === undefined) {
|
|
1290
|
-
warnings.push(`Physical browser containment is unverified for lane ${args.laneId}; X window bounds could not be measured. Page-reported outer dimensions can be emulated and do not prove physical visibility.`);
|
|
1291
|
-
}
|
|
1292
|
-
let cdpUnavailable;
|
|
1293
|
-
const chromeGeometry = args.browserFamily === "chromium"
|
|
1294
|
-
? await makeChromeDesktopGeometryObserver(args.desktop, args.requestTimeoutMs, {
|
|
1295
|
-
...(args.launchIdentity?.cdpPort === undefined ? {} : { cdpPort: args.launchIdentity.cdpPort }),
|
|
1296
|
-
...(args.launchIdentity?.profileDir === undefined ? {} : { profileDir: args.launchIdentity.profileDir }),
|
|
1297
|
-
targetUrl: args.targetUrl
|
|
1298
|
-
}, args.browserTargetId, (reason) => {
|
|
1299
|
-
cdpUnavailable = reason;
|
|
1300
|
-
}, args.pagePreference ?? "pinned")().catch((error) => {
|
|
1301
|
-
cdpUnavailable = toErrorMessage(error);
|
|
1302
|
-
return undefined;
|
|
1303
|
-
})
|
|
1304
|
-
: undefined;
|
|
1305
|
-
const browserWindow = xdotoolWindow ?? chromeGeometry?.browserWindow;
|
|
1306
|
-
const viewport = chromeGeometry?.viewport;
|
|
1307
|
-
// The fill check reads the X window when it was measured: under mobile emulation (#221) the
|
|
1308
|
-
// page's window.outerWidth reports the EMULATED screen (414), which is not a fill failure.
|
|
1309
|
-
const fillBounds = xdotoolWindow;
|
|
1310
|
-
if (!browserWindow) {
|
|
1311
|
-
warnings.push(`Browser outer bounds could not be measured for lane ${args.laneId}.`);
|
|
1312
|
-
}
|
|
1313
|
-
else if (unusable === undefined && fillBounds !== undefined && (fillBounds.x !== 0 || fillBounds.y !== 0 || fillBounds.width !== args.requestedScreen[0] || fillBounds.height !== args.requestedScreen[1])) {
|
|
1314
|
-
warnings.push(`Browser window fill did not reach the requested ${args.requestedScreen[0]}x${args.requestedScreen[1]} screen for lane ${args.laneId}; measured physical bounds are ${fillBounds.width}x${fillBounds.height} at (${fillBounds.x}, ${fillBounds.y}).`);
|
|
1315
|
-
}
|
|
1316
|
-
if (!viewport) {
|
|
1317
|
-
// Name the cause, not only the symptom: the same dead DevTools channel that loses the viewport
|
|
1318
|
-
// loses every url/text observation, and a reader of the bundle should learn that here (#514).
|
|
1319
|
-
const cause = cdpUnavailable === undefined ? "" : ` DevTools probe: ${redactText(cdpUnavailable)}.`;
|
|
1320
|
-
warnings.push(args.browserFamily === "firefox"
|
|
1321
|
-
? `Browser CSS viewport measurement is unavailable for Firefox on lane ${args.laneId}; stream.viewport is omitted instead of reading a different browser's CDP endpoint.`
|
|
1322
|
-
: `Browser CSS viewport could not be measured for lane ${args.laneId}; stream.viewport is omitted instead of copying the requested screen resolution.${cause}`);
|
|
1323
|
-
}
|
|
1324
|
-
return {
|
|
1325
|
-
...(unusable === undefined ? {} : { unusable }),
|
|
1326
|
-
...(browserWindowId === undefined ? {} : { browserWindowId }),
|
|
1327
|
-
...(chromeGeometry?.targetId === undefined ? {} : { browserTargetId: chromeGeometry.targetId }),
|
|
1328
|
-
...(browserWindow === undefined ? {} : { browserWindow }),
|
|
1329
|
-
...(viewport === undefined ? {} : { viewport }),
|
|
1330
|
-
warnings
|
|
1331
|
-
};
|
|
1332
|
-
}
|
|
1333
|
-
function shellSingleQuote(value) {
|
|
1334
|
-
return `'${value.replace(/'/g, "'\\''")}'`;
|
|
1335
|
-
}
|
|
1336
|
-
/**
|
|
1337
|
-
* Prepare a CLI study's runtime and, only when declared, its product (#495, #515).
|
|
1338
|
-
*
|
|
1339
|
-
* The install runs UNKEYED and before the session starts, for the same reason the clone route
|
|
1340
|
-
* provisions its subject first: what is being studied begins when the participant looks at the
|
|
1341
|
-
* screen. Omitting install deliberately studies product installation; Node/npm remain a
|
|
1342
|
-
* harness prerequisite so the participant can follow the product's public npm instructions.
|
|
1343
|
-
*/
|
|
1344
|
-
async function provisionDesktopCli(desktop, args) {
|
|
1345
|
-
const install = args.install;
|
|
1346
|
-
const now = () => Date.now();
|
|
1347
|
-
if (install === undefined || needsNodeRuntime([install])) {
|
|
1348
|
-
const startedAt = now();
|
|
1349
|
-
emitPhaseStarted(args.onPhase, now, "runtime", "providing Node/npm for the desktop CLI study");
|
|
1350
|
-
const bootstrap = await runDetachedStep(desktop, {
|
|
1351
|
-
name: "desktop-cli-runtime-node",
|
|
1352
|
-
command: TERMINAL_NODE_BOOTSTRAP_COMMAND,
|
|
1353
|
-
cwd: "/home/user",
|
|
1354
|
-
timeoutMs: INSTALL_TIMEOUT_MS,
|
|
1355
|
-
requestTimeoutMs: args.requestTimeoutMs
|
|
1356
|
-
});
|
|
1357
|
-
emitPhaseCompleted(args.onPhase, now, startedAt, "runtime", bootstrap.ok, bootstrap.ok
|
|
1358
|
-
? "Node runtime ready"
|
|
1359
|
-
: "Node runtime bootstrap failed");
|
|
1360
|
-
if (!bootstrap.ok) {
|
|
1361
|
-
throw new Error(`desktop-cli runtime bootstrap failed for "${args.product}"`);
|
|
1362
|
-
}
|
|
1363
|
-
}
|
|
1364
|
-
if (install === undefined)
|
|
1365
|
-
return;
|
|
1366
|
-
const startedAt = now();
|
|
1367
|
-
emitPhaseStarted(args.onPhase, now, "install", `installing ${args.product} on the desktop`);
|
|
1368
|
-
const result = await runDetachedStep(desktop, {
|
|
1369
|
-
name: "desktop-cli-install",
|
|
1370
|
-
command: install,
|
|
1371
|
-
cwd: "/home/user",
|
|
1372
|
-
timeoutMs: INSTALL_TIMEOUT_MS,
|
|
1373
|
-
requestTimeoutMs: args.requestTimeoutMs
|
|
1374
|
-
});
|
|
1375
|
-
emitPhaseCompleted(args.onPhase, now, startedAt, "install", result.ok, result.ok
|
|
1376
|
-
? `${args.product} installed`
|
|
1377
|
-
: `installing ${args.product} failed`);
|
|
1378
|
-
if (!result.ok) {
|
|
1379
|
-
// Fail closed: a participant handed a desktop where the product is not installed would produce
|
|
1380
|
-
// a transcript about a missing command, and that finding belongs to the harness, not the tool.
|
|
1381
|
-
// The tail rides along, scrubbed before truncation like every other provisioning failure — a
|
|
1382
|
-
// bare "install failed" is unactionable to whoever wrote the command.
|
|
1383
|
-
throw new Error(args.scrub(`desktop-cli install failed for "${args.product}" (${result.timedOut ? "timed out" : `exit ${result.exitCode ?? "?"}`}): ${tailOf(args.scrub(result.logTail))}`));
|
|
1384
|
-
}
|
|
1385
|
-
}
|
|
1386
|
-
/**
|
|
1387
|
-
* Open a terminal window on the desktop.
|
|
1388
|
-
*
|
|
1389
|
-
* The stock template is XFCE and ships xfce4-terminal (also aliased x-terminal-emulator), verified
|
|
1390
|
-
* live before this route was built. `x-terminal-emulator` is tried first so a template that swaps
|
|
1391
|
-
* the emulator still works; a desktop with neither is a template problem and fails closed rather
|
|
1392
|
-
* than handing a participant an empty screen and calling it a study.
|
|
1393
|
-
*/
|
|
1394
|
-
async function openDesktopTerminal(desktop, requestTimeoutMs, workdir) {
|
|
1395
|
-
const dir = workdir ?? "/home/user";
|
|
1396
|
-
const result = await runDetachedStep(desktop, {
|
|
1397
|
-
name: "desktop-cli-terminal",
|
|
1398
|
-
command: [
|
|
1399
|
-
"for candidate in x-terminal-emulator xfce4-terminal gnome-terminal konsole xterm; do",
|
|
1400
|
-
' if command -v "$candidate" >/dev/null 2>&1; then',
|
|
1401
|
-
// LANG is set on the terminal we open, not globally: the stock image declares no locale, and
|
|
1402
|
-
// a study that measures our own mojibake against an unconfigured template would be measuring
|
|
1403
|
-
// the template. The PRODUCT-side fix (an ASCII fallback when the locale is not UTF-8) is in
|
|
1404
|
-
// src/terminal-encoding.ts, and it is the one that matters for real users.
|
|
1405
|
-
` (cd ${shellSingleQuote(dir)} 2>/dev/null || cd /home/user; DISPLAY=:0 LANG=C.UTF-8 LC_ALL=C.UTF-8 HUMANISH_STUDY_PARTICIPANT=1 nohup "$candidate" >/dev/null 2>&1 &)`,
|
|
1406
|
-
" sleep 3",
|
|
1407
|
-
' echo "humanish: opened $candidate"',
|
|
1408
|
-
" exit 0",
|
|
1409
|
-
" fi",
|
|
1410
|
-
"done",
|
|
1411
|
-
"echo 'humanish: no terminal emulator on this desktop template' >&2",
|
|
1412
|
-
"exit 1"
|
|
1413
|
-
].join("\n"),
|
|
1414
|
-
cwd: "/home/user",
|
|
1415
|
-
timeoutMs: 60_000,
|
|
1416
|
-
requestTimeoutMs
|
|
1417
|
-
});
|
|
1418
|
-
if (!result.ok) {
|
|
1419
|
-
throw new Error("desktop-cli lane could not open a terminal on this desktop template");
|
|
1420
|
-
}
|
|
1421
|
-
}
|
|
1422
|
-
async function startDesktopStream(desktop, browserWindowId) {
|
|
1423
|
-
if (!browserWindowId) {
|
|
1424
|
-
await desktop.stream.start({ requireAuth: true });
|
|
1425
|
-
return;
|
|
1426
|
-
}
|
|
1427
|
-
try {
|
|
1428
|
-
await desktop.stream.start({ requireAuth: true, windowId: browserWindowId });
|
|
1429
|
-
}
|
|
1430
|
-
catch {
|
|
1431
|
-
await desktop.stream.start({ requireAuth: true });
|
|
1432
|
-
}
|
|
1433
|
-
}
|
|
1434
541
|
// "can't" followed by a PERCEPTION verb describes what the screen showed, not an inability to
|
|
1435
542
|
// proceed: "the canvas truncates it so you can't even read the whole thing", "I can't tell from
|
|
1436
543
|
// the screen whether the rename is persisted", "so I could not read its full description". Five
|
|
@@ -1640,118 +747,18 @@ export function resolveSelfReportedFriction(session) {
|
|
|
1640
747
|
return reports.join("\n\n");
|
|
1641
748
|
return undefined;
|
|
1642
749
|
}
|
|
1643
|
-
/**
|
|
1644
|
-
*
|
|
1645
|
-
* resolution), prepareDesktop, verify geometry, (clone+serve+seed the subject per lane), open the
|
|
1646
|
-
* browser, run the session, and ALWAYS tear down THIS lane's sandbox BY ID in a finally. Never
|
|
1647
|
-
* enumerates sandboxes. Extracted from the former single-lane block; at N=1 it writes the exact
|
|
1648
|
-
* same artifacts (actor.json, screenshots/<name>) the bundle has always referenced.
|
|
1649
|
-
*/
|
|
750
|
+
/** Run one participant against a prepared desktop. The adapter owns provisioning, final
|
|
751
|
+
* evidence and cleanup; this runner owns the model, trace and participant outcome. */
|
|
1650
752
|
export async function runCuaLane(spec, deps) {
|
|
1651
|
-
const { config,
|
|
1652
|
-
const desktopCliRoute = deps.desktopCliRoute === true;
|
|
1653
|
-
// The local brain, when there is one. `appServer` / `claudeSession` own a process, so the lane
|
|
1654
|
-
// closes it.
|
|
753
|
+
const { config, env } = deps;
|
|
1655
754
|
let appServer;
|
|
1656
755
|
let claudeSession;
|
|
1657
756
|
let localAgentProvider;
|
|
1658
|
-
const subjectEnvValues = config.subject.envValues ?? {};
|
|
1659
|
-
const targetUrl = spec.targetUrl ?? appUrl;
|
|
1660
|
-
const env = deps.env;
|
|
1661
|
-
// Off-app comms (#297): on an in-sandbox subject route, redirect the app's email-API sends into an
|
|
1662
|
-
// in-sandbox catch (loopback) so its verification mail is CAPTURED, not sent to the internet. Gated
|
|
1663
|
-
// ENTIRELY on config.comms — no comms declared → zero change. The base-URL env is injected at
|
|
1664
|
-
// sandbox-create (below, so the app reads it at boot); the catch is started right after create.
|
|
1665
|
-
const commsEmail = (cloneRoute || localTreeRoute) && config.comms?.email?.kind === "fake" ? config.comms.email : undefined;
|
|
1666
|
-
const commsPort = commsEmail ? (commsEmail.port ?? DEFAULT_SANDBOX_CATCH_PORT) : undefined;
|
|
1667
|
-
// Hoisted so the finally can drain the catch before teardown; `commsArtifactPath` is the written
|
|
1668
|
-
// evidence path folded into the lane outcome.
|
|
1669
|
-
let deployedComms;
|
|
1670
|
-
let commsArtifactPath;
|
|
1671
|
-
let receivingInboxUrl;
|
|
1672
|
-
// injectEnv is absent on an adopter-hosted plane (#328): there is no subject env to inject
|
|
1673
|
-
// because the operator points their own app at their own catch.
|
|
1674
|
-
const commsEnv = commsEmail?.injectEnv !== undefined && commsPort !== undefined
|
|
1675
|
-
? { [commsEmail.injectEnv]: `http://127.0.0.1:${commsPort}` }
|
|
1676
|
-
: {};
|
|
1677
|
-
// SMTP transport: the same idea as injectEnv, but an app that speaks SMTP needs a host and a port
|
|
1678
|
-
// rather than a base URL. The catch accepts any credentials (loopback only), yet many apps refuse
|
|
1679
|
-
// to boot unless the user/password vars exist at all, so those are injected when declared.
|
|
1680
|
-
const commsSmtpPort = commsEmail?.smtp?.port;
|
|
1681
|
-
if (commsEmail?.smtp && commsSmtpPort !== undefined) {
|
|
1682
|
-
commsEnv[commsEmail.smtp.hostEnv] = "127.0.0.1";
|
|
1683
|
-
commsEnv[commsEmail.smtp.portEnv] = String(commsSmtpPort);
|
|
1684
|
-
if (commsEmail.smtp.userEnv)
|
|
1685
|
-
commsEnv[commsEmail.smtp.userEnv] = commsEmail.smtp.user ?? "humanish";
|
|
1686
|
-
if (commsEmail.smtp.passwordEnv)
|
|
1687
|
-
commsEnv[commsEmail.smtp.passwordEnv] = commsEmail.smtp.password ?? "humanish";
|
|
1688
|
-
}
|
|
1689
|
-
// Persona inbox SURFACE (#297 slice B): the loopback URL the persona opens to read captured mail; the
|
|
1690
|
-
// origin-rewrite map (identity on this same-sandbox route, but covers localhost/0.0.0.0 alias skew + an
|
|
1691
|
-
// operator-declared linkOrigin); and a disposable background loop that renders the surface DURING the
|
|
1692
|
-
// session so the inbox is live when the persona checks. The surface uses its OWN FakeInbox + cursor,
|
|
1693
|
-
// independent of the teardown evidence drain (two readers of the append-only NDJSON — no double-count).
|
|
1694
|
-
const commsInboxUrl = commsEmail && commsPort !== undefined ? `http://127.0.0.1:${commsPort}/inbox` : undefined;
|
|
1695
|
-
const commsOriginMap = commsEmail
|
|
1696
|
-
? buildOriginMap({
|
|
1697
|
-
...(config.subject.serve?.url === undefined ? {} : { internalServeUrl: config.subject.serve.url }),
|
|
1698
|
-
reachableBaseUrl: targetUrl,
|
|
1699
|
-
...(commsEmail.linkOrigin === undefined ? {} : { linkOrigin: commsEmail.linkOrigin })
|
|
1700
|
-
})
|
|
1701
|
-
: [];
|
|
1702
|
-
const surfaceRecipients = (commsEmail?.recipients ?? [])
|
|
1703
|
-
.filter((recipient) => recipient.address !== undefined)
|
|
1704
|
-
.map((recipient) => ({ lane: recipient.lane, address: recipient.address }));
|
|
1705
|
-
let surfaceRenderedCount = 0;
|
|
1706
|
-
let surfaceDisposed = false;
|
|
1707
|
-
let releaseSurface = () => { };
|
|
1708
|
-
const surfaceDispose = new Promise((resolve) => { releaseSurface = resolve; });
|
|
1709
|
-
let surfaceLoop;
|
|
1710
757
|
const warnings = [];
|
|
1711
758
|
const screenshots = [];
|
|
1712
759
|
const writeScreenshot = makeLaneWriteScreenshot(deps.artifactRoot, spec, screenshots);
|
|
1713
|
-
const stateStepRecords = [];
|
|
1714
|
-
// Completed-only trail (durationMs/ok are set on completed events, never on started ones):
|
|
1715
|
-
// this is what survives into bundle.events. The default/injected sink below sees EVERY event,
|
|
1716
|
-
// started and completed alike, so an operator watching stderr sees both halves of each phase.
|
|
1717
|
-
const phaseRecords = [];
|
|
1718
|
-
const onSubjectPhase = (event) => {
|
|
1719
|
-
if (event.ok !== undefined) {
|
|
1720
|
-
phaseRecords.push(event);
|
|
1721
|
-
}
|
|
1722
|
-
(deps.hooks.onPhase ?? defaultSubjectPhaseSink)(event, { laneId: spec.laneId, laneCount: deps.laneCount });
|
|
1723
|
-
};
|
|
1724
760
|
let session;
|
|
1725
761
|
let sessionError;
|
|
1726
|
-
let failureCode;
|
|
1727
|
-
let sandboxId;
|
|
1728
|
-
// Host-side E2B desktop billed-span endpoints, measured via the injected clock. Captured right
|
|
1729
|
-
// after create() succeeds and again in the finally after teardown resolves (both the killed and
|
|
1730
|
-
// kept-for-debug paths). This measured span excludes allocation before the acquired handle;
|
|
1731
|
-
// a kept/unconfirmed allocation gets an extra unknown lifetime cost line.
|
|
1732
|
-
let sandboxCreatedAtMs;
|
|
1733
|
-
let sandboxTornDownAtMs;
|
|
1734
|
-
let desktopResources;
|
|
1735
|
-
let killed = false;
|
|
1736
|
-
let streamUrl;
|
|
1737
|
-
let subjectCommit;
|
|
1738
|
-
let desktopBrowser;
|
|
1739
|
-
let launchedBrowserFamily = "unknown";
|
|
1740
|
-
let browserLaunchIdentity;
|
|
1741
|
-
let browserLaunched = false;
|
|
1742
|
-
let initialBrowserGeometry;
|
|
1743
|
-
let appliedFidelity;
|
|
1744
|
-
let emulatedTargetId;
|
|
1745
|
-
let emulationHolderName;
|
|
1746
|
-
let browserWindowId;
|
|
1747
|
-
let browserTargetId;
|
|
1748
|
-
const declaredScreen = declaredScreenForRender(spec.devicePreset, spec.deviceName, spec.resolution);
|
|
1749
|
-
let desktopGeometry = {
|
|
1750
|
-
screen: {
|
|
1751
|
-
requested: { width: spec.resolution[0], height: spec.resolution[1] },
|
|
1752
|
-
...(declaredScreen ? { declared: declaredScreen } : {})
|
|
1753
|
-
}
|
|
1754
|
-
};
|
|
1755
762
|
let provisioned = false;
|
|
1756
763
|
let signaled = false;
|
|
1757
764
|
const signal = (ok) => {
|
|
@@ -1760,635 +767,137 @@ export async function runCuaLane(spec, deps) {
|
|
|
1760
767
|
deps.signalProvisioned(ok);
|
|
1761
768
|
}
|
|
1762
769
|
};
|
|
1763
|
-
|
|
1764
|
-
let desktop;
|
|
770
|
+
const desktopLane = deps.createDesktopLane?.(spec, warnings) ?? createE2BCuaDesktopLane(spec, deps, warnings);
|
|
1765
771
|
try {
|
|
1766
|
-
|
|
1767
|
-
//
|
|
1768
|
-
// the
|
|
1769
|
-
|
|
1770
|
-
|
|
1771
|
-
|
|
1772
|
-
|
|
1773
|
-
|
|
1774
|
-
|
|
1775
|
-
labId: config.id,
|
|
1776
|
-
simId: spec.simId,
|
|
1777
|
-
laneId: spec.laneId,
|
|
1778
|
-
laneIndex: String(spec.laneIndex),
|
|
1779
|
-
laneCount: String(deps.laneCount)
|
|
1780
|
-
},
|
|
1781
|
-
// Env placement per the doctrine: the ACTOR's key never enters the sandbox (the model drives
|
|
1782
|
-
// from outside). The SUBJECT's declared env NAMES are provisioned here on the clone route.
|
|
1783
|
-
// Three sources, in precedence order: committed non-secret config (subject.envValues), then
|
|
1784
|
-
// secret values forwarded from the caller's environment (subject.env), then the harness's own
|
|
1785
|
-
// comms wiring, which must win because only it knows the catch's address.
|
|
1786
|
-
...(subjectEnvNames.length > 0 || Object.keys(subjectEnvValues).length > 0 || Object.keys(commsEnv).length > 0
|
|
1787
|
-
? {
|
|
1788
|
-
envs: {
|
|
1789
|
-
...subjectEnvValues,
|
|
1790
|
-
...Object.fromEntries(subjectEnvNames.map((name) => [name, env[name]])),
|
|
1791
|
-
...commsEnv
|
|
1792
|
-
}
|
|
1793
|
-
}
|
|
1794
|
-
: {}),
|
|
1795
|
-
resolution: spec.resolution,
|
|
1796
|
-
dpi: 96,
|
|
1797
|
-
lifecycle: { onTimeout: "kill" }
|
|
1798
|
-
}, config.execution?.desktop?.template, {
|
|
1799
|
-
// The default loader reclaims an acquired handle before retrying failed desktop startup.
|
|
1800
|
-
// Its error names the cleanup outcome; pre-construction allocation failures remain unowned.
|
|
1801
|
-
onRetry: (reason) => {
|
|
1802
|
-
const named = redactText(deps.scrubKnownValues(reason));
|
|
1803
|
-
warnings.push(`Sandbox create for lane ${spec.laneId} retried once after a transient provider error (${named}).`);
|
|
1804
|
-
onSubjectPhase({ at: new Date(deps.now()).toISOString(), type: "cua-lab.sandbox.create.retry", message: `sandbox create retried once (${named})` });
|
|
1805
|
-
}
|
|
1806
|
-
});
|
|
1807
|
-
sandboxId = desktop.sandboxId;
|
|
1808
|
-
// #358 salvage: journal the id to disk before any work — an interrupted run reclaims by
|
|
1809
|
-
// exact recorded id (`humanish reclaim`), never by enumerating the account.
|
|
1810
|
-
await appendSandboxReceipt(deps.artifactRoot, { at: new Date(deps.now()).toISOString(), laneId: spec.laneId, sandboxId, timeoutMs: deps.perLaneSandboxMs });
|
|
1811
|
-
// The billed span starts the instant the sandbox exists.
|
|
1812
|
-
sandboxCreatedAtMs = deps.now();
|
|
1813
|
-
desktopResources = await observeDesktopResources(desktop);
|
|
1814
|
-
if ("reason" in desktopResources) {
|
|
1815
|
-
warnings.push(`Desktop resource size unavailable (${desktopResources.reason}); compute cost remains unpriced.`);
|
|
1816
|
-
}
|
|
1817
|
-
if (deps.hooks.prepareDesktop) {
|
|
1818
|
-
await deps.hooks.prepareDesktop(desktop, { laneId: spec.laneId, laneIndex: spec.laneIndex, laneCount: deps.laneCount });
|
|
1819
|
-
}
|
|
1820
|
-
// Start the in-sandbox email catch BEFORE the subject serve, so the app's send-API base URL (injected
|
|
1821
|
-
// into its env at create) resolves the moment it boots. A comms-declared lab that can't stand the
|
|
1822
|
-
// catch up is a setup failure (fail closed) rather than silently sending real mail.
|
|
1823
|
-
if (deps.receiving) {
|
|
1824
|
-
const surface = await deployReceivingInbox(desktop, { leaseId: spec.streamId, requestTimeoutMs: Math.min(deps.requestTimeoutMs, 30_000) });
|
|
1825
|
-
receivingInboxUrl = surface.url;
|
|
1826
|
-
const email = config.comms?.email;
|
|
1827
|
-
try {
|
|
1828
|
-
await deps.receiving.attach(spec.laneId, {
|
|
1829
|
-
surface,
|
|
1830
|
-
allowedOrigins: [...new Set([new URL(targetUrl).origin, ...(email?.allowedOrigins ?? [])])],
|
|
1831
|
-
originMap: buildOriginMap({
|
|
1832
|
-
...(config.subject.serve?.url === undefined ? {} : { internalServeUrl: config.subject.serve.url }),
|
|
1833
|
-
reachableBaseUrl: targetUrl,
|
|
1834
|
-
...(email?.linkOrigin === undefined ? {} : { linkOrigin: email.linkOrigin })
|
|
1835
|
-
})
|
|
1836
|
-
});
|
|
1837
|
-
commsArtifactPath = "comms/receiving.json";
|
|
1838
|
-
}
|
|
1839
|
-
catch (error) {
|
|
1840
|
-
await surface.stop().catch(() => { });
|
|
1841
|
-
throw error;
|
|
1842
|
-
}
|
|
1843
|
-
}
|
|
1844
|
-
if (commsEmail && commsPort !== undefined) {
|
|
1845
|
-
deployedComms = await deployCommsCatch(desktop, {
|
|
1846
|
-
port: commsPort,
|
|
1847
|
-
...(commsSmtpPort === undefined ? {} : { smtpPort: commsSmtpPort }),
|
|
1848
|
-
requestTimeoutMs: deps.requestTimeoutMs
|
|
772
|
+
await desktopLane.prepare();
|
|
773
|
+
// Start the brain BEFORE the first screenshot: the app-server handshake is ~500ms, and it
|
|
774
|
+
// is paid here, while the sandbox is still settling, rather than inside turn one.
|
|
775
|
+
if (deps.localAgent === "codex") {
|
|
776
|
+
appServer = await startAppServerSession({
|
|
777
|
+
...(spec.reasoningEffort === undefined ? {} : { reasoningEffort: spec.reasoningEffort }),
|
|
778
|
+
...(config.actors[0]?.model === undefined ? {} : { model: config.actors[0].model }),
|
|
779
|
+
// The persona lives on the THREAD, so it is stated once instead of re-sent every turn.
|
|
780
|
+
baseInstructions: spec.instructions
|
|
1849
781
|
});
|
|
1850
|
-
|
|
1851
|
-
|
|
1852
|
-
|
|
1853
|
-
//
|
|
1854
|
-
//
|
|
1855
|
-
//
|
|
1856
|
-
|
|
1857
|
-
|
|
1858
|
-
|
|
1859
|
-
|
|
1860
|
-
|
|
1861
|
-
|
|
1862
|
-
|
|
1863
|
-
|
|
1864
|
-
|
|
1865
|
-
|
|
1866
|
-
|
|
1867
|
-
const refreshed = await refreshInboxSurface({
|
|
1868
|
-
desktop,
|
|
1869
|
-
deployed: deployedRef,
|
|
1870
|
-
recipients: surfaceRecipients,
|
|
1871
|
-
sinceCount: surfaceRenderedCount,
|
|
1872
|
-
originMap: commsOriginMap,
|
|
1873
|
-
requestTimeoutMs: deps.requestTimeoutMs
|
|
1874
|
-
});
|
|
1875
|
-
if (refreshed.rendered)
|
|
1876
|
-
surfaceRenderedCount = refreshed.count;
|
|
1877
|
-
}
|
|
1878
|
-
catch {
|
|
1879
|
-
// Never throw into the render loop; the teardown drain + by-id teardown must still run.
|
|
1880
|
-
}
|
|
1881
|
-
if (surfaceDisposed)
|
|
1882
|
-
break;
|
|
1883
|
-
await new Promise((resolve) => {
|
|
1884
|
-
const timer = setTimeout(resolve, INBOX_SURFACE_CADENCE_MS);
|
|
1885
|
-
void surfaceDispose.then(() => { clearTimeout(timer); resolve(); });
|
|
1886
|
-
});
|
|
1887
|
-
if (surfaceDisposed)
|
|
1888
|
-
break;
|
|
1889
|
-
}
|
|
1890
|
-
})();
|
|
1891
|
-
}
|
|
1892
|
-
// Per-lane geometry assertion (fail-closed) — the device claim is verified in-sandbox.
|
|
1893
|
-
const screenGeometry = await inspectDesktopScreenGeometry({
|
|
1894
|
-
desktop,
|
|
1895
|
-
laneId: spec.laneId,
|
|
1896
|
-
requestedScreen: spec.resolution,
|
|
1897
|
-
requestTimeoutMs: deps.requestTimeoutMs
|
|
1898
|
-
});
|
|
1899
|
-
if (screenGeometry.verified) {
|
|
1900
|
-
desktopGeometry = {
|
|
1901
|
-
...desktopGeometry,
|
|
1902
|
-
screen: { ...desktopGeometry.screen, verified: screenGeometry.verified }
|
|
1903
|
-
};
|
|
1904
|
-
}
|
|
1905
|
-
if (screenGeometry.warning) {
|
|
1906
|
-
warnings.push(screenGeometry.warning);
|
|
1907
|
-
desktopGeometry = { ...desktopGeometry, warnings: [screenGeometry.warning] };
|
|
1908
|
-
}
|
|
1909
|
-
if (screenGeometry.error && deps.screenMismatchPolicy !== "record-evidence") {
|
|
1910
|
-
sessionError = screenGeometry.error;
|
|
1911
|
-
failureCode = "HUMANISH_CUA_LAB_DEVICE_GEOMETRY";
|
|
1912
|
-
}
|
|
1913
|
-
else {
|
|
1914
|
-
if (screenGeometry.error && screenGeometry.verified) {
|
|
1915
|
-
// record-evidence policy: the bundle keeps requested vs verified as separate facts and
|
|
1916
|
-
// discloses the divergence instead of failing this lane's world mid-flight.
|
|
1917
|
-
const mismatchWarning = deps.scrubKnownValues(`Lane ${spec.laneId} requested a ${spec.resolution[0]}x${spec.resolution[1]} screen but xdpyinfo reports ${screenGeometry.verified.width}x${screenGeometry.verified.height}; recording requested vs verified separately instead of failing the lane closed.`);
|
|
1918
|
-
warnings.push(mismatchWarning);
|
|
1919
|
-
desktopGeometry = {
|
|
1920
|
-
...desktopGeometry,
|
|
1921
|
-
warnings: [...(desktopGeometry.warnings ?? []), mismatchWarning]
|
|
1922
|
-
};
|
|
1923
|
-
}
|
|
1924
|
-
if (desktopCliRoute) {
|
|
1925
|
-
// Prepare the runtime and any declared product install, UNKEYED. With install omitted,
|
|
1926
|
-
// the participant discovers and installs the product from its public surfaces.
|
|
1927
|
-
await provisionDesktopCli(desktop, {
|
|
1928
|
-
product: config.subject.product?.name ?? "",
|
|
1929
|
-
...(config.subject.product?.install === undefined ? {} : { install: config.subject.product.install }),
|
|
1930
|
-
requestTimeoutMs: deps.requestTimeoutMs,
|
|
1931
|
-
scrub: deps.scrubKnownValues,
|
|
1932
|
-
onPhase: onSubjectPhase
|
|
1933
|
-
});
|
|
1934
|
-
}
|
|
1935
|
-
if (cloneRoute && serve && subjectRepo) {
|
|
1936
|
-
subjectCommit = await provisionCloneSubject(desktop, {
|
|
1937
|
-
repo: subjectRepo,
|
|
1938
|
-
depth: config.subject.clone?.depth ?? 1,
|
|
1939
|
-
serve,
|
|
1940
|
-
...(config.subject.state === undefined ? {} : { state: config.subject.state }),
|
|
1941
|
-
hasGithubToken: deps.hasGithubToken,
|
|
1942
|
-
requestTimeoutMs: deps.requestTimeoutMs,
|
|
1943
|
-
scrub: deps.scrubKnownValues,
|
|
1944
|
-
onCommit: (commit) => {
|
|
1945
|
-
subjectCommit = commit;
|
|
1946
|
-
},
|
|
1947
|
-
onStateStep: (record) => {
|
|
1948
|
-
stateStepRecords.push(record);
|
|
1949
|
-
},
|
|
1950
|
-
onPhase: onSubjectPhase,
|
|
1951
|
-
...(deps.hooks.detachedTimers ?? {})
|
|
1952
|
-
});
|
|
1953
|
-
}
|
|
1954
|
-
else if (localTreeRoute && serve && deps.localTreeArchiveBuffer) {
|
|
1955
|
-
await provisionLocalTreeSubject(desktop, {
|
|
1956
|
-
archiveBuffer: deps.localTreeArchiveBuffer,
|
|
1957
|
-
serve,
|
|
1958
|
-
...(config.subject.state === undefined ? {} : { state: config.subject.state }),
|
|
1959
|
-
requestTimeoutMs: deps.requestTimeoutMs,
|
|
1960
|
-
scrub: deps.scrubKnownValues,
|
|
1961
|
-
onStateStep: (record) => {
|
|
1962
|
-
stateStepRecords.push(record);
|
|
1963
|
-
},
|
|
1964
|
-
onPhase: onSubjectPhase,
|
|
1965
|
-
...(deps.hooks.detachedTimers ?? {})
|
|
782
|
+
localAgentProvider = appServer.provider;
|
|
783
|
+
}
|
|
784
|
+
else if (deps.localAgent === "claude") {
|
|
785
|
+
// One session for the whole run, like the codex thread above (#520). The one-shot
|
|
786
|
+
// provider (createLocalAgentProvider) spawned `claude -p` per turn, and every turn
|
|
787
|
+
// started with no memory of the last. HUMANISH_LOCAL_AGENT_ONE_SHOT=1 keeps that path
|
|
788
|
+
// reachable as a MEASUREMENT switch: MemTrapBench (2026-08) reports memory frameworks
|
|
789
|
+
// degrading agent performance by 10-40% on some tasks, so "remembers" has to be measured
|
|
790
|
+
// against "does not" on the same lab, not assumed. The trace records which one ran.
|
|
791
|
+
const oneShot = env.HUMANISH_LOCAL_AGENT_ONE_SHOT !== undefined
|
|
792
|
+
&& env.HUMANISH_LOCAL_AGENT_ONE_SHOT !== ""
|
|
793
|
+
&& env.HUMANISH_LOCAL_AGENT_ONE_SHOT !== "0";
|
|
794
|
+
if (oneShot) {
|
|
795
|
+
localAgentProvider = createLocalAgentProvider({
|
|
796
|
+
agent: "claude",
|
|
797
|
+
...(spec.reasoningEffort === undefined ? {} : { reasoningEffort: spec.reasoningEffort }),
|
|
798
|
+
...(config.actors[0]?.model === undefined ? {} : { model: config.actors[0].model })
|
|
1966
799
|
});
|
|
1967
800
|
}
|
|
1968
|
-
if (!desktopCliRoute) {
|
|
1969
|
-
const requestedFidelity = config.execution?.desktop?.fidelity;
|
|
1970
|
-
// A declared camera (#509) is in place before the browser starts: the feed is generated or
|
|
1971
|
-
// uploaded first, and a feed that cannot be produced fails the lane closed here.
|
|
1972
|
-
const requestedMedia = config.execution?.desktop?.media;
|
|
1973
|
-
const mediaEvidence = requestedMedia === undefined
|
|
1974
|
-
? undefined
|
|
1975
|
-
: await prepareDesktopMedia(desktop, requestedMedia, config.policies?.mediaPermission ?? "prompt", deps.labCwd, deps.requestTimeoutMs);
|
|
1976
|
-
const browserLaunch = await openDesktopBrowserTarget(desktop, targetUrl, deps.requestTimeoutMs, config.execution?.desktop?.browser, [
|
|
1977
|
-
...(requestedFidelity?.mobileEmulation && spec.devicePreset.isMobile
|
|
1978
|
-
? [
|
|
1979
|
-
`--user-agent=${requestedFidelity.userAgent ?? DEFAULT_MOBILE_USER_AGENT}`,
|
|
1980
|
-
...(requestedFidelity.touch === false ? [] : ["--touch-events=enabled"])
|
|
1981
|
-
]
|
|
1982
|
-
: []),
|
|
1983
|
-
...(mediaEvidence?.flags ?? [])
|
|
1984
|
-
]);
|
|
1985
|
-
desktopBrowser = mediaEvidence === undefined
|
|
1986
|
-
? browserLaunch.evidence
|
|
1987
|
-
: { requested: config.execution?.desktop?.browser ?? "default", ...(browserLaunch.evidence ?? {}), media: mediaEvidence };
|
|
1988
|
-
if (mediaEvidence !== undefined && browserLaunch.family !== "chromium") {
|
|
1989
|
-
throw new Error(`execution.desktop.media needs Chrome or Chromium on lane ${spec.laneId} (the fake-device flags are Chromium's); the launched browser family is ${browserLaunch.family}. Set execution.desktop.browser: chrome.`);
|
|
1990
|
-
}
|
|
1991
|
-
launchedBrowserFamily = browserLaunch.family;
|
|
1992
|
-
browserLaunchIdentity = browserLaunch.identity;
|
|
1993
|
-
browserLaunched = true;
|
|
1994
|
-
await desktop.wait(BROWSER_SETTLE_MS).catch(() => undefined);
|
|
1995
|
-
// Mobile fidelity beyond viewport size (#221): applied to the launch page before the
|
|
1996
|
-
// geometry capture and the participant's first observation, OUTSIDE the stream/geometry
|
|
1997
|
-
// try below (whose catch degrades to a warning): a request that cannot be applied fails
|
|
1998
|
-
// the lane closed with the reason.
|
|
1999
|
-
// Only lanes on a mobile preset are emulated: a run-wide flag must not hand a desktop or
|
|
2000
|
-
// tablet lane an iPhone user agent (the first live proof did exactly that to the desktop
|
|
2001
|
-
// newcomer beside the phone lane). Those lanes carry no fidelity block, which is honest.
|
|
2002
|
-
const fidelityRequest = config.execution?.desktop?.fidelity;
|
|
2003
|
-
if (fidelityRequest?.mobileEmulation && spec.devicePreset.isMobile) {
|
|
2004
|
-
if (launchedBrowserFamily !== "chromium") {
|
|
2005
|
-
throw new Error(`execution.desktop.fidelity.mobileEmulation needs Chrome or Chromium on lane ${spec.laneId}; the launched browser family is ${launchedBrowserFamily}. Set execution.desktop.browser: chrome.`);
|
|
2006
|
-
}
|
|
2007
|
-
const applied = await applyMobileEmulation(desktop, deps.requestTimeoutMs, {
|
|
2008
|
-
...(browserLaunchIdentity?.cdpPort === undefined ? {} : { cdpPort: browserLaunchIdentity.cdpPort }),
|
|
2009
|
-
...(browserLaunchIdentity?.profileDir === undefined ? {} : { profileDir: browserLaunchIdentity.profileDir }),
|
|
2010
|
-
targetUrl
|
|
2011
|
-
}, browserTargetId, {
|
|
2012
|
-
width: spec.devicePreset.width,
|
|
2013
|
-
height: spec.devicePreset.height,
|
|
2014
|
-
deviceScaleFactor: fidelityRequest.deviceScaleFactor ?? spec.devicePreset.deviceScaleFactor,
|
|
2015
|
-
touch: fidelityRequest.touch ?? true,
|
|
2016
|
-
userAgent: fidelityRequest.userAgent ?? DEFAULT_MOBILE_USER_AGENT
|
|
2017
|
-
});
|
|
2018
|
-
appliedFidelity = applied.fidelity;
|
|
2019
|
-
emulatedTargetId = applied.targetId;
|
|
2020
|
-
emulationHolderName = applied.holderName;
|
|
2021
|
-
warnings.push(...applied.warnings);
|
|
2022
|
-
}
|
|
2023
|
-
}
|
|
2024
801
|
else {
|
|
2025
|
-
|
|
2026
|
-
// participant arrives at a desktop with the thing they were asked to use already in front
|
|
2027
|
-
// of them. They can still open another from the dock — that is the point of a desktop.
|
|
2028
|
-
await openDesktopTerminal(desktop, deps.requestTimeoutMs, config.subject.product?.workdir);
|
|
2029
|
-
await desktop.wait(BROWSER_SETTLE_MS).catch(() => undefined);
|
|
2030
|
-
}
|
|
2031
|
-
// Start the brain BEFORE the first screenshot: the app-server handshake is ~500ms, and it
|
|
2032
|
-
// is paid here, while the sandbox is still settling, rather than inside turn one.
|
|
2033
|
-
if (deps.localAgent === "codex") {
|
|
2034
|
-
appServer = await startAppServerSession({
|
|
802
|
+
claudeSession = await startClaudeSession({
|
|
2035
803
|
...(spec.reasoningEffort === undefined ? {} : { reasoningEffort: spec.reasoningEffort }),
|
|
2036
|
-
...(config.actors[0]?.model === undefined ? {} : { model: config.actors[0].model })
|
|
2037
|
-
// The persona lives on the THREAD, so it is stated once instead of re-sent every turn.
|
|
2038
|
-
baseInstructions: spec.instructions
|
|
804
|
+
...(config.actors[0]?.model === undefined ? {} : { model: config.actors[0].model })
|
|
2039
805
|
});
|
|
2040
|
-
localAgentProvider =
|
|
2041
|
-
}
|
|
2042
|
-
else if (deps.localAgent === "claude") {
|
|
2043
|
-
// One session for the whole run, like the codex thread above (#520). The one-shot
|
|
2044
|
-
// provider (createLocalAgentProvider) spawned `claude -p` per turn, and every turn
|
|
2045
|
-
// started with no memory of the last. HUMANISH_LOCAL_AGENT_ONE_SHOT=1 keeps that path
|
|
2046
|
-
// reachable as a MEASUREMENT switch: MemTrapBench (2026-08) reports memory frameworks
|
|
2047
|
-
// degrading agent performance by 10-40% on some tasks, so "remembers" has to be measured
|
|
2048
|
-
// against "does not" on the same lab, not assumed. The trace records which one ran.
|
|
2049
|
-
const oneShot = env.HUMANISH_LOCAL_AGENT_ONE_SHOT !== undefined
|
|
2050
|
-
&& env.HUMANISH_LOCAL_AGENT_ONE_SHOT !== ""
|
|
2051
|
-
&& env.HUMANISH_LOCAL_AGENT_ONE_SHOT !== "0";
|
|
2052
|
-
if (oneShot) {
|
|
2053
|
-
localAgentProvider = createLocalAgentProvider({
|
|
2054
|
-
agent: "claude",
|
|
2055
|
-
...(spec.reasoningEffort === undefined ? {} : { reasoningEffort: spec.reasoningEffort }),
|
|
2056
|
-
...(config.actors[0]?.model === undefined ? {} : { model: config.actors[0].model })
|
|
2057
|
-
});
|
|
2058
|
-
}
|
|
2059
|
-
else {
|
|
2060
|
-
claudeSession = await startClaudeSession({
|
|
2061
|
-
...(spec.reasoningEffort === undefined ? {} : { reasoningEffort: spec.reasoningEffort }),
|
|
2062
|
-
...(config.actors[0]?.model === undefined ? {} : { model: config.actors[0].model })
|
|
2063
|
-
});
|
|
2064
|
-
localAgentProvider = claudeSession.provider;
|
|
2065
|
-
}
|
|
2066
|
-
}
|
|
2067
|
-
// World is ready: release the pipeline gate so the remaining lanes may start.
|
|
2068
|
-
provisioned = true;
|
|
2069
|
-
signal(true);
|
|
2070
|
-
try {
|
|
2071
|
-
// No browser means no browser geometry, and none is invented: the CSS-viewport facts a
|
|
2072
|
-
// browser reports have no counterpart in a terminal window, and an empty record shaped like
|
|
2073
|
-
// a measurement would read as one. The screen geometry above is still verified.
|
|
2074
|
-
if (!desktopCliRoute) {
|
|
2075
|
-
const browserGeometry = await captureDesktopBrowserGeometry({
|
|
2076
|
-
desktop,
|
|
2077
|
-
browserFamily: launchedBrowserFamily,
|
|
2078
|
-
...(browserLaunchIdentity === undefined ? {} : { launchIdentity: browserLaunchIdentity }),
|
|
2079
|
-
laneId: spec.laneId,
|
|
2080
|
-
targetUrl,
|
|
2081
|
-
requestedScreen: spec.resolution,
|
|
2082
|
-
requestTimeoutMs: deps.requestTimeoutMs
|
|
2083
|
-
});
|
|
2084
|
-
initialBrowserGeometry = browserGeometry;
|
|
2085
|
-
browserWindowId = browserGeometry.browserWindowId;
|
|
2086
|
-
browserTargetId = browserGeometry.browserTargetId;
|
|
2087
|
-
}
|
|
2088
|
-
// The WHOLE desktop, not one window: a person studying a terminal app opens other windows,
|
|
2089
|
-
// and a stream bound to the first one would quietly stop being evidence.
|
|
2090
|
-
await startDesktopStream(desktop, browserWindowId);
|
|
2091
|
-
const candidateStreamUrl = desktop.stream.getUrl({
|
|
2092
|
-
authKey: desktop.stream.getAuthKey(),
|
|
2093
|
-
autoConnect: true,
|
|
2094
|
-
viewOnly: true,
|
|
2095
|
-
resize: "scale"
|
|
2096
|
-
});
|
|
2097
|
-
if (typeof candidateStreamUrl === "string" && candidateStreamUrl.trim().length > 0) {
|
|
2098
|
-
streamUrl = candidateStreamUrl;
|
|
2099
|
-
await deps.hooks.onRuntimeStreamReady?.({
|
|
2100
|
-
laneId: spec.laneId,
|
|
2101
|
-
sandboxId: desktop.sandboxId,
|
|
2102
|
-
simId: spec.simId,
|
|
2103
|
-
streamId: spec.streamId,
|
|
2104
|
-
url: streamUrl
|
|
2105
|
-
});
|
|
2106
|
-
}
|
|
2107
|
-
else {
|
|
2108
|
-
warnings.push("Live desktop stream started but did not return a usable watch URL; Observer will fall back to screenshots.");
|
|
2109
|
-
}
|
|
806
|
+
localAgentProvider = claudeSession.provider;
|
|
2110
807
|
}
|
|
2111
|
-
catch (error) {
|
|
2112
|
-
warnings.push(`Live desktop stream unavailable (run continues; evidence still captured): ${redactText(deps.scrubKnownValues(toErrorMessage(error)))}`);
|
|
2113
|
-
}
|
|
2114
|
-
// This is outside the stream's best-effort catch: unusable geometry is a harness failure,
|
|
2115
|
-
// never a participant finding about missing controls. Both per-lane and concurrent seats
|
|
2116
|
-
// use this route; sequential seats enforce the same capture result in shared-world-lab.
|
|
2117
|
-
if (initialBrowserGeometry?.unusable !== undefined) {
|
|
2118
|
-
failureCode = "HUMANISH_CUA_LAB_DEVICE_GEOMETRY";
|
|
2119
|
-
throw new Error(`${failureCode}: ${initialBrowserGeometry.unusable} Participant actions were not started.`);
|
|
2120
|
-
}
|
|
2121
|
-
// The FAIL-CLOSED spend cap (execution.caps.maxUsd) is wired into the loop as maxUsd + an
|
|
2122
|
-
// injected pure per-turn estimator keyed on the resolved model. Preflight already refused a
|
|
2123
|
-
// cap on an unpriced model, so the estimate is measurable whenever a cap is in force. The
|
|
2124
|
-
// model id here matches provider.version (openai-responses-cu resolves the default when unset).
|
|
2125
|
-
const capModelId = config.actors[0]?.model ?? DEFAULT_OPENAI_CU_MODEL;
|
|
2126
|
-
const maxUsd = config.execution?.caps?.maxUsd;
|
|
2127
|
-
const sessionOptions = {
|
|
2128
|
-
// Tell the persona where its inbox is — but only when comms is live AND this lane has a declared
|
|
2129
|
-
// recipient it can actually receive mail into (else it would stall on an inbox that stays
|
|
2130
|
-
// empty). Two comms planes, mutually exclusive by parse: the in-sandbox catch humanish
|
|
2131
|
-
// deployed, or the adopter-hosted one (#380).
|
|
2132
|
-
instructions: deps.receiving && receivingInboxUrl
|
|
2133
|
-
? withInboxMission(spec, receivingInboxUrl, deps.receiving.address(spec.laneId), true).instructions
|
|
2134
|
-
: commsEmail && commsInboxUrl && deployedComms?.ready && laneHasInboxRecipient(commsEmail, spec.laneId)
|
|
2135
|
-
? withInboxMission(spec, commsInboxUrl, inboxRecipientFor(commsEmail, spec.laneId)?.address).instructions
|
|
2136
|
-
: deps.externalComms && laneHasInboxRecipient(deps.externalComms.email, spec.laneId)
|
|
2137
|
-
? withInboxMission(spec, deps.externalComms.inboxUrl, inboxRecipientFor(deps.externalComms.email, spec.laneId)?.address).instructions
|
|
2138
|
-
: spec.instructions,
|
|
2139
|
-
persona: spec.persona,
|
|
2140
|
-
timeoutMs: deps.timeoutMs,
|
|
2141
|
-
// The brain is either a keyed API client or a CLI the operator is already signed in to.
|
|
2142
|
-
// Everything below this line — loop, executor, trace, affordances — is identical either
|
|
2143
|
-
// way, which is what makes a local-agent run comparable to an API one.
|
|
2144
|
-
...(localAgentProvider === undefined ? {} : { provider: localAgentProvider }),
|
|
2145
|
-
openai: {
|
|
2146
|
-
apiKey: deps.openaiApiKey,
|
|
2147
|
-
...(config.actors[0]?.model ? { model: config.actors[0].model } : {}),
|
|
2148
|
-
// Per-LANE, not per-actor: two lanes at different efforts is the control this exists for.
|
|
2149
|
-
...(spec.reasoningEffort === undefined ? {} : { reasoningEffort: spec.reasoningEffort }),
|
|
2150
|
-
...(spec.maxOutputTokens === undefined ? {} : { maxOutputTokens: spec.maxOutputTokens })
|
|
2151
|
-
},
|
|
2152
|
-
...(maxUsd === undefined
|
|
2153
|
-
? {}
|
|
2154
|
-
: {
|
|
2155
|
-
maxUsd,
|
|
2156
|
-
estimateTurnCostUsd: (usage) => estimateActorCost(usage, capModelId).estimatedCostUsd
|
|
2157
|
-
}),
|
|
2158
|
-
desktop: desktop,
|
|
2159
|
-
...(launchedBrowserFamily === "chromium"
|
|
2160
|
-
? {
|
|
2161
|
-
executorOptions: {
|
|
2162
|
-
observeBrowserState: makeChromeBrowserStateObserver(desktop, deps.requestTimeoutMs, {
|
|
2163
|
-
...(browserLaunchIdentity?.cdpPort === undefined ? {} : { cdpPort: browserLaunchIdentity.cdpPort }),
|
|
2164
|
-
...(browserLaunchIdentity?.profileDir === undefined ? {} : { profileDir: browserLaunchIdentity.profileDir }),
|
|
2165
|
-
targetUrl
|
|
2166
|
-
}, browserTargetId,
|
|
2167
|
-
// Once per lane: a dark observation channel is a gap in the instrument, and the
|
|
2168
|
-
// funnel's NEVER MEASURED count needs this line to explain itself (#514).
|
|
2169
|
-
(reason) => {
|
|
2170
|
-
warnings.push(`Browser-state observer unavailable for lane ${spec.laneId} (${redactText(deps.scrubKnownValues(reason))}); ` +
|
|
2171
|
-
"urlIncludes/urlPathEquals/textIncludes stop conditions and task criteria are NOT being measured this session.");
|
|
2172
|
-
}, emulatedTargetId === undefined
|
|
2173
|
-
? undefined
|
|
2174
|
-
: {
|
|
2175
|
-
emulatedTargetId,
|
|
2176
|
-
expectedWidth: spec.devicePreset.width,
|
|
2177
|
-
expectTouch: appliedFidelity?.requested.touch === true,
|
|
2178
|
-
onDrift: (reason) => {
|
|
2179
|
-
warnings.push(`Mobile emulation drift on lane ${spec.laneId}: ${reason} (#623).`);
|
|
2180
|
-
},
|
|
2181
|
-
onCovered: (coveredTargetId, read) => {
|
|
2182
|
-
// A later tab the page itself reported at the phone width: evidence that
|
|
2183
|
-
// the emulation followed the participant (#623), kept on the bundle.
|
|
2184
|
-
if (appliedFidelity === undefined)
|
|
2185
|
-
return;
|
|
2186
|
-
appliedFidelity = {
|
|
2187
|
-
...appliedFidelity,
|
|
2188
|
-
laterTargets: [...(appliedFidelity.laterTargets ?? []), { targetId: coveredTargetId, ...read }]
|
|
2189
|
-
};
|
|
2190
|
-
}
|
|
2191
|
-
})
|
|
2192
|
-
}
|
|
2193
|
-
}
|
|
2194
|
-
: {}),
|
|
2195
|
-
redactScreenshots: deps.redactScreenshots,
|
|
2196
|
-
scrubText: deps.scrubKnownValues,
|
|
2197
|
-
writeScreenshot,
|
|
2198
|
-
...(spec.idleSteps === undefined ? {} : { idleSteps: spec.idleSteps }),
|
|
2199
|
-
...(spec.noProgressSteps === undefined ? {} : { noProgressSteps: spec.noProgressSteps }),
|
|
2200
|
-
...(spec.stopWhen === undefined ? {} : { stopWhen: spec.stopWhen }),
|
|
2201
|
-
...(spec.dwell === undefined ? {} : { dwell: spec.dwell }),
|
|
2202
|
-
...(spec.tasks === undefined ? {} : { tasks: spec.tasks }),
|
|
2203
|
-
// The STUDY budget (#299): this lane notes its own running estimate on the shared ledger
|
|
2204
|
-
// and stops when the RUN total crosses the cap — independent of the per-lane maxUsd above.
|
|
2205
|
-
...(deps.runBudget === undefined
|
|
2206
|
-
? {}
|
|
2207
|
-
: {
|
|
2208
|
-
overRunBudget: (usage) => {
|
|
2209
|
-
const estimate = estimateActorCost(usage, capModelId).estimatedCostUsd;
|
|
2210
|
-
const totalUsd = deps.runBudget.note(spec.laneId, estimate);
|
|
2211
|
-
return totalUsd > deps.runBudget.maxTotalUsd
|
|
2212
|
-
? `study budget reached: the run's estimated model spend $${round6(totalUsd)} crossed execution.caps.maxTotalUsd=$${deps.runBudget.maxTotalUsd}; this lane stops here and sibling lanes stop at their next turn`
|
|
2213
|
-
: null;
|
|
2214
|
-
}
|
|
2215
|
-
}),
|
|
2216
|
-
...(deps.onObservedUrl === undefined ? {} : { onObservedUrl: deps.onObservedUrl }),
|
|
2217
|
-
...(deps.onMessage === undefined ? {} : { onMessage: deps.onMessage }),
|
|
2218
|
-
...(deps.onScreenshot === undefined ? {} : { onScreenshot: deps.onScreenshot }),
|
|
2219
|
-
...(deps.onTrace === undefined
|
|
2220
|
-
? {}
|
|
2221
|
-
: {
|
|
2222
|
-
// Forwards the RUNNING usage as well: the lane is where both are known, and usage
|
|
2223
|
-
// without it never reaches the flush — which is how the live cost stayed unknown.
|
|
2224
|
-
onTrace: (items, usage) => deps.onTrace?.(spec.laneId, items, usage)
|
|
2225
|
-
})
|
|
2226
|
-
};
|
|
2227
|
-
session = await deps.runSession(sessionOptions);
|
|
2228
808
|
}
|
|
809
|
+
// World is ready: release the pipeline gate so the remaining lanes may start.
|
|
810
|
+
provisioned = true;
|
|
811
|
+
signal(true);
|
|
812
|
+
const ready = await desktopLane.openSession();
|
|
813
|
+
// The FAIL-CLOSED spend cap (execution.caps.maxUsd) is wired into the loop as maxUsd + an
|
|
814
|
+
// injected pure per-turn estimator keyed on the resolved model. Preflight already refused a
|
|
815
|
+
// cap on an unpriced model, so the estimate is measurable whenever a cap is in force. The
|
|
816
|
+
// model id here matches provider.version (openai-responses-cu resolves the default when unset).
|
|
817
|
+
const capModelId = config.actors[0]?.model ?? DEFAULT_OPENAI_CU_MODEL;
|
|
818
|
+
const maxUsd = config.execution?.caps?.maxUsd;
|
|
819
|
+
const sessionOptions = {
|
|
820
|
+
instructions: ready.inbox
|
|
821
|
+
? withInboxMission(spec, ready.inbox.url, ready.inbox.address, ready.inbox.receiving).instructions
|
|
822
|
+
: spec.instructions,
|
|
823
|
+
persona: spec.persona,
|
|
824
|
+
timeoutMs: deps.timeoutMs,
|
|
825
|
+
// The brain is either a keyed API client or a CLI the operator is already signed in to.
|
|
826
|
+
// Everything below this line — loop, executor, trace, affordances — is identical either
|
|
827
|
+
// way, which is what makes a local-agent run comparable to an API one.
|
|
828
|
+
...(localAgentProvider === undefined ? {} : { provider: localAgentProvider }),
|
|
829
|
+
openai: {
|
|
830
|
+
apiKey: deps.openaiApiKey,
|
|
831
|
+
...(config.actors[0]?.model ? { model: config.actors[0].model } : {}),
|
|
832
|
+
// Per-LANE, not per-actor: two lanes at different efforts is the control this exists for.
|
|
833
|
+
...(spec.reasoningEffort === undefined ? {} : { reasoningEffort: spec.reasoningEffort }),
|
|
834
|
+
...(spec.maxOutputTokens === undefined ? {} : { maxOutputTokens: spec.maxOutputTokens })
|
|
835
|
+
},
|
|
836
|
+
...(maxUsd === undefined
|
|
837
|
+
? {}
|
|
838
|
+
: {
|
|
839
|
+
maxUsd,
|
|
840
|
+
estimateTurnCostUsd: (usage) => estimateActorCost(usage, capModelId).estimatedCostUsd
|
|
841
|
+
}),
|
|
842
|
+
executor: ready.executor,
|
|
843
|
+
redactScreenshots: deps.redactScreenshots,
|
|
844
|
+
scrubText: deps.scrubKnownValues,
|
|
845
|
+
writeScreenshot,
|
|
846
|
+
...(spec.idleSteps === undefined ? {} : { idleSteps: spec.idleSteps }),
|
|
847
|
+
...(spec.noProgressSteps === undefined ? {} : { noProgressSteps: spec.noProgressSteps }),
|
|
848
|
+
...(spec.stopWhen === undefined ? {} : { stopWhen: spec.stopWhen }),
|
|
849
|
+
...(spec.dwell === undefined ? {} : { dwell: spec.dwell }),
|
|
850
|
+
...(spec.tasks === undefined ? {} : { tasks: spec.tasks }),
|
|
851
|
+
// The STUDY budget (#299): this lane notes its own running estimate on the shared ledger
|
|
852
|
+
// and stops when the RUN total crosses the cap — independent of the per-lane maxUsd above.
|
|
853
|
+
...(deps.runBudget === undefined
|
|
854
|
+
? {}
|
|
855
|
+
: {
|
|
856
|
+
overRunBudget: (usage) => {
|
|
857
|
+
const estimate = estimateActorCost(usage, capModelId).estimatedCostUsd;
|
|
858
|
+
const totalUsd = deps.runBudget.note(spec.laneId, estimate);
|
|
859
|
+
return totalUsd > deps.runBudget.maxTotalUsd
|
|
860
|
+
? `study budget reached: the run's estimated model spend $${round6(totalUsd)} crossed execution.caps.maxTotalUsd=$${deps.runBudget.maxTotalUsd}; this lane stops here and sibling lanes stop at their next turn`
|
|
861
|
+
: null;
|
|
862
|
+
}
|
|
863
|
+
}),
|
|
864
|
+
...(deps.onObservedUrl === undefined ? {} : { onObservedUrl: deps.onObservedUrl }),
|
|
865
|
+
...(deps.onMessage === undefined ? {} : { onMessage: deps.onMessage }),
|
|
866
|
+
...(deps.onScreenshot === undefined ? {} : { onScreenshot: deps.onScreenshot }),
|
|
867
|
+
...(deps.onTrace === undefined
|
|
868
|
+
? {}
|
|
869
|
+
: {
|
|
870
|
+
// Forwards the RUNNING usage as well: the lane is where both are known, and usage
|
|
871
|
+
// without it never reaches the flush — which is how the live cost stayed unknown.
|
|
872
|
+
onTrace: (items, usage) => deps.onTrace?.(spec.laneId, items, usage)
|
|
873
|
+
})
|
|
874
|
+
};
|
|
875
|
+
session = await deps.runSession(sessionOptions);
|
|
2229
876
|
}
|
|
2230
877
|
catch (error) {
|
|
2231
878
|
sessionError = redactText(deps.scrubKnownValues(toErrorMessage(error)));
|
|
2232
879
|
}
|
|
2233
880
|
finally {
|
|
2234
|
-
|
|
2235
|
-
|
|
2236
|
-
appServer?.close();
|
|
2237
|
-
await claudeSession?.close();
|
|
2238
|
-
// Stop the mid-run inbox-surface loop FIRST — before the teardown evidence drain below — so the two
|
|
2239
|
-
// `cat`s never overlap and the final surface state is deterministic. A surface failure can never
|
|
2240
|
-
// block teardown (the loop body is fully try/caught and this await is on its already-caught promise).
|
|
2241
|
-
surfaceDisposed = true;
|
|
2242
|
-
releaseSurface();
|
|
2243
|
-
if (surfaceLoop)
|
|
2244
|
-
await surfaceLoop.catch(() => undefined);
|
|
2245
|
-
if (!provisioned) {
|
|
2246
|
-
signal(false);
|
|
881
|
+
try {
|
|
882
|
+
appServer?.close();
|
|
2247
883
|
}
|
|
2248
|
-
|
|
2249
|
-
|
|
2250
|
-
|
|
2251
|
-
|
|
2252
|
-
|
|
2253
|
-
|
|
2254
|
-
|
|
2255
|
-
|
|
2256
|
-
|
|
2257
|
-
|
|
2258
|
-
|
|
2259
|
-
|
|
2260
|
-
|
|
2261
|
-
|
|
2262
|
-
|
|
2263
|
-
warnings: [`Final browser geometry measurement failed for lane ${spec.laneId}: ${redactText(deps.scrubKnownValues(toErrorMessage(error)))}`]
|
|
2264
|
-
}));
|
|
2265
|
-
// Chosen capture rule: final-if-it-measured-anything, else launch-time. A final capture
|
|
2266
|
-
// that measured EITHER field wins whole, so a partial final capture omits fields the
|
|
2267
|
-
// launch-time capture had (honest omission); only a final capture that measured NOTHING
|
|
2268
|
-
// falls back to the launch-time capture.
|
|
2269
|
-
const chosenGeometry = finalGeometry.browserWindow !== undefined || finalGeometry.viewport !== undefined
|
|
2270
|
-
? finalGeometry
|
|
2271
|
-
: initialBrowserGeometry ?? finalGeometry;
|
|
2272
|
-
const geometryWarnings = [...new Set([...(initialBrowserGeometry?.warnings ?? []), ...chosenGeometry.warnings].map((warning) => deps.scrubKnownValues(warning)))];
|
|
2273
|
-
warnings.push(...geometryWarnings);
|
|
2274
|
-
// The emulation holder's own log, after its announce line: which later targets it
|
|
2275
|
-
// attached to, what it sent, and any reply that came back as an error (#623). Read while
|
|
2276
|
-
// the sandbox is alive; the first live proof had no way to say what the holder did.
|
|
2277
|
-
if (appliedFidelity !== undefined && emulationHolderName !== undefined) {
|
|
2278
|
-
const holderLog = await readDetachedLog(desktop, emulationHolderName, deps.requestTimeoutMs).catch(() => "");
|
|
2279
|
-
const lines = holderLog.split("\n").map((line) => line.trim()).filter((line) => line.startsWith("{")).slice(1, 51);
|
|
2280
|
-
if (lines.length > 0)
|
|
2281
|
-
appliedFidelity = { ...appliedFidelity, holderLog: lines.map((line) => deps.scrubKnownValues(line)) };
|
|
2282
|
-
}
|
|
2283
|
-
desktopGeometry = {
|
|
2284
|
-
screen: desktopGeometry.screen,
|
|
2285
|
-
...(chosenGeometry.browserWindow === undefined ? {} : { browserWindow: chosenGeometry.browserWindow }),
|
|
2286
|
-
...(chosenGeometry.viewport === undefined ? {} : { viewport: chosenGeometry.viewport }),
|
|
2287
|
-
...(appliedFidelity === undefined ? {} : { fidelity: appliedFidelity }),
|
|
2288
|
-
...((desktopGeometry.warnings?.length ?? 0) + geometryWarnings.length === 0
|
|
2289
|
-
? {}
|
|
2290
|
-
: { warnings: [...(desktopGeometry.warnings ?? []), ...geometryWarnings] })
|
|
2291
|
-
};
|
|
2292
|
-
}
|
|
2293
|
-
if (deps.receiving) {
|
|
2294
|
-
try {
|
|
2295
|
-
await deps.receiving.finishParticipant(spec.laneId);
|
|
2296
|
-
}
|
|
2297
|
-
catch {
|
|
2298
|
-
warnings.push("Real email finalization is incomplete. Inspect communication cleanup with humanish comms recover.");
|
|
2299
|
-
}
|
|
2300
|
-
}
|
|
2301
|
-
// Off-app comms evidence (#297): before this lane's sandbox is torn down, drain everything the
|
|
2302
|
-
// in-sandbox catch captured, route it into a host fake inbox addressed to the declared
|
|
2303
|
-
// recipients, and write the digest-only thread artifact. Wrapped so a drain failure NEVER
|
|
2304
|
-
// breaks teardown — the sandbox must still be killed either way. Runs only for a ready catch.
|
|
2305
|
-
if (commsEmail && deployedComms?.ready) {
|
|
2306
|
-
try {
|
|
2307
|
-
const commsChannel = new FakeInbox();
|
|
2308
|
-
const commsInboxes = [];
|
|
2309
|
-
for (const recipient of commsEmail.recipients ?? []) {
|
|
2310
|
-
if (recipient.address !== undefined) {
|
|
2311
|
-
commsInboxes.push(await commsChannel.provisionAddress(recipient.lane, recipient.address));
|
|
2312
|
-
}
|
|
2313
|
-
}
|
|
2314
|
-
const collected = await collectCommsThread({
|
|
2315
|
-
desktop,
|
|
2316
|
-
deployed: deployedComms,
|
|
2317
|
-
channel: commsChannel,
|
|
2318
|
-
inboxes: commsInboxes,
|
|
2319
|
-
requestTimeoutMs: deps.requestTimeoutMs
|
|
2320
|
-
});
|
|
2321
|
-
if (collected.artifact) {
|
|
2322
|
-
const path = deps.laneCount === 1 ? "comms/thread.json" : `comms/${spec.streamId}.thread.json`;
|
|
2323
|
-
await writeContainedOutputFile(deps.artifactRoot, path, `${JSON.stringify(collected.artifact, null, 2)}\n`, "utf8");
|
|
2324
|
-
commsArtifactPath = path;
|
|
2325
|
-
}
|
|
2326
|
-
else if (collected.captured > 0) {
|
|
2327
|
-
// Captured mail that matched no declared recipient must not vanish silently (invariant 6:
|
|
2328
|
-
// honest signals): tell the operator to declare comms.email.recipients[].address to match
|
|
2329
|
-
// the address the app actually sends to (e.g. the one the persona surface will sign up with).
|
|
2330
|
-
warnings.push(`Comms catch captured ${collected.captured} email send(s) but none matched a declared recipient inbox — no comms evidence written. Declare comms.email.recipients[].address to match the address the app sends to.`);
|
|
2331
|
-
}
|
|
2332
|
-
else {
|
|
2333
|
-
// Zero captures is the silent-broken shape (#351): the app never posted to the catch at
|
|
2334
|
-
// all, so the personas stared at an empty inbox. Most common cause: the app does not
|
|
2335
|
-
// actually read the declared injectEnv var for its email API base URL.
|
|
2336
|
-
const transportHint = commsEmail.smtp
|
|
2337
|
-
? `Verify the app reads ${commsEmail.smtp.hostEnv}/${commsEmail.smtp.portEnv} for its SMTP host and port`
|
|
2338
|
-
: `Verify the app reads ${commsEmail.injectEnv} for its email API base URL (an SDK that ignores it sends real mail or throws)`;
|
|
2339
|
-
warnings.push(`Comms catch captured ZERO email sends — the app never delivered mail through the catch. ${transportHint} and that the flow reached an email step.`);
|
|
2340
|
-
}
|
|
2341
|
-
}
|
|
2342
|
-
catch (error) {
|
|
2343
|
-
warnings.push(`Comms evidence collection failed (run continues; sandbox still torn down): ${redactText(deps.scrubKnownValues(toErrorMessage(error)))}`);
|
|
2344
|
-
}
|
|
2345
|
-
}
|
|
2346
|
-
const failed = sessionError !== undefined || session === undefined;
|
|
2347
|
-
// Each route's own keep flag gates its own lane only: a clone.keep can never leak into
|
|
2348
|
-
// a local-tree lane's teardown decision, and vice versa.
|
|
2349
|
-
const keepReason = cloneRoute && config.subject.clone?.keep === true
|
|
2350
|
-
? "subject.clone.keep"
|
|
2351
|
-
: localTreeRoute && config.subject.localTree?.keep === true
|
|
2352
|
-
? "subject.localTree.keep"
|
|
2353
|
-
: undefined;
|
|
2354
|
-
const keepForDebug = keepReason !== undefined && failed;
|
|
2355
|
-
if (keepForDebug) {
|
|
2356
|
-
warnings.push(`Sandbox ${desktop.sandboxId} kept for debugging (${keepReason} on failure); reclaim it via E2B or it will be killed on its server-side timeout.`);
|
|
2357
|
-
}
|
|
2358
|
-
else if (typeof desktopModule.Sandbox.kill === "function") {
|
|
2359
|
-
try {
|
|
2360
|
-
await desktopModule.Sandbox.kill(desktop.sandboxId, { requestTimeoutMs: 60_000 });
|
|
2361
|
-
killed = true;
|
|
2362
|
-
}
|
|
2363
|
-
catch (error) {
|
|
2364
|
-
warnings.push(`Sandbox teardown failed (server-side kill-on-timeout will reclaim it): ${redactText(deps.scrubKnownValues(toErrorMessage(error)))}`);
|
|
2365
|
-
}
|
|
2366
|
-
}
|
|
2367
|
-
else {
|
|
2368
|
-
warnings.push("Installed @e2b/desktop SDK does not expose Sandbox.kill; server-side kill-on-timeout will reclaim the sandbox.");
|
|
2369
|
-
}
|
|
2370
|
-
// Close the observed span. A kept or unconfirmed sandbox can still accrue compute cost;
|
|
2371
|
-
// the summary records that remaining lifetime as unknown instead of calling this complete.
|
|
2372
|
-
sandboxTornDownAtMs = deps.now();
|
|
2373
|
-
// The lane's live stream is now a dead page whichever teardown path ran (killed, kept, or
|
|
2374
|
-
// kill-failed-awaiting-TTL) — tell the watch overlay so the tile falls back to recorded
|
|
2375
|
-
// evidence instead of "sandbox not found" (#357). Guarded: a viewer callback must never
|
|
2376
|
-
// break teardown.
|
|
2377
|
-
if (streamUrl !== undefined) {
|
|
2378
|
-
try {
|
|
2379
|
-
await deps.hooks.onRuntimeStreamEnded?.({ laneId: spec.laneId, simId: spec.simId, streamId: spec.streamId });
|
|
2380
|
-
}
|
|
2381
|
-
catch {
|
|
2382
|
-
// viewer-side only; nothing to record
|
|
2383
|
-
}
|
|
2384
|
-
}
|
|
884
|
+
catch {
|
|
885
|
+
warnings.push('Codex session cleanup failed; desktop cleanup will still run.');
|
|
886
|
+
}
|
|
887
|
+
try {
|
|
888
|
+
await claudeSession?.close();
|
|
889
|
+
}
|
|
890
|
+
catch {
|
|
891
|
+
warnings.push('Claude session cleanup failed; desktop cleanup will still run.');
|
|
892
|
+
}
|
|
893
|
+
try {
|
|
894
|
+
if (!provisioned)
|
|
895
|
+
signal(false);
|
|
896
|
+
}
|
|
897
|
+
finally {
|
|
898
|
+
await desktopLane.finalize({ failed: sessionError !== undefined || session === undefined });
|
|
2385
899
|
}
|
|
2386
900
|
}
|
|
2387
|
-
// Host-side approximation of the E2B desktop's billed lifetime; feeds the desktop-minute cost
|
|
2388
|
-
// estimate. Never negative.
|
|
2389
|
-
const desktopDurationMs = sandboxCreatedAtMs !== undefined && sandboxTornDownAtMs !== undefined
|
|
2390
|
-
? Math.max(0, sandboxTornDownAtMs - sandboxCreatedAtMs)
|
|
2391
|
-
: undefined;
|
|
2392
901
|
if (session) {
|
|
2393
902
|
// Per-lane model-token cost ESTIMATE, attached to the trace before it is persisted (the model
|
|
2394
903
|
// id is authoritative here — provider.version). Kept at the lab boundary so the pure loop
|
|
@@ -2419,24 +928,13 @@ export async function runCuaLane(spec, deps) {
|
|
|
2419
928
|
spec,
|
|
2420
929
|
...(session ? { session } : {}),
|
|
2421
930
|
...(sessionError === undefined ? {} : { sessionError }),
|
|
2422
|
-
...(
|
|
2423
|
-
...(desktopDurationMs === undefined ? {} : { desktopDurationMs }),
|
|
2424
|
-
...(desktopResources === undefined ? {} : { desktopResources }),
|
|
2425
|
-
killed,
|
|
2426
|
-
streamUrlPresent: streamUrl !== undefined,
|
|
931
|
+
...desktopLane.snapshot(),
|
|
2427
932
|
screenshots,
|
|
2428
|
-
...(subjectCommit === undefined ? {} : { subjectCommit }),
|
|
2429
|
-
...(desktopBrowser === undefined ? {} : { desktopBrowser }),
|
|
2430
|
-
desktopGeometry,
|
|
2431
|
-
stateStepRecords,
|
|
2432
|
-
phaseRecords,
|
|
2433
933
|
warnings,
|
|
2434
934
|
noEngagement,
|
|
2435
935
|
selfReportedBlocker,
|
|
2436
936
|
reportedFriction,
|
|
2437
937
|
harnessError,
|
|
2438
|
-
...(failureCode === undefined ? {} : { failureCode }),
|
|
2439
|
-
...(commsArtifactPath === undefined ? {} : { commsArtifactPath })
|
|
2440
938
|
};
|
|
2441
939
|
}
|
|
2442
940
|
/** Run the single IN-PROCESS lane (a custom executor + provider; NO E2B). Always one lane. */
|
|
@@ -3763,300 +2261,6 @@ function buildSingleLaneBundle(args) {
|
|
|
3763
2261
|
phaseEvents: outcome?.phaseRecords ?? []
|
|
3764
2262
|
});
|
|
3765
2263
|
}
|
|
3766
|
-
/**
|
|
3767
|
-
* Shared post-populate provisioning pipeline (clone AND local-tree routes): (install) ->
|
|
3768
|
-
* state(before-build) -> (build) -> state(before-start) -> detached start -> readiness probe ->
|
|
3769
|
-
* state(after-ready). Both provisioning routes populate SUBJECT_DIR by different means (git
|
|
3770
|
-
* clone vs. upload+extract) and then run this identical pipeline unchanged.
|
|
3771
|
-
*
|
|
3772
|
-
* State steps run through the same detached primitive as serve steps (author-trusted, the
|
|
3773
|
-
* "serve commands are author-trusted" corollary) under the reserved `subject-state-<name>`
|
|
3774
|
-
* label prefix, so a step name can never collide with subject-clone/subject-extract/install/
|
|
3775
|
-
* build/start. after-ready steps complete BEFORE the caller opens the browser: the actor never
|
|
3776
|
-
* drives a half-seeded subject and seeding never eats the session budget.
|
|
3777
|
-
*/
|
|
3778
|
-
async function runSubjectServePipeline(desktop, args) {
|
|
3779
|
-
const timers = {
|
|
3780
|
-
...(args.now === undefined ? {} : { now: args.now }),
|
|
3781
|
-
...(args.sleep === undefined ? {} : { sleep: args.sleep })
|
|
3782
|
-
};
|
|
3783
|
-
const now = args.now ?? Date.now;
|
|
3784
|
-
const refresh = args.onPhaseComplete ?? (() => Promise.resolve());
|
|
3785
|
-
const stateSteps = args.state?.seed ?? [];
|
|
3786
|
-
const runStateSteps = async (when) => {
|
|
3787
|
-
const steps = stateSteps.filter((step) => (step.when ?? "before-start") === when);
|
|
3788
|
-
if (steps.length === 0) {
|
|
3789
|
-
// No declared steps for this group: no boundary to report (avoids empty-group noise on
|
|
3790
|
-
// every run, since before-build/before-start/after-ready are always called).
|
|
3791
|
-
return;
|
|
3792
|
-
}
|
|
3793
|
-
const groupStartedAt = now();
|
|
3794
|
-
emitPhaseStarted(args.onPhase, now, `state.${when}`, `running subject state seed steps (${when})`);
|
|
3795
|
-
for (const step of steps) {
|
|
3796
|
-
const stepTimeoutMs = step.timeoutMs ?? DEFAULT_STATE_STEP_TIMEOUT_MS;
|
|
3797
|
-
const startedAt = now();
|
|
3798
|
-
const result = await runDetachedStep(desktop, {
|
|
3799
|
-
name: `subject-state-${step.name}`,
|
|
3800
|
-
command: step.command,
|
|
3801
|
-
cwd: SUBJECT_DIR,
|
|
3802
|
-
timeoutMs: stepTimeoutMs,
|
|
3803
|
-
requestTimeoutMs: args.requestTimeoutMs,
|
|
3804
|
-
...timers
|
|
3805
|
-
});
|
|
3806
|
-
args.onStateStep?.({
|
|
3807
|
-
name: step.name,
|
|
3808
|
-
when,
|
|
3809
|
-
// Digest only (sha256-16): the command text never persists: the lab YAML in the
|
|
3810
|
-
// consumer's repo is the plaintext source of truth.
|
|
3811
|
-
commandDigest: commandDigestOf(step.command),
|
|
3812
|
-
ok: result.ok,
|
|
3813
|
-
...(result.exitCode === undefined ? {} : { exitCode: result.exitCode }),
|
|
3814
|
-
...(result.timedOut ? { timedOut: true } : {}),
|
|
3815
|
-
durationMs: Math.max(0, now() - startedAt)
|
|
3816
|
-
});
|
|
3817
|
-
if (!result.ok) {
|
|
3818
|
-
emitPhaseCompleted(args.onPhase, now, groupStartedAt, `state.${when}`, false, `subject state seed steps failed (${when})`);
|
|
3819
|
-
// Fail closed with the existing scrub-before-truncate tail chain: literal scrub of
|
|
3820
|
-
// every provisioned value PRE-truncation, then pattern redaction + cap in tailOf.
|
|
3821
|
-
throw new Error(`subject state step "${step.name}" ${result.timedOut ? `timed out after ${stepTimeoutMs}ms` : `failed (exit ${result.exitCode})`}: ${tailOf(args.scrub(result.logTail))}`);
|
|
3822
|
-
}
|
|
3823
|
-
}
|
|
3824
|
-
emitPhaseCompleted(args.onPhase, now, groupStartedAt, `state.${when}`, true, `subject state seed steps complete (${when})`);
|
|
3825
|
-
};
|
|
3826
|
-
// Provide the runtime the pipeline needs before running it (#371). The stock desktop template
|
|
3827
|
-
// ships python3 and curl but no Node, so an `npm install` here used to die at exit 127 after the
|
|
3828
|
-
// sandbox was already paid for. Probe-first, so a template that ships its own Node pays nothing.
|
|
3829
|
-
const serveCommands = [args.serve.install, args.serve.build, args.serve.start];
|
|
3830
|
-
if (needsNodeRuntime(serveCommands)) {
|
|
3831
|
-
const runtimeStartedAt = now();
|
|
3832
|
-
emitPhaseStarted(args.onPhase, now, "runtime", "providing the Node runtime the serve pipeline needs");
|
|
3833
|
-
const bootstrap = await runProvisioningStepWithOneRetry(desktop, {
|
|
3834
|
-
name: "subject-runtime-node",
|
|
3835
|
-
command: nodeBootstrapCommand(),
|
|
3836
|
-
cwd: SUBJECT_DIR,
|
|
3837
|
-
timeoutMs: args.serve.installTimeoutMs ?? INSTALL_TIMEOUT_MS,
|
|
3838
|
-
requestTimeoutMs: args.requestTimeoutMs,
|
|
3839
|
-
timers,
|
|
3840
|
-
retryPhase: "runtime-retry",
|
|
3841
|
-
retryMessage: "Node runtime bootstrap",
|
|
3842
|
-
onPhase: args.onPhase,
|
|
3843
|
-
now
|
|
3844
|
-
});
|
|
3845
|
-
let ok = bootstrap.ok;
|
|
3846
|
-
const corepack = ok ? corepackCommandFor(serveCommands) : undefined;
|
|
3847
|
-
if (corepack) {
|
|
3848
|
-
const pm = await runDetachedStep(desktop, {
|
|
3849
|
-
name: "subject-runtime-pm",
|
|
3850
|
-
command: corepack,
|
|
3851
|
-
cwd: SUBJECT_DIR,
|
|
3852
|
-
timeoutMs: args.serve.installTimeoutMs ?? INSTALL_TIMEOUT_MS,
|
|
3853
|
-
requestTimeoutMs: args.requestTimeoutMs,
|
|
3854
|
-
...timers
|
|
3855
|
-
});
|
|
3856
|
-
ok = pm.ok;
|
|
3857
|
-
}
|
|
3858
|
-
emitPhaseCompleted(args.onPhase, now, runtimeStartedAt, "runtime", ok, ok ? "Node runtime ready" : "could not provide a Node runtime");
|
|
3859
|
-
if (!ok) {
|
|
3860
|
-
throw new Error(`the subject's serve pipeline needs a Node runtime and this desktop template has none, and bootstrapping one failed${bootstrap.attempts === 2 ? " twice" : ""}: ${tailOf(args.scrub(bootstrap.logTail))}. Use execution.desktop.template with an image that ships Node, or change serve.install to a runtime the template provides.`);
|
|
3861
|
-
}
|
|
3862
|
-
}
|
|
3863
|
-
if (args.serve.install) {
|
|
3864
|
-
const installStartedAt = now();
|
|
3865
|
-
emitPhaseStarted(args.onPhase, now, "install", "installing subject dependencies");
|
|
3866
|
-
const install = await runProvisioningStepWithOneRetry(desktop, {
|
|
3867
|
-
name: "subject-install",
|
|
3868
|
-
command: args.serve.install,
|
|
3869
|
-
cwd: SUBJECT_DIR,
|
|
3870
|
-
timeoutMs: args.serve.installTimeoutMs ?? INSTALL_TIMEOUT_MS,
|
|
3871
|
-
requestTimeoutMs: args.requestTimeoutMs,
|
|
3872
|
-
timers,
|
|
3873
|
-
retryPhase: "install-retry",
|
|
3874
|
-
retryMessage: "subject install",
|
|
3875
|
-
onPhase: args.onPhase,
|
|
3876
|
-
now
|
|
3877
|
-
});
|
|
3878
|
-
emitPhaseCompleted(args.onPhase, now, installStartedAt, "install", install.ok, install.ok
|
|
3879
|
-
? install.attempts === 2
|
|
3880
|
-
? "subject dependencies installed (on the second attempt)"
|
|
3881
|
-
: "subject dependencies installed"
|
|
3882
|
-
: install.attempts === 2
|
|
3883
|
-
? "subject install failed twice"
|
|
3884
|
-
: "subject install failed");
|
|
3885
|
-
if (!install.ok) {
|
|
3886
|
-
// Lead with the line a person can act on; npm's own trace follows it (#602).
|
|
3887
|
-
const headline = install.timedOut
|
|
3888
|
-
? `subject install timed out after ${args.serve.installTimeoutMs ?? INSTALL_TIMEOUT_MS}ms`
|
|
3889
|
-
: install.attempts === 2
|
|
3890
|
-
? `subject install failed twice (exit ${install.firstExitCode ?? "null"}, then exit ${install.exitCode ?? "null"}); the sandbox could not complete serve.install`
|
|
3891
|
-
: `subject install failed (exit ${install.exitCode ?? "null"})`;
|
|
3892
|
-
throw new Error(`${headline}: ${tailOf(args.scrub(install.logTail))}`);
|
|
3893
|
-
}
|
|
3894
|
-
await refresh();
|
|
3895
|
-
}
|
|
3896
|
-
// before-build: after install, before build (builds that read seeded state, e.g. SSG).
|
|
3897
|
-
// When no build is declared this simply precedes start: equivalent to before-start.
|
|
3898
|
-
await runStateSteps("before-build");
|
|
3899
|
-
await refresh();
|
|
3900
|
-
if (args.serve.build) {
|
|
3901
|
-
const buildStartedAt = now();
|
|
3902
|
-
emitPhaseStarted(args.onPhase, now, "build", "building subject");
|
|
3903
|
-
const build = await runDetachedStep(desktop, {
|
|
3904
|
-
name: "subject-build",
|
|
3905
|
-
command: args.serve.build,
|
|
3906
|
-
cwd: SUBJECT_DIR,
|
|
3907
|
-
timeoutMs: args.serve.buildTimeoutMs ?? BUILD_TIMEOUT_MS,
|
|
3908
|
-
requestTimeoutMs: args.requestTimeoutMs,
|
|
3909
|
-
...timers
|
|
3910
|
-
});
|
|
3911
|
-
emitPhaseCompleted(args.onPhase, now, buildStartedAt, "build", build.ok, build.ok ? "subject build complete" : "subject build failed");
|
|
3912
|
-
if (!build.ok) {
|
|
3913
|
-
throw new Error(`subject build ${build.timedOut ? "timed out" : `failed (exit ${build.exitCode})`}: ${tailOf(args.scrub(build.logTail))}`);
|
|
3914
|
-
}
|
|
3915
|
-
await refresh();
|
|
3916
|
-
}
|
|
3917
|
-
// before-start (the default phase): migrations, SQL/file fixtures, an in-sandbox DB server
|
|
3918
|
-
// (`sudo service postgresql start && pg_isready` is a bounded step; the daemon it forks is
|
|
3919
|
-
// reclaimed by the sandbox lifecycle like everything else).
|
|
3920
|
-
await runStateSteps("before-start");
|
|
3921
|
-
await refresh();
|
|
3922
|
-
await startDetachedProcess(desktop, {
|
|
3923
|
-
name: "subject-start",
|
|
3924
|
-
command: args.serve.start,
|
|
3925
|
-
cwd: SUBJECT_DIR,
|
|
3926
|
-
requestTimeoutMs: args.requestTimeoutMs
|
|
3927
|
-
});
|
|
3928
|
-
// Fire-and-forget: startDetachedProcess never waits for the long-lived server to exit, so
|
|
3929
|
-
// there is no matching completed event here (no ok/durationMs to report yet); readiness is
|
|
3930
|
-
// the next boundary.
|
|
3931
|
-
args.onPhase?.({ at: isoNow(now), type: "cua-lab.subject.serve.started", message: "subject server launched (detached)" });
|
|
3932
|
-
const readyStartedAt = now();
|
|
3933
|
-
emitPhaseStarted(args.onPhase, now, "ready", "waiting for subject to become ready");
|
|
3934
|
-
const ready = await probeUrl(desktop, args.serve.url, {
|
|
3935
|
-
timeoutMs: args.serve.readyTimeoutMs ?? DEFAULT_READY_TIMEOUT_MS,
|
|
3936
|
-
requestTimeoutMs: args.requestTimeoutMs,
|
|
3937
|
-
...timers
|
|
3938
|
-
});
|
|
3939
|
-
emitPhaseCompleted(args.onPhase, now, readyStartedAt, "ready", ready, ready ? "subject is ready" : "subject did not become ready in time");
|
|
3940
|
-
if (!ready) {
|
|
3941
|
-
const startLog = await readDetachedLog(desktop, "subject-start", args.requestTimeoutMs).catch(() => "");
|
|
3942
|
-
throw new Error(`subject did not answer at ${args.serve.url} within ${args.serve.readyTimeoutMs ?? DEFAULT_READY_TIMEOUT_MS}ms; server log tail: ${tailOf(args.scrub(startLog))}`);
|
|
3943
|
-
}
|
|
3944
|
-
// after-ready: fixture loading through the RUNNING app (loopback curl from in-sandbox:
|
|
3945
|
-
// steps are author-trusted provisioning, not actors, so no new URL policy surface). These
|
|
3946
|
-
// complete before the caller opens the browser and the session timer starts.
|
|
3947
|
-
await runStateSteps("after-ready");
|
|
3948
|
-
await refresh();
|
|
3949
|
-
}
|
|
3950
|
-
/**
|
|
3951
|
-
* Provision a clone subject inside the sandbox: clone → the shared serve pipeline
|
|
3952
|
-
* (install → state(before-build) → build → state(before-start) → start → readiness
|
|
3953
|
-
* probe → state(after-ready)). Returns the latest subject HEAD after successful
|
|
3954
|
-
* provisioning. Throws (with a capped log tail for the caller to redact) on any failing step:
|
|
3955
|
-
* the lab persists that as a failed-evidence bundle.
|
|
3956
|
-
*
|
|
3957
|
-
* Auth: when GITHUB_TOKEN is among the declared subject env names, the clone authenticates
|
|
3958
|
-
* via an Authorization header computed IN-SANDBOX from the provisioned env: the token never
|
|
3959
|
-
* appears in the script text, the process argv beyond the transient git call, the clone URL,
|
|
3960
|
-
* or .git/config.
|
|
3961
|
-
*/
|
|
3962
|
-
export async function provisionCloneSubject(desktop, args) {
|
|
3963
|
-
const timers = {
|
|
3964
|
-
...(args.now === undefined ? {} : { now: args.now }),
|
|
3965
|
-
...(args.sleep === undefined ? {} : { sleep: args.sleep })
|
|
3966
|
-
};
|
|
3967
|
-
const now = args.now ?? Date.now;
|
|
3968
|
-
let latestCommit;
|
|
3969
|
-
const refreshCommit = async () => {
|
|
3970
|
-
const head = await desktop.commands.run(`git -C ${SUBJECT_DIR} rev-parse HEAD 2>/dev/null || true`, { requestTimeoutMs: args.requestTimeoutMs });
|
|
3971
|
-
const commit = (head.stdout ?? "").trim() || undefined;
|
|
3972
|
-
if (commit) {
|
|
3973
|
-
latestCommit = commit;
|
|
3974
|
-
args.onCommit?.(commit);
|
|
3975
|
-
}
|
|
3976
|
-
};
|
|
3977
|
-
const cloneCommand = args.hasGithubToken
|
|
3978
|
-
? `auth=$(printf 'x-access-token:%s' "$GITHUB_TOKEN" | base64 -w0) && git -c http.extraHeader="Authorization: Basic $auth" clone --depth ${args.depth} https://github.com/${args.repo}.git ${SUBJECT_DIR}`
|
|
3979
|
-
: `git clone --depth ${args.depth} https://github.com/${args.repo}.git ${SUBJECT_DIR}`;
|
|
3980
|
-
const cloneStartedAt = now();
|
|
3981
|
-
emitPhaseStarted(args.onPhase, now, "clone", "cloning subject repository");
|
|
3982
|
-
const clone = await runDetachedStep(desktop, {
|
|
3983
|
-
name: "subject-clone",
|
|
3984
|
-
command: cloneCommand,
|
|
3985
|
-
timeoutMs: CLONE_TIMEOUT_MS,
|
|
3986
|
-
requestTimeoutMs: args.requestTimeoutMs,
|
|
3987
|
-
...timers
|
|
3988
|
-
});
|
|
3989
|
-
emitPhaseCompleted(args.onPhase, now, cloneStartedAt, "clone", clone.ok, clone.ok ? "subject repository cloned" : "subject clone failed");
|
|
3990
|
-
if (!clone.ok) {
|
|
3991
|
-
throw new Error(`subject clone ${clone.timedOut ? "timed out" : `failed (exit ${clone.exitCode})`}: ${tailOf(args.scrub(clone.logTail))}`);
|
|
3992
|
-
}
|
|
3993
|
-
await refreshCommit();
|
|
3994
|
-
await runSubjectServePipeline(desktop, {
|
|
3995
|
-
serve: args.serve,
|
|
3996
|
-
...(args.state === undefined ? {} : { state: args.state }),
|
|
3997
|
-
requestTimeoutMs: args.requestTimeoutMs,
|
|
3998
|
-
scrub: args.scrub,
|
|
3999
|
-
...(args.onStateStep === undefined ? {} : { onStateStep: args.onStateStep }),
|
|
4000
|
-
...(args.onPhase === undefined ? {} : { onPhase: args.onPhase }),
|
|
4001
|
-
onPhaseComplete: refreshCommit,
|
|
4002
|
-
...timers
|
|
4003
|
-
});
|
|
4004
|
-
return latestCommit;
|
|
4005
|
-
}
|
|
4006
|
-
/**
|
|
4007
|
-
* Provision a local-tree subject inside the sandbox: upload the once-per-run packed archive
|
|
4008
|
-
* (identical bytes across every fan-out lane) → extract it into SUBJECT_DIR → the
|
|
4009
|
-
* same shared serve pipeline provisionCloneSubject uses. Unlike the clone route there is no
|
|
4010
|
-
* in-sandbox git refresh: the archive excludes .git entirely (see source-archive.ts), so
|
|
4011
|
-
* subject identity is the host-side LocalTreeArchive captured at pack time, never anything
|
|
4012
|
-
* resolved in-sandbox.
|
|
4013
|
-
*/
|
|
4014
|
-
export async function provisionLocalTreeSubject(desktop, args) {
|
|
4015
|
-
const timers = {
|
|
4016
|
-
...(args.now === undefined ? {} : { now: args.now }),
|
|
4017
|
-
...(args.sleep === undefined ? {} : { sleep: args.sleep })
|
|
4018
|
-
};
|
|
4019
|
-
const now = args.now ?? Date.now;
|
|
4020
|
-
const uploadStartedAt = now();
|
|
4021
|
-
emitPhaseStarted(args.onPhase, now, "upload", "uploading packed local-tree archive");
|
|
4022
|
-
try {
|
|
4023
|
-
await withOneRetryOnTransientE2BError(() => desktop.files.write(LOCAL_TREE_REMOTE_ARCHIVE_PATH, args.archiveBuffer, {
|
|
4024
|
-
requestTimeoutMs: args.requestTimeoutMs,
|
|
4025
|
-
useOctetStream: true
|
|
4026
|
-
}), {
|
|
4027
|
-
onRetry: (reason) => emitPhaseStarted(args.onPhase, now, "upload-retry", `local-tree archive upload retried once (${tailOf(args.scrub(reason))})`),
|
|
4028
|
-
...(args.sleep === undefined ? {} : { sleep: args.sleep })
|
|
4029
|
-
});
|
|
4030
|
-
}
|
|
4031
|
-
catch (error) {
|
|
4032
|
-
emitPhaseCompleted(args.onPhase, now, uploadStartedAt, "upload", false, "local-tree archive upload failed");
|
|
4033
|
-
throw new Error(`subject-upload failed: ${tailOf(args.scrub(toErrorMessage(error)))}`);
|
|
4034
|
-
}
|
|
4035
|
-
emitPhaseCompleted(args.onPhase, now, uploadStartedAt, "upload", true, "local-tree archive uploaded");
|
|
4036
|
-
const extractCommand = `rm -rf ${SUBJECT_DIR} && mkdir -p ${SUBJECT_DIR} && tar -xzf ${LOCAL_TREE_REMOTE_ARCHIVE_PATH} -C ${SUBJECT_DIR} && rm -f ${LOCAL_TREE_REMOTE_ARCHIVE_PATH}`;
|
|
4037
|
-
const extractStartedAt = now();
|
|
4038
|
-
emitPhaseStarted(args.onPhase, now, "extract", "extracting local-tree archive");
|
|
4039
|
-
const extract = await runDetachedStep(desktop, {
|
|
4040
|
-
name: "subject-extract",
|
|
4041
|
-
command: extractCommand,
|
|
4042
|
-
timeoutMs: CLONE_TIMEOUT_MS,
|
|
4043
|
-
requestTimeoutMs: args.requestTimeoutMs,
|
|
4044
|
-
...timers
|
|
4045
|
-
});
|
|
4046
|
-
emitPhaseCompleted(args.onPhase, now, extractStartedAt, "extract", extract.ok, extract.ok ? "local-tree archive extracted" : "local-tree archive extraction failed");
|
|
4047
|
-
if (!extract.ok) {
|
|
4048
|
-
throw new Error(`subject extract ${extract.timedOut ? "timed out" : `failed (exit ${extract.exitCode})`}: ${tailOf(args.scrub(extract.logTail))}`);
|
|
4049
|
-
}
|
|
4050
|
-
await runSubjectServePipeline(desktop, {
|
|
4051
|
-
serve: args.serve,
|
|
4052
|
-
...(args.state === undefined ? {} : { state: args.state }),
|
|
4053
|
-
requestTimeoutMs: args.requestTimeoutMs,
|
|
4054
|
-
scrub: args.scrub,
|
|
4055
|
-
...(args.onStateStep === undefined ? {} : { onStateStep: args.onStateStep }),
|
|
4056
|
-
...(args.onPhase === undefined ? {} : { onPhase: args.onPhase }),
|
|
4057
|
-
...timers
|
|
4058
|
-
});
|
|
4059
|
-
}
|
|
4060
2264
|
/**
|
|
4061
2265
|
* Default local-tree packing implementation: createLocalTreeArchive(root, opts) on the host,
|
|
4062
2266
|
* then a single read of the produced archive file into an ArrayBuffer for upload. The DI seam
|
|
@@ -4076,10 +2280,6 @@ export async function defaultPackLocalTree(args) {
|
|
|
4076
2280
|
await rm(path.dirname(archive.archivePath), { recursive: true, force: true }).catch(() => undefined);
|
|
4077
2281
|
return { archive, buffer };
|
|
4078
2282
|
}
|
|
4079
|
-
/** sha256 hex of the exact command string, first 16 chars (the promptDigest convention). */
|
|
4080
|
-
export function commandDigestOf(command) {
|
|
4081
|
-
return digestText(command, 16);
|
|
4082
|
-
}
|
|
4083
2283
|
/**
|
|
4084
2284
|
* Resolve the bundle's state marker from the declaration and what actually ran.
|
|
4085
2285
|
* Precedence: external declared → "unpinned" (seed records, if any, stay attached — a
|
|
@@ -4135,10 +2335,6 @@ function describeSubjectState(state, dryRun) {
|
|
|
4135
2335
|
return "external-public (operator-declared, operator-owned public deployment; neither provisioned nor seeded)";
|
|
4136
2336
|
}
|
|
4137
2337
|
}
|
|
4138
|
-
// The in-sandbox `tail -c` upstream is a fundamental log-tail limit we cannot redact past.
|
|
4139
|
-
function tailOf(log) {
|
|
4140
|
-
return redactedTail(log, ERROR_TAIL_CHARS);
|
|
4141
|
-
}
|
|
4142
2338
|
export function buildCuaCostSummary(args) {
|
|
4143
2339
|
const breakdown = [];
|
|
4144
2340
|
let sumInput = 0;
|