humanish 0.96.1 → 0.98.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +86 -79
- package/CONTRIBUTING.md +7 -2
- package/README.md +11 -2
- package/dist/actor-contract.d.ts +35 -1
- package/dist/actor-contract.js +38 -0
- package/dist/actor-contract.js.map +1 -1
- package/dist/adapter-extension.js +1 -0
- package/dist/adapter-extension.js.map +1 -1
- package/dist/automatic-analysis-config.d.ts +15 -5
- package/dist/automatic-analysis-config.js +25 -4
- package/dist/automatic-analysis-config.js.map +1 -1
- package/dist/automatic-study-analysis.js +4 -2
- package/dist/automatic-study-analysis.js.map +1 -1
- package/dist/browser-control-client.d.ts +14 -0
- package/dist/browser-control-client.js +134 -0
- package/dist/browser-control-client.js.map +1 -0
- package/dist/browser-control-dispatcher.d.ts +14 -0
- package/dist/browser-control-dispatcher.js +109 -0
- package/dist/browser-control-dispatcher.js.map +1 -0
- package/dist/browser-control-protocol.d.ts +371 -0
- package/dist/browser-control-protocol.js +155 -0
- package/dist/browser-control-protocol.js.map +1 -0
- package/dist/browser-control-transport.d.ts +24 -0
- package/dist/browser-control-transport.js +156 -0
- package/dist/browser-control-transport.js.map +1 -0
- package/dist/comms-lease-store.d.ts +1 -0
- package/dist/comms-lease-store.js +9 -3
- package/dist/comms-lease-store.js.map +1 -1
- package/dist/computer-use-actor.d.ts +2 -2
- package/dist/computer-use-actor.js +6 -1
- package/dist/computer-use-actor.js.map +1 -1
- package/dist/computer-use.d.ts +23 -1
- package/dist/computer-use.js +253 -70
- package/dist/computer-use.js.map +1 -1
- package/dist/cua-actor-lab.d.ts +24 -289
- package/dist/cua-actor-lab.js +203 -1983
- package/dist/cua-actor-lab.js.map +1 -1
- package/dist/cua-desktop-lane.d.ts +35 -0
- package/dist/cua-desktop-lane.js +13 -0
- package/dist/cua-desktop-lane.js.map +1 -0
- package/dist/cua-executor-error.d.ts +31 -0
- package/dist/cua-executor-error.js +48 -0
- package/dist/cua-executor-error.js.map +1 -0
- package/dist/cua-provider-error.d.ts +12 -0
- package/dist/cua-provider-error.js +28 -0
- package/dist/cua-provider-error.js.map +1 -0
- package/dist/desktop-session.d.ts +41 -0
- package/dist/desktop-session.js +46 -0
- package/dist/desktop-session.js.map +1 -0
- package/dist/doctor-lab.d.ts +8 -1
- package/dist/doctor-lab.js +40 -8
- package/dist/doctor-lab.js.map +1 -1
- package/dist/e2b-cua-desktop.d.ts +3 -0
- package/dist/e2b-cua-desktop.js +675 -0
- package/dist/e2b-cua-desktop.js.map +1 -0
- package/dist/e2b-cua-provisioning.d.ts +311 -0
- package/dist/e2b-cua-provisioning.js +1213 -0
- package/dist/e2b-cua-provisioning.js.map +1 -0
- package/dist/e2b-desktop-executor.d.ts +1 -24
- package/dist/e2b-desktop-executor.js +2 -127
- package/dist/e2b-desktop-executor.js.map +1 -1
- package/dist/e2b-desktop-session.d.ts +7 -0
- package/dist/e2b-desktop-session.js +29 -0
- package/dist/e2b-desktop-session.js.map +1 -0
- package/dist/e2b-terminal-lab.js +1 -0
- package/dist/e2b-terminal-lab.js.map +1 -1
- package/dist/frame-signature.d.ts +24 -0
- package/dist/frame-signature.js +128 -0
- package/dist/frame-signature.js.map +1 -0
- package/dist/guest-bootstrap.d.ts +43 -0
- package/dist/guest-bootstrap.js +240 -0
- package/dist/guest-bootstrap.js.map +1 -0
- package/dist/guest-browser-tools.d.ts +8 -0
- package/dist/guest-browser-tools.js +66 -0
- package/dist/guest-browser-tools.js.map +1 -0
- package/dist/guest-chromium-text.d.ts +27 -0
- package/dist/guest-chromium-text.js +281 -0
- package/dist/guest-chromium-text.js.map +1 -0
- package/dist/guest-desktop-executor.d.ts +22 -0
- package/dist/guest-desktop-executor.js +177 -0
- package/dist/guest-desktop-executor.js.map +1 -0
- package/dist/guest-desktop-native.d.ts +14 -0
- package/dist/guest-desktop-native.js +131 -0
- package/dist/guest-desktop-native.js.map +1 -0
- package/dist/guest-runtime-desktop.d.ts +35 -0
- package/dist/guest-runtime-desktop.js +231 -0
- package/dist/guest-runtime-desktop.js.map +1 -0
- package/dist/guest-runtime-main.d.ts +1 -0
- package/dist/guest-runtime-main.js +31 -0
- package/dist/guest-runtime-main.js.map +1 -0
- package/dist/guest-runtime-revision.d.ts +1 -0
- package/dist/guest-runtime-revision.js +3 -0
- package/dist/guest-runtime-revision.js.map +1 -0
- package/dist/guest-runtime.d.ts +25 -0
- package/dist/guest-runtime.js +96 -0
- package/dist/guest-runtime.js.map +1 -0
- package/dist/index.d.ts +1 -1
- package/dist/lab-config.js +10 -3
- package/dist/lab-config.js.map +1 -1
- package/dist/lab-engine.js +6 -0
- package/dist/lab-engine.js.map +1 -1
- package/dist/lab-summary.d.ts +5 -1
- package/dist/lab-summary.js +5 -0
- package/dist/lab-summary.js.map +1 -1
- package/dist/local-agent-cli.js +1 -1
- package/dist/local-agent-cli.js.map +1 -1
- package/dist/local-firecracker-desktop.d.ts +13 -0
- package/dist/local-firecracker-desktop.js +150 -0
- package/dist/local-firecracker-desktop.js.map +1 -0
- package/dist/local-firecracker-study.d.ts +9 -0
- package/dist/local-firecracker-study.js +93 -0
- package/dist/local-firecracker-study.js.map +1 -0
- package/dist/local-runtime-config.d.ts +6 -0
- package/dist/local-runtime-config.js +56 -0
- package/dist/local-runtime-config.js.map +1 -0
- package/dist/local-runtime-release.d.ts +3 -0
- package/dist/local-runtime-release.js +8 -0
- package/dist/local-runtime-release.js.map +1 -0
- package/dist/local-runtime.d.ts +25 -0
- package/dist/local-runtime.js +113 -0
- package/dist/local-runtime.js.map +1 -0
- package/dist/observer-app.html +4 -4
- package/dist/pricing.d.ts +22 -1
- package/dist/pricing.js +22 -0
- package/dist/pricing.js.map +1 -1
- package/dist/program.js +50 -9
- package/dist/program.js.map +1 -1
- package/dist/restricted-codex-analysis.d.ts +15 -0
- package/dist/restricted-codex-analysis.js +13 -0
- package/dist/restricted-codex-analysis.js.map +1 -0
- package/dist/restricted-codex-participant-policy.d.ts +39 -0
- package/dist/restricted-codex-participant-policy.js +69 -0
- package/dist/restricted-codex-participant-policy.js.map +1 -0
- package/dist/restricted-codex-participant-run.d.ts +20 -0
- package/dist/restricted-codex-participant-run.js +78 -0
- package/dist/restricted-codex-participant-run.js.map +1 -0
- package/dist/restricted-codex-participant.d.ts +14 -0
- package/dist/restricted-codex-participant.js +178 -0
- package/dist/restricted-codex-participant.js.map +1 -0
- package/dist/restricted-codex-policy.d.ts +56 -0
- package/dist/restricted-codex-policy.js +151 -0
- package/dist/restricted-codex-policy.js.map +1 -0
- package/dist/restricted-codex-session.d.ts +19 -0
- package/dist/restricted-codex-session.js +413 -0
- package/dist/restricted-codex-session.js.map +1 -0
- package/dist/restricted-codex-transport.d.ts +58 -0
- package/dist/restricted-codex-transport.js +233 -0
- package/dist/restricted-codex-transport.js.map +1 -0
- package/dist/run-detail.js +4 -2
- package/dist/run-detail.js.map +1 -1
- package/dist/run.d.ts +12 -5
- package/dist/run.js +17 -1
- package/dist/run.js.map +1 -1
- package/dist/shared-world-lab.js +2 -2
- package/dist/shared-world-lab.js.map +1 -1
- package/dist/study-analysis-codex-config.d.ts +11 -0
- package/dist/study-analysis-codex-config.js +34 -0
- package/dist/study-analysis-codex-config.js.map +1 -0
- package/dist/study-analysis-engine.d.ts +6 -3
- package/dist/study-analysis-engine.js +19 -10
- package/dist/study-analysis-engine.js.map +1 -1
- package/dist/study-analysis-job.d.ts +3 -2
- package/dist/study-analysis-job.js +1 -1
- package/dist/study-analysis-job.js.map +1 -1
- package/dist/study-analysis-provider.d.ts +4 -2
- package/dist/study-analysis-provider.js +1 -1
- package/dist/study-analysis-provider.js.map +1 -1
- package/dist/study-analysis-service.d.ts +3 -0
- package/dist/study-analysis-service.js +24 -6
- package/dist/study-analysis-service.js.map +1 -1
- package/dist/study-analysis-validation.d.ts +43 -19
- package/dist/study-analysis-validation.js +23 -10
- package/dist/study-analysis-validation.js.map +1 -1
- package/dist/study-analysis.d.ts +27 -2
- package/dist/study-costs.js +6 -0
- package/dist/study-costs.js.map +1 -1
- package/dist/tui-app.js +102 -102
- package/docs/architecture/browser-control.md +117 -0
- package/docs/architecture/desktop-sessions.md +80 -0
- package/docs/architecture/guest-desktop.md +87 -0
- package/docs/architecture/local-browser-runtime.md +100 -0
- package/docs/architecture/restricted-codex-analysis.md +91 -0
- package/docs/architecture/runtime-broker-core.md +30 -0
- package/docs/contracts/schemas.md +1 -1
- package/docs/contracts/study-analysis.md +44 -2
- package/docs/goals/current.md +25 -9
- package/docs/product/automatic-analysis.md +24 -3
- package/docs/product/open-source-install-experience.md +7 -0
- package/docs/ramp/README.md +30 -13
- package/docs/release/0.97.0-codex-account-analysis.md +45 -0
- package/docs/release/0.98.0-local-browser-studies.md +31 -0
- package/package.json +4 -2
- package/skills/humanish/SKILL.md +39 -0
package/dist/cua-actor-lab.js
CHANGED
|
@@ -1,7 +1,12 @@
|
|
|
1
|
-
|
|
1
|
+
export { inboxRecipientFor, laneHasInboxRecipient } from "./cua-desktop-lane.js";
|
|
2
|
+
export { CUA_ACTOR_LAB_PROVIDER_METADATA, DEFAULT_MOBILE_USER_AGENT, SANDBOX_CAMERA_PATH, SANDBOX_MEDIA_DIR, SUBJECT_DIR, SYNTHETIC_CAMERA_COMMAND, applyMobileEmulation, buildFillDesktopWindowCommand, captureDesktopBrowserGeometry, commandDigestOf, declaredScreenForRender, desktopBrowserFamily, inspectDesktopScreenGeometry, makeChromeBrowserStateObserver, makeChromeDesktopGeometryObserver, parseXwininfoGeometry, prepareDesktopMedia, provisionCloneSubject, provisionLocalTreeSubject } from "./e2b-cua-provisioning.js";
|
|
2
3
|
import { prepareReceivingRun, receivingPublication } from "./comms-receiving-runtime.js";
|
|
3
|
-
import {
|
|
4
|
+
import { laneHasInboxRecipient } from "./cua-desktop-lane.js";
|
|
5
|
+
import { createE2BCuaDesktopLane } from "./e2b-cua-desktop.js";
|
|
6
|
+
import { isLocalBrowserLab, LOCAL_BROWSER_LIFETIME_MS } from "./local-runtime-config.js";
|
|
7
|
+
import { DEFAULT_STATE_STEP_TIMEOUT_MS, commandDigestOf, declaredScreenForRender } from "./e2b-cua-provisioning.js";
|
|
4
8
|
import { receivingEmailValidationReason } from "./lab-config.js";
|
|
9
|
+
import { withTransientCommsSecrets } from "./run-narration-secrets.js";
|
|
5
10
|
// The computer-use lab backend: a subject (an app-url the caller provisioned, or a repo the
|
|
6
11
|
// lab clones AND serves in-sandbox) driven by a REGISTRY-RESOLVED computer-use actor inside a
|
|
7
12
|
// hosted E2B desktop. This is the path that makes `actors[].type` load-bearing — the
|
|
@@ -24,52 +29,47 @@ import { receivingEmailValidationReason } from "./lab-config.js";
|
|
|
24
29
|
// harness errors are redacted at THIS boundary; the bundle's `stream.actor` carries the
|
|
25
30
|
// conformant humanish.actor-trace.v1 projection, whose `redaction.screenshots` records the
|
|
26
31
|
// run's actual mode ("raw" | "blurred" | "n/a") — every label downstream derives from it.
|
|
27
|
-
import { resolveAutomaticAnalysis } from "./automatic-analysis-config.js";
|
|
28
|
-
import { completeAutomaticAnalysis, markFinalizedStudyResult } from "./automatic-analysis-completion.js";
|
|
29
|
-
import { desktopMediaValidationReason, taskProtocolValidationReason } from "./lab-config.js";
|
|
30
32
|
import { randomBytes } from "node:crypto";
|
|
31
|
-
import { describeMissingKeys } from "./key-resolution.js";
|
|
32
33
|
import { readFile, realpath, rm } from "node:fs/promises";
|
|
33
34
|
import path from "node:path";
|
|
35
|
+
import { completeAutomaticAnalysis, markFinalizedStudyResult } from "./automatic-analysis-completion.js";
|
|
36
|
+
import { resolveAutomaticAnalysis } from "./automatic-analysis-config.js";
|
|
37
|
+
import { describeMissingKeys } from "./key-resolution.js";
|
|
38
|
+
import { desktopMediaValidationReason, taskProtocolValidationReason } from "./lab-config.js";
|
|
39
|
+
import { pathToFileURL } from "node:url";
|
|
40
|
+
import { toErrorMessage } from "./command-failure.js";
|
|
34
41
|
import { cuaLaneDiagnostics, summarizeCuaDiagnostics } from "./cua-diagnostics.js";
|
|
35
42
|
import { feedbackProofCommands } from "./feedback-proof.js";
|
|
36
|
-
import { runDesktopCommandOrThrow, toErrorMessage } from "./command-failure.js";
|
|
37
|
-
import { pathToFileURL } from "node:url";
|
|
38
|
-
import { beginRunStatus, withRunStatusScope } from "./run-status.js";
|
|
39
|
-
import { adapterScoreFailureMessage, applyBrowserAdapterHooks } from "./adapter-extension.js";
|
|
40
43
|
import { actorRegistry, isCuaActorDescriptor } from "./actor-registry.js";
|
|
41
|
-
import {
|
|
42
|
-
import {
|
|
43
|
-
import { createLocalAgentProvider, detectLocalAgents } from "./local-agent-cli.js";
|
|
44
|
-
import { startAppServerSession } from "./local-agent-appserver.js";
|
|
45
|
-
import { startClaudeSession } from "./local-agent-claude-session.js";
|
|
46
|
-
import { createDesktopSandbox, withOneRetryOnTransientE2BError, loadE2BDesktopModule } from "./e2b-desktop-launch.js";
|
|
47
|
-
import { probeUrl, readDetachedLog, runDetachedStep, startDetachedProcess } from "./e2b-detached.js";
|
|
48
|
-
import { DEFAULT_SANDBOX_CATCH_PORT, collectCommsThread, collectExternalCommsThread, deployCommsCatch, externalCatchHealthy, externalInboxUrl, refreshInboxSurface, writeInboxSurface } from "./comms-sandbox-catch.js";
|
|
44
|
+
import { actorEnding } from "./actor-stop-cause.js";
|
|
45
|
+
import { adapterScoreFailureMessage, applyBrowserAdapterHooks } from "./adapter-extension.js";
|
|
49
46
|
import { FakeInbox } from "./comms-fake-inbox.js";
|
|
50
|
-
import {
|
|
51
|
-
import {
|
|
52
|
-
import { cuaLaneValidationReason, outputTokenLimitValidationReason, isHttpUrl, isLoopbackUrl, MAX_CUA_LANES, subjectStateInvalidReason } from "./lab-config.js";
|
|
47
|
+
import { recipientInboxUrl } from "./comms-inbox.js";
|
|
48
|
+
import { collectExternalCommsThread, externalCatchHealthy, externalInboxUrl } from "./comms-sandbox-catch.js";
|
|
53
49
|
import { mapWithConcurrency } from "./concurrency.js";
|
|
54
|
-
import {
|
|
50
|
+
import { DEFAULT_DEVICE_PRESET, isDevicePresetName, resolveDevicePreset } from "./device-presets.js";
|
|
51
|
+
import {} from "./e2b-desktop-launch.js";
|
|
52
|
+
import {} from "./e2b-desktop-resources.js";
|
|
53
|
+
import {} from "./e2b-detached.js";
|
|
55
54
|
import { assertScreenshotEvidence } from "./image-evidence.js";
|
|
55
|
+
import { MAX_CUA_LANES, cuaLaneValidationReason, isHttpUrl, isLoopbackUrl, outputTokenLimitValidationReason, subjectStateInvalidReason } from "./lab-config.js";
|
|
56
|
+
import { startAppServerSession } from "./local-agent-appserver.js";
|
|
57
|
+
import { startClaudeSession } from "./local-agent-claude-session.js";
|
|
58
|
+
import { createLocalAgentProvider, detectLocalAgents } from "./local-agent-cli.js";
|
|
56
59
|
import { buildObserverData } from "./observer-data.js";
|
|
57
|
-
import { corepackCommandFor, needsNodeRuntime, nodeBootstrapCommand } from "./subject-runtime.js";
|
|
58
|
-
import { TERMINAL_NODE_BOOTSTRAP_COMMAND } from "./terminal-node-bootstrap.js";
|
|
59
|
-
import { chromeCdpProbeCommand, parseChromeCdpProbeOutput } from "./chrome-cdp-probe.js";
|
|
60
|
-
import { personaToDirectives, renderPersonaPromptSection } from "./persona.js";
|
|
61
|
-
import { labPersonaIds, resolveCommittedPersonas } from "./persona-resolve.js";
|
|
62
|
-
import { renderTaskPrompt } from "./tasks.js";
|
|
63
60
|
import { attachObserverRuntimeStreamUrls, renderObserver } from "./observer.js";
|
|
64
|
-
import {
|
|
61
|
+
import { DEFAULT_OPENAI_CU_MODEL } from "./openai-responses-cu.js";
|
|
65
62
|
import { participantAssignment } from "./participant-assignment.js";
|
|
66
|
-
import {
|
|
67
|
-
import {
|
|
63
|
+
import { labPersonaIds, resolveCommittedPersonas } from "./persona-resolve.js";
|
|
64
|
+
import { personaToDirectives, renderPersonaPromptSection } from "./persona.js";
|
|
65
|
+
import { MODEL_RATES, estimateActorCost, estimateActorCostForExecution, estimateAllocatedDesktopCost, estimateDesktopCost, round6 } from "./pricing.js";
|
|
66
|
+
import { containsSensitive, digestText, redactText } from "./redaction.js";
|
|
68
67
|
import { prepareRunArtifactPaths, validatePreparedRunArtifactPaths } from "./run-paths.js";
|
|
68
|
+
import { beginRunStatus, withRunStatusScope } from "./run-status.js";
|
|
69
|
+
import { PUBLIC_TARGET_CWD, REVIEW_SCHEMA, RUN_BUNDLE_SCHEMA, aggregateTaskFunnels, buildRunSource, formatParticipantOutcomes, formatStudyTaskFunnel, loadRunBundle, tallyParticipantOutcomes, withCuaReviewProvenance } from "./run.js";
|
|
70
|
+
import { assertPreparedSelectedOutputDirectory, assertSafeOutputPathSegment, prepareContainedOutputDirectory, prepareSelectedOutputDirectory, writeContainedOutputFile, writePreparedRunLatestPointer } from "./selected-output-paths.js";
|
|
69
71
|
import { createLocalTreeArchive } from "./source-archive.js";
|
|
70
|
-
import {
|
|
71
|
-
import { estimateActorCost, estimateDesktopCost, estimateAllocatedDesktopCost, MODEL_RATES, round6 } from "./pricing.js";
|
|
72
|
-
import { observeDesktopResources } from "./e2b-desktop-resources.js";
|
|
72
|
+
import { renderTaskPrompt } from "./tasks.js";
|
|
73
73
|
export const CUA_ACTOR_LAB_SCHEMA = "humanish.cua-lab-result.v2";
|
|
74
74
|
// The only fan-out topology this slice ships: N lanes = N independent E2B desktop sandboxes,
|
|
75
75
|
// each its own world (clone/serve + subject.state per lane). Shared-world is layer 7 (#164).
|
|
@@ -77,10 +77,6 @@ export const CUA_FANOUT_STRATEGY = "per-lane-worlds";
|
|
|
77
77
|
// Env override that may only LOWER the effective concurrency (never raise concurrent paid
|
|
78
78
|
// desktops — invariant 3). Read names-only into a local; the value never persists.
|
|
79
79
|
const CUA_MAX_CONCURRENCY_ENV = "HUMANISH_CUA_MAX_CONCURRENCY";
|
|
80
|
-
export const CUA_ACTOR_LAB_PROVIDER_METADATA = {
|
|
81
|
-
mode: "cua-actor-lab",
|
|
82
|
-
tool: "humanish"
|
|
83
|
-
};
|
|
84
80
|
// The DEFAULT session budget, sized so a study can FINISH (docs/principles/three-roles.md: a
|
|
85
81
|
// session ends because the participant is done, not because a timer fired — the time-box is a
|
|
86
82
|
// session-level cap a researcher sets generously; spend protection is the dollar caps' job).
|
|
@@ -103,60 +99,6 @@ function defaultSessionTimeoutMs(config) {
|
|
|
103
99
|
const room = MAX_SANDBOX_MS - SUBJECT_PROVISION_BUDGET_MS - stateBudgetMs - SANDBOX_TIMEOUT_BUFFER_MS;
|
|
104
100
|
return Math.max(MIN_DERIVED_SESSION_TIMEOUT_MS, Math.min(DEFAULT_APP_URL_SESSION_TIMEOUT_MS, room));
|
|
105
101
|
}
|
|
106
|
-
// Settle after opening the browser, before the first screenshot — long enough for a cold
|
|
107
|
-
// browser + page load to paint (2s captured a blank desktop; the render empirically needs ~6-9s).
|
|
108
|
-
const BROWSER_SETTLE_MS = 8_000;
|
|
109
|
-
/** Where a lane's synthetic camera feed lives inside the sandbox: a tmpfs the sandbox user can
|
|
110
|
-
* write, and a path that contains neither /tmp/ nor /home/, which the public-safety scan reads
|
|
111
|
-
* as an operator's local path (this one is the harness's own and belongs in the bundle). */
|
|
112
|
-
export const SANDBOX_MEDIA_DIR = "/dev/shm/humanish-media";
|
|
113
|
-
export const SANDBOX_CAMERA_PATH = `${SANDBOX_MEDIA_DIR}/camera.y4m`;
|
|
114
|
-
/** The synthetic feed: ffmpeg's test pattern, 640x480 at 10 fps, six seconds (about 28 MB of
|
|
115
|
-
* raw Y4M on the tmpfs), looped by Chrome's fake capture device. */
|
|
116
|
-
export const SYNTHETIC_CAMERA_COMMAND = `mkdir -p ${SANDBOX_MEDIA_DIR} && ffmpeg -y -loglevel error -f lavfi -i testsrc=size=640x480:rate=10 -t 6 -pix_fmt yuv420p ${SANDBOX_CAMERA_PATH}`;
|
|
117
|
-
/**
|
|
118
|
-
* Put the declared camera feed in the sandbox and return the Chromium flags that present it as a
|
|
119
|
-
* capture device (#509). Fails CLOSED: a feed that cannot be produced (no ffmpeg on the image, an
|
|
120
|
-
* unreadable host file) is named before the browser launches, because a participant told it has
|
|
121
|
-
* a camera and finds none reports the instrument's gap as the product's.
|
|
122
|
-
*/
|
|
123
|
-
export async function prepareDesktopMedia(desktop, media, permission, cwd, requestTimeoutMs, readHostFile = (absolutePath) => readFile(absolutePath)) {
|
|
124
|
-
if (media.microphone !== undefined) {
|
|
125
|
-
throw new Error("execution.desktop.media.microphone.source injection is unsupported; the declared microphone file cannot be delivered, including on custom templates.");
|
|
126
|
-
}
|
|
127
|
-
const flags = [];
|
|
128
|
-
let camera;
|
|
129
|
-
if (media.camera !== undefined) {
|
|
130
|
-
if (media.camera.source === "synthetic") {
|
|
131
|
-
const made = await desktop.commands.run(SYNTHETIC_CAMERA_COMMAND, { requestTimeoutMs, timeoutMs: 60_000 });
|
|
132
|
-
if (made.exitCode !== undefined && made.exitCode !== 0) {
|
|
133
|
-
throw new Error(`the synthetic camera feed could not be generated on this desktop image (ffmpeg exited ${made.exitCode}: ${tailOf(made.stderr ?? made.stdout ?? "")}); give execution.desktop.media.camera.source a .y4m file instead`);
|
|
134
|
-
}
|
|
135
|
-
camera = { source: "synthetic", file: SANDBOX_CAMERA_PATH };
|
|
136
|
-
}
|
|
137
|
-
else {
|
|
138
|
-
const absolutePath = path.resolve(cwd, media.camera.source);
|
|
139
|
-
let bytes;
|
|
140
|
-
try {
|
|
141
|
-
bytes = await readHostFile(absolutePath);
|
|
142
|
-
}
|
|
143
|
-
catch (error) {
|
|
144
|
-
throw new Error(`execution.desktop.media.camera.source could not be read (${toErrorMessage(error)})`);
|
|
145
|
-
}
|
|
146
|
-
if (bytes.length > 64 * 1024 * 1024) {
|
|
147
|
-
throw new Error(`execution.desktop.media.camera.source is ${bytes.length} bytes; the camera feed is capped at 64 MiB`);
|
|
148
|
-
}
|
|
149
|
-
await desktop.commands.run(`mkdir -p ${SANDBOX_MEDIA_DIR}`, { requestTimeoutMs, timeoutMs: 15_000 });
|
|
150
|
-
const payload = bytes.buffer.slice(bytes.byteOffset, bytes.byteOffset + bytes.byteLength);
|
|
151
|
-
await desktop.files.write(SANDBOX_CAMERA_PATH, payload, { requestTimeoutMs, useOctetStream: true });
|
|
152
|
-
camera = { source: "file", file: SANDBOX_CAMERA_PATH };
|
|
153
|
-
}
|
|
154
|
-
flags.push("--use-fake-device-for-media-stream", `--use-file-for-fake-video-capture=${SANDBOX_CAMERA_PATH}`);
|
|
155
|
-
}
|
|
156
|
-
if (permission === "granted")
|
|
157
|
-
flags.push("--use-fake-ui-for-media-stream");
|
|
158
|
-
return { ...(camera === undefined ? {} : { camera }), permission, flags };
|
|
159
|
-
}
|
|
160
102
|
// Device/screen size comes from the named-preset registry (device-presets.ts), selectable per run
|
|
161
103
|
// via execution.desktop.device (default `desktop`=1440x950). NOTE: this is run-wide for now; a
|
|
162
104
|
// per-PERSONA device dimension (N personas × devices, as the bespoke sims author) lands with
|
|
@@ -172,19 +114,6 @@ const SUBJECT_PROVISION_BUDGET_MS = 30 * 60_000;
|
|
|
172
114
|
* The derived per-lane deadline has to stay under it, and saying so at plan time beats discovering
|
|
173
115
|
* it from a raw provider 400 after a plan has already printed. */
|
|
174
116
|
const MAX_SANDBOX_MS = 60 * 60_000;
|
|
175
|
-
export const SUBJECT_DIR = "/home/user/subject";
|
|
176
|
-
// Remote path for the once-per-run packed local-tree archive; removed by the extract step
|
|
177
|
-
// after it unpacks into SUBJECT_DIR.
|
|
178
|
-
const LOCAL_TREE_REMOTE_ARCHIVE_PATH = "/home/user/.humanish-source.tar.gz";
|
|
179
|
-
const CLONE_TIMEOUT_MS = 5 * 60_000;
|
|
180
|
-
const INSTALL_TIMEOUT_MS = 10 * 60_000;
|
|
181
|
-
const BUILD_TIMEOUT_MS = 10 * 60_000;
|
|
182
|
-
const DEFAULT_READY_TIMEOUT_MS = 180_000;
|
|
183
|
-
// Per-step budget for subject.state seed steps; each step's declared (or default) budget is
|
|
184
|
-
// also summed into the default sandbox deadline so seeding never eats the session's room.
|
|
185
|
-
const DEFAULT_STATE_STEP_TIMEOUT_MS = 5 * 60_000;
|
|
186
|
-
// How much of a failing step's log tail rides the (redacted) error message.
|
|
187
|
-
const ERROR_TAIL_CHARS = 2000;
|
|
188
117
|
const DEFAULT_MISSION = "You are testing a web application. The browser is already open at the subject URL. Explore it, accomplish what the scenario asks, and stop when done.";
|
|
189
118
|
/**
|
|
190
119
|
* The participant's outcome as ONE fixed first line of its last message (#570, second half). The
|
|
@@ -262,21 +191,6 @@ export function withInboxMission(spec, inboxUrl, address, receiving = false) {
|
|
|
262
191
|
instructions: `${spec.instructions}\n\nEmail inbox:${identity} When the app tells you it has emailed you (a verification link, confirmation code, or magic link), open ${recipientInboxUrl(inboxUrl, address)} in the browser to read that email and follow its link or enter its code. All email the app sends you arrives there. Waiting for an email is normal, not a blocker — do not end your session while waiting; open the inbox and refresh it until the email appears.`
|
|
263
192
|
};
|
|
264
193
|
}
|
|
265
|
-
/** The lane's addressed comms recipient, when one exists — the gate AND the address source for the
|
|
266
|
-
* inbox instruction (#351). A lane told to check an inbox it can never receive into would stall,
|
|
267
|
-
* so no addressed recipient means no instruction. */
|
|
268
|
-
export function inboxRecipientFor(commsEmail, laneId) {
|
|
269
|
-
return (commsEmail.recipients ?? []).find((recipient) => recipient.lane === laneId && recipient.address !== undefined);
|
|
270
|
-
}
|
|
271
|
-
/** True when a lane has a declared comms recipient WITH an address, so the drain can actually match the
|
|
272
|
-
* mail the persona will be told to read. Gates the inbox instruction to lanes that can receive mail —
|
|
273
|
-
* a lane told to check an inbox it can never receive into would just stall. */
|
|
274
|
-
export function laneHasInboxRecipient(commsEmail, laneId) {
|
|
275
|
-
return inboxRecipientFor(commsEmail, laneId) !== undefined;
|
|
276
|
-
}
|
|
277
|
-
/** Mid-run inbox-surface render cadence (ms). Coarse enough that the per-tick `cat` + file writes stay
|
|
278
|
-
* cheap; fine enough that a verification email is visible seconds after the app sends it. */
|
|
279
|
-
const INBOX_SURFACE_CADENCE_MS = 2500;
|
|
280
194
|
/**
|
|
281
195
|
* The narrowest browser WINDOW Chrome/Chromium will render on the E2B desktop. Chrome refuses to
|
|
282
196
|
* make its window narrower than this (~500 CSS px observed: a 414-wide X screen produced a 500-wide
|
|
@@ -292,20 +206,6 @@ export const MIN_DESKTOP_RENDER_WIDTH = 500;
|
|
|
292
206
|
export function floorRenderResolution(resolution) {
|
|
293
207
|
return [Math.max(resolution[0], MIN_DESKTOP_RENDER_WIDTH), resolution[1]];
|
|
294
208
|
}
|
|
295
|
-
/**
|
|
296
|
-
* The DECLARED preset to record alongside the rendered screen, or undefined when the preset
|
|
297
|
-
* rendered faithfully.
|
|
298
|
-
*
|
|
299
|
-
* `desktopGeometry.screen.verified` compares the FLOORED number with itself, so on its own a
|
|
300
|
-
* floored run is indistinguishable from a faithful one: a reader sees requested 500 / verified 500
|
|
301
|
-
* and concludes a 500-wide screen was asked for. Recording the declared preset is what makes
|
|
302
|
-
* "the preset width did not render" legible in the bundle.
|
|
303
|
-
*/
|
|
304
|
-
export function declaredScreenForRender(preset, presetName, rendered) {
|
|
305
|
-
if (preset.width === rendered[0] && preset.height === rendered[1])
|
|
306
|
-
return undefined;
|
|
307
|
-
return { width: preset.width, height: preset.height, preset: presetName };
|
|
308
|
-
}
|
|
309
209
|
/**
|
|
310
210
|
* Resolve a lane's device + rendered resolution (most-specific wins, exactly as the single-lane
|
|
311
211
|
* path always has): a raw execution.desktop.resolution escape hatch (only legal when no lane
|
|
@@ -333,6 +233,8 @@ export function resolveLaneDevice(config, lane) {
|
|
|
333
233
|
* git clone for an upload+extract, but the shared install/build/state/start/probe pipeline
|
|
334
234
|
* costs the same wall-clock room either way. */
|
|
335
235
|
function resolvePerLaneSandboxMs(config) {
|
|
236
|
+
if (isLocalBrowserLab(config))
|
|
237
|
+
return LOCAL_BROWSER_LIFETIME_MS;
|
|
336
238
|
const timeoutMs = config.execution?.timeoutMs ?? defaultSessionTimeoutMs(config);
|
|
337
239
|
const provisionedRoute = config.subject.source === "clone" || config.subject.source === "local-tree";
|
|
338
240
|
const stateBudgetMs = provisionedRoute
|
|
@@ -582,65 +484,6 @@ function formatLanePlanEntry(lane) {
|
|
|
582
484
|
].filter((part) => part !== undefined);
|
|
583
485
|
return `${lane.id}: persona=${lane.persona}${taxonomy.length > 0 ? ` ${taxonomy.join(" ")}` : ""} device=${lane.device} ${lane.resolution[0]}x${lane.resolution[1]} prompt#${lane.instructionDigest}${lane.targetDigest ? ` target#${lane.targetDigest}` : ""}`;
|
|
584
486
|
}
|
|
585
|
-
/** ISO timestamp from an injectable clock (tests freeze `now` for deterministic durationMs). */
|
|
586
|
-
function isoNow(now) {
|
|
587
|
-
return new Date(now()).toISOString();
|
|
588
|
-
}
|
|
589
|
-
/** Emit a phase-started event (no ok/durationMs: those belong to the matching completed event). */
|
|
590
|
-
/**
|
|
591
|
-
* Run a provisioning step and, when it fails with an EXIT CODE, run it once more (#602). A cold
|
|
592
|
-
* install of 0.74.0 lost its whole first live study to one transient TLS error inside the
|
|
593
|
-
* sandbox's `npm install`; the parallel install twenty seconds later passed, as had the ten
|
|
594
|
-
* before it. One retry clears that class. A TIMEOUT is not retried: its budget is already spent,
|
|
595
|
-
* and a second wait would double it. The retry runs under its own step name so both logs stay.
|
|
596
|
-
*/
|
|
597
|
-
async function runProvisioningStepWithOneRetry(desktop, args) {
|
|
598
|
-
const first = await runDetachedStep(desktop, {
|
|
599
|
-
name: args.name,
|
|
600
|
-
command: args.command,
|
|
601
|
-
cwd: args.cwd,
|
|
602
|
-
timeoutMs: args.timeoutMs,
|
|
603
|
-
requestTimeoutMs: args.requestTimeoutMs,
|
|
604
|
-
...args.timers
|
|
605
|
-
});
|
|
606
|
-
if (first.ok || first.timedOut)
|
|
607
|
-
return { ...first, attempts: 1 };
|
|
608
|
-
const retryStartedAt = args.now();
|
|
609
|
-
emitPhaseStarted(args.onPhase, args.now, args.retryPhase, `${args.retryMessage} (first attempt exited ${first.exitCode ?? "null"}; retrying once)`);
|
|
610
|
-
const second = await runDetachedStep(desktop, {
|
|
611
|
-
name: `${args.name}-retry`,
|
|
612
|
-
command: args.command,
|
|
613
|
-
cwd: args.cwd,
|
|
614
|
-
timeoutMs: args.timeoutMs,
|
|
615
|
-
requestTimeoutMs: args.requestTimeoutMs,
|
|
616
|
-
...args.timers
|
|
617
|
-
});
|
|
618
|
-
emitPhaseCompleted(args.onPhase, args.now, retryStartedAt, args.retryPhase, second.ok, second.ok ? `${args.retryMessage}: succeeded on the second attempt` : `${args.retryMessage}: failed twice`);
|
|
619
|
-
return { ...second, attempts: 2, ...(first.exitCode === undefined ? {} : { firstExitCode: first.exitCode }) };
|
|
620
|
-
}
|
|
621
|
-
function emitPhaseStarted(onPhase, now, phase, message) {
|
|
622
|
-
onPhase?.({ at: isoNow(now), type: `cua-lab.subject.${phase}.started`, message });
|
|
623
|
-
}
|
|
624
|
-
/** Emit the matching phase-completed event: always carries ok and durationMs (>= 0). */
|
|
625
|
-
function emitPhaseCompleted(onPhase, now, startedAt, phase, ok, message) {
|
|
626
|
-
onPhase?.({
|
|
627
|
-
at: isoNow(now),
|
|
628
|
-
type: `cua-lab.subject.${phase}.completed`,
|
|
629
|
-
ok,
|
|
630
|
-
durationMs: Math.max(0, now() - startedAt),
|
|
631
|
-
message
|
|
632
|
-
});
|
|
633
|
-
}
|
|
634
|
-
/** Default phase-boundary sink (stderr): one line per event, prefixed with the lane id ONLY
|
|
635
|
-
* when laneCount > 1. Single-lane emission is unconditional: total single-lane silence for the
|
|
636
|
-
* whole clone/install/build/ready boot is the bug this event stream exists to close.
|
|
637
|
-
* Overridable via CuaActorLabHooks.onPhase so deterministic tests capture instead of writing to
|
|
638
|
-
* the real stderr. */
|
|
639
|
-
function defaultSubjectPhaseSink(event, ctx) {
|
|
640
|
-
const durationSuffix = event.durationMs === undefined ? "" : ` (${event.durationMs}ms)`;
|
|
641
|
-
const prefix = ctx.laneCount > 1 ? `humanish cua [${ctx.laneId}]` : "humanish cua";
|
|
642
|
-
process.stderr.write(`${prefix}: ${event.message}${durationSuffix}\n`);
|
|
643
|
-
}
|
|
644
487
|
/** Short id-safe suffix for a subject-phase RunEvent: drops the shared prefix/suffix so each
|
|
645
488
|
* phase gets a distinct bundle event id (e.g. "clone", "state-before-build"). */
|
|
646
489
|
function phaseEventIdSuffix(type) {
|
|
@@ -681,36 +524,6 @@ export function makeLaneWriteScreenshot(artifactRoot, spec, screenshots) {
|
|
|
681
524
|
return rel;
|
|
682
525
|
};
|
|
683
526
|
}
|
|
684
|
-
/**
|
|
685
|
-
* Verify the desktop screen geometry IN-SANDBOX (the per-lane device claim is checked, never
|
|
686
|
-
* assumed). A parseable mismatch fails closed. Unavailable/unparseable evidence is returned as
|
|
687
|
-
* an explicit warning: the lane may still run, but its bundle records only the requested screen
|
|
688
|
-
* and never upgrades that request into a verified measurement.
|
|
689
|
-
*/
|
|
690
|
-
export async function inspectDesktopScreenGeometry(args) {
|
|
691
|
-
let out = "";
|
|
692
|
-
try {
|
|
693
|
-
const result = await args.desktop.commands.run("xdpyinfo 2>/dev/null | grep -i dimensions || true", { requestTimeoutMs: args.requestTimeoutMs });
|
|
694
|
-
out = (result.stdout ?? "").trim();
|
|
695
|
-
}
|
|
696
|
-
catch {
|
|
697
|
-
return { warning: `Desktop screen geometry could not be measured for lane ${args.laneId}; requested geometry remains unverified.` };
|
|
698
|
-
}
|
|
699
|
-
const match = out.match(/(\d+)\s*x\s*(\d+)\s*pixels/i);
|
|
700
|
-
if (!match) {
|
|
701
|
-
return { warning: `Desktop screen geometry could not be parsed for lane ${args.laneId}; requested geometry remains unverified.` };
|
|
702
|
-
}
|
|
703
|
-
const width = Number(match[1]);
|
|
704
|
-
const height = Number(match[2]);
|
|
705
|
-
const [expectedWidth, expectedHeight] = args.requestedScreen;
|
|
706
|
-
if (width === expectedWidth && height === expectedHeight) {
|
|
707
|
-
return { verified: { width, height, source: "xdpyinfo" } };
|
|
708
|
-
}
|
|
709
|
-
return {
|
|
710
|
-
verified: { width, height, source: "xdpyinfo" },
|
|
711
|
-
error: `HUMANISH_CUA_LAB_DEVICE_GEOMETRY: lane ${args.laneId} requested a ${expectedWidth}x${expectedHeight} desktop but xdpyinfo reports ${width}x${height} in-sandbox; the per-lane device geometry is unverified (fail-closed).`
|
|
712
|
-
};
|
|
713
|
-
}
|
|
714
527
|
/** A blocked lane outcome (pipeline gate / fail-fast skipped it before it ran). */
|
|
715
528
|
function blockedLaneOutcome(spec, reason) {
|
|
716
529
|
return {
|
|
@@ -728,725 +541,6 @@ function blockedLaneOutcome(spec, reason) {
|
|
|
728
541
|
harnessError: false
|
|
729
542
|
};
|
|
730
543
|
}
|
|
731
|
-
async function findVisibleBrowserWindowId(desktop, requestTimeoutMs, browserFamily, launchIdentity) {
|
|
732
|
-
if (browserFamily === "unknown")
|
|
733
|
-
return undefined;
|
|
734
|
-
// The candidate loop keeps the LAST identity match: with a launch identity the match is
|
|
735
|
-
// unique anyway, and without one every family candidate matches, so the newest visible
|
|
736
|
-
// window of the launched family wins (the window this lane just opened).
|
|
737
|
-
const finder = browserFamily === "firefox"
|
|
738
|
-
? [
|
|
739
|
-
"find_firefox_window() {",
|
|
740
|
-
" timeout 2s xdotool search --onlyvisible --class 'firefox|Firefox' 2>/dev/null || true",
|
|
741
|
-
"}",
|
|
742
|
-
"window_id=",
|
|
743
|
-
"for _ in $(seq 1 10); do",
|
|
744
|
-
" for candidate in $(find_firefox_window); do",
|
|
745
|
-
" window_pid=\"$(xdotool getwindowpid \"$candidate\" 2>/dev/null || true)\"",
|
|
746
|
-
" if matches_launch_identity \"$window_pid\"; then window_id=\"$candidate\"; fi",
|
|
747
|
-
" done",
|
|
748
|
-
" if [ -n \"$window_id\" ]; then break; fi",
|
|
749
|
-
" sleep 0.5",
|
|
750
|
-
"done"
|
|
751
|
-
]
|
|
752
|
-
: [
|
|
753
|
-
"find_chrome_window() {",
|
|
754
|
-
" timeout 2s xdotool search --onlyvisible --class 'google-chrome|Google-chrome|chromium|Chromium|chrome|Chrome' 2>/dev/null || true",
|
|
755
|
-
"}",
|
|
756
|
-
"window_id=",
|
|
757
|
-
"for _ in $(seq 1 10); do",
|
|
758
|
-
" for candidate in $(find_chrome_window); do",
|
|
759
|
-
" window_pid=\"$(xdotool getwindowpid \"$candidate\" 2>/dev/null || true)\"",
|
|
760
|
-
" if matches_launch_identity \"$window_pid\"; then window_id=\"$candidate\"; fi",
|
|
761
|
-
" done",
|
|
762
|
-
" if [ -n \"$window_id\" ]; then break; fi",
|
|
763
|
-
" sleep 0.5",
|
|
764
|
-
"done"
|
|
765
|
-
];
|
|
766
|
-
const result = await desktop.commands.run([
|
|
767
|
-
"set -euo pipefail",
|
|
768
|
-
"export DISPLAY=\"${DISPLAY:-:0}\"",
|
|
769
|
-
`launch_pid=${shellSingleQuote(launchIdentity?.processId ?? "")}`,
|
|
770
|
-
`profile_dir=${shellSingleQuote(launchIdentity?.profileDir ?? "")}`,
|
|
771
|
-
"matches_launch_identity() {",
|
|
772
|
-
" if [ -z \"$launch_pid\" ] && [ -z \"$profile_dir\" ]; then return 0; fi",
|
|
773
|
-
" local current=\"${1:-}\"",
|
|
774
|
-
" while [[ \"$current\" =~ ^[0-9]+$ ]] && [ \"$current\" -gt 1 ]; do",
|
|
775
|
-
" cmdline=\"$(tr '\\0' ' ' < \"/proc/$current/cmdline\" 2>/dev/null || true)\"",
|
|
776
|
-
" if [ -n \"$profile_dir\" ] && [[ \"$cmdline\" == *\"$profile_dir\"* ]]; then return 0; fi",
|
|
777
|
-
" if [ \"$current\" = \"$launch_pid\" ]; then return 0; fi",
|
|
778
|
-
" current=\"$(ps -o ppid= -p \"$current\" 2>/dev/null | tr -d ' ' || true)\"",
|
|
779
|
-
" done",
|
|
780
|
-
" return 1",
|
|
781
|
-
"}",
|
|
782
|
-
...finder,
|
|
783
|
-
"if [ -n \"$window_id\" ]; then printf 'WINDOW_ID=%s\\n' \"$window_id\"; fi"
|
|
784
|
-
].join("\n"), {
|
|
785
|
-
requestTimeoutMs,
|
|
786
|
-
timeoutMs: 15_000
|
|
787
|
-
});
|
|
788
|
-
return (result.stdout ?? "").match(/^WINDOW_ID=(\S+)$/m)?.[1];
|
|
789
|
-
}
|
|
790
|
-
/**
|
|
791
|
-
* Build the xdotool command that makes a browser window fill the desktop.
|
|
792
|
-
* Exported (pure) for contract tests. A window manager can ignore Chrome's
|
|
793
|
-
* --window-size, so xdotool is the robust path: move the window to the origin,
|
|
794
|
-
* then size it to the exact desktop resolution so Observer screenshots carry no
|
|
795
|
-
* dead margin around the browser.
|
|
796
|
-
*/
|
|
797
|
-
export function buildFillDesktopWindowCommand(windowId, width, height) {
|
|
798
|
-
return [
|
|
799
|
-
"set -euo pipefail",
|
|
800
|
-
`win=${shellSingleQuote(windowId)}`,
|
|
801
|
-
`xdotool windowactivate "$win" >/dev/null 2>&1 || true`,
|
|
802
|
-
`xdotool windowmove "$win" 0 0 >/dev/null 2>&1 || true`,
|
|
803
|
-
`xdotool windowsize "$win" ${width} ${height} >/dev/null 2>&1 || true`,
|
|
804
|
-
].join("\n");
|
|
805
|
-
}
|
|
806
|
-
/**
|
|
807
|
-
* Best-effort initial fill. A contained smaller window remains usable; the capture
|
|
808
|
-
* below checks for clipping and refuses an uncorrectable window before the actor runs.
|
|
809
|
-
*/
|
|
810
|
-
async function fillDesktopBrowserWindow(desktop, windowId, resolution, requestTimeoutMs) {
|
|
811
|
-
const [width, height] = resolution;
|
|
812
|
-
await desktop.commands
|
|
813
|
-
.run(buildFillDesktopWindowCommand(windowId, width, height), {
|
|
814
|
-
requestTimeoutMs,
|
|
815
|
-
timeoutMs: 10_000,
|
|
816
|
-
})
|
|
817
|
-
.catch(() => undefined);
|
|
818
|
-
}
|
|
819
|
-
async function openDesktopBrowserTarget(desktop, targetUrl, requestTimeoutMs, browserPreference,
|
|
820
|
-
/** Launch-time flags that make mobile fidelity (#221) hold across every tab: the user agent and
|
|
821
|
-
* touch events are browser-wide here, where the CDP holder covers only the launch page. */
|
|
822
|
-
extraChromiumFlags = []) {
|
|
823
|
-
const requestedBrowser = browserPreference ?? "default";
|
|
824
|
-
if (isHttpUrl(targetUrl)) {
|
|
825
|
-
const chromiumFlags = [...CHROMIUM_EVIDENCE_HYGIENE_FLAGS, ...extraChromiumFlags].map(shellSingleQuote).join(" ");
|
|
826
|
-
const browserLaunchCommand = [
|
|
827
|
-
"set -euo pipefail",
|
|
828
|
-
`target_url=${shellSingleQuote(targetUrl)}`,
|
|
829
|
-
`browser_preference=${shellSingleQuote(requestedBrowser)}`,
|
|
830
|
-
"chrome_profile_dir=",
|
|
831
|
-
`chrome_preferences_json=${shellSingleQuote(chromiumEvidenceProfilePreferencesJson())}`,
|
|
832
|
-
"prepare_chrome_profile() {",
|
|
833
|
-
" chrome_profile_dir=\"$(mktemp -d /tmp/humanish-chrome-profile.XXXXXX)\"",
|
|
834
|
-
" mkdir -p \"$chrome_profile_dir/Default\"",
|
|
835
|
-
" printf '%s\\n' \"$chrome_preferences_json\" > \"$chrome_profile_dir/Default/Preferences\"",
|
|
836
|
-
"}",
|
|
837
|
-
"launch_browser() {",
|
|
838
|
-
" local label=\"$1\"",
|
|
839
|
-
" local binary=\"$2\"",
|
|
840
|
-
" shift 2",
|
|
841
|
-
" if command -v \"$binary\" >/dev/null 2>&1; then",
|
|
842
|
-
" nohup \"$binary\" \"$@\" \"$target_url\" >/tmp/humanish-browser-open.log 2>&1 &",
|
|
843
|
-
" local launch_pid=$!",
|
|
844
|
-
" printf 'HUMANISH_BROWSER_RESOLVED=%s\\n' \"$label\"",
|
|
845
|
-
" printf 'HUMANISH_BROWSER_PID=%s\\n' \"$launch_pid\"",
|
|
846
|
-
" printf 'HUMANISH_BROWSER_PROFILE_DIR=%s\\n' \"$chrome_profile_dir\"",
|
|
847
|
-
" if [[ \"$label\" =~ ^(google-chrome|google-chrome-stable|chromium|chromium-browser)$ ]]; then",
|
|
848
|
-
" for _ in $(seq 1 30); do",
|
|
849
|
-
" if [ -s \"$chrome_profile_dir/DevToolsActivePort\" ]; then",
|
|
850
|
-
" head -n 1 \"$chrome_profile_dir/DevToolsActivePort\" | sed 's/^/HUMANISH_BROWSER_CDP_PORT=/'",
|
|
851
|
-
" break",
|
|
852
|
-
" fi",
|
|
853
|
-
" sleep 0.1",
|
|
854
|
-
" done",
|
|
855
|
-
" fi",
|
|
856
|
-
" return 0",
|
|
857
|
-
" fi",
|
|
858
|
-
" return 1",
|
|
859
|
-
"}",
|
|
860
|
-
// Fixed CDP port (not :0/random): each seat has its OWN desktop sandbox, so a known port
|
|
861
|
-
// cannot conflict, and it makes the observer's port resolution deterministic. With :0 the
|
|
862
|
-
// real port lives only in DevToolsActivePort; when the launch-time capture misses on a cold
|
|
863
|
-
// start the observer falls back to 9222 and — being wrong — every CDP read fails for the
|
|
864
|
-
// whole run (the lobby-code handoff then never sees the host's /lobby URL). 9222 is already
|
|
865
|
-
// the fallback, so making it the actual port aligns launch, capture, and fallback.
|
|
866
|
-
`chrome_debug_flags=(--remote-debugging-address=127.0.0.1 --remote-debugging-port=9222 ${chromiumFlags})`,
|
|
867
|
-
"open_target() {",
|
|
868
|
-
" case \"$browser_preference\" in",
|
|
869
|
-
" chrome)",
|
|
870
|
-
" prepare_chrome_profile",
|
|
871
|
-
" launch_browser google-chrome google-chrome --new-window \"--user-data-dir=$chrome_profile_dir\" \"${chrome_debug_flags[@]}\" && return 0",
|
|
872
|
-
" launch_browser google-chrome-stable google-chrome-stable --new-window \"--user-data-dir=$chrome_profile_dir\" \"${chrome_debug_flags[@]}\" && return 0",
|
|
873
|
-
" echo 'requested browser chrome was not found' >&2",
|
|
874
|
-
" return 127",
|
|
875
|
-
" ;;",
|
|
876
|
-
" chromium)",
|
|
877
|
-
" prepare_chrome_profile",
|
|
878
|
-
" launch_browser chromium chromium --new-window \"--user-data-dir=$chrome_profile_dir\" \"${chrome_debug_flags[@]}\" && return 0",
|
|
879
|
-
" launch_browser chromium-browser chromium-browser --new-window \"--user-data-dir=$chrome_profile_dir\" \"${chrome_debug_flags[@]}\" && return 0",
|
|
880
|
-
" echo 'requested browser chromium was not found' >&2",
|
|
881
|
-
" return 127",
|
|
882
|
-
" ;;",
|
|
883
|
-
" firefox)",
|
|
884
|
-
" prepare_chrome_profile",
|
|
885
|
-
" launch_browser firefox firefox --new-instance --no-remote --new-window --profile \"$chrome_profile_dir\" && return 0",
|
|
886
|
-
" echo 'requested browser firefox was not found' >&2",
|
|
887
|
-
" return 127",
|
|
888
|
-
" ;;",
|
|
889
|
-
" default)",
|
|
890
|
-
" prepare_chrome_profile",
|
|
891
|
-
" launch_browser google-chrome google-chrome --new-window \"--user-data-dir=$chrome_profile_dir\" \"${chrome_debug_flags[@]}\" && return 0",
|
|
892
|
-
" launch_browser google-chrome-stable google-chrome-stable --new-window \"--user-data-dir=$chrome_profile_dir\" \"${chrome_debug_flags[@]}\" && return 0",
|
|
893
|
-
" launch_browser chromium chromium --new-window \"--user-data-dir=$chrome_profile_dir\" \"${chrome_debug_flags[@]}\" && return 0",
|
|
894
|
-
" launch_browser chromium-browser chromium-browser --new-window \"--user-data-dir=$chrome_profile_dir\" \"${chrome_debug_flags[@]}\" && return 0",
|
|
895
|
-
" launch_browser firefox firefox --new-instance --no-remote --new-window --profile \"$chrome_profile_dir\" && return 0",
|
|
896
|
-
" launch_browser xdg-open xdg-open && return 0",
|
|
897
|
-
" echo 'no browser opener found' >&2",
|
|
898
|
-
" return 127",
|
|
899
|
-
" ;;",
|
|
900
|
-
" esac",
|
|
901
|
-
"}",
|
|
902
|
-
"open_target"
|
|
903
|
-
].join("\n");
|
|
904
|
-
const result = await runDesktopCommandOrThrow(() => desktop.commands.run(browserLaunchCommand, {
|
|
905
|
-
requestTimeoutMs,
|
|
906
|
-
timeoutMs: 15_000,
|
|
907
|
-
}), ({ exitCode, stderrTail }) => new Error(`browser launch failed${exitCode === undefined ? "" : ` with exit ${exitCode}`}: ${stderrTail}`));
|
|
908
|
-
if (result.exitCode !== undefined && result.exitCode !== 0) {
|
|
909
|
-
throw new Error(`browser launch failed with exit ${result.exitCode}: ${tailOf(result.stderr ?? result.stdout ?? "")}`);
|
|
910
|
-
}
|
|
911
|
-
const resolved = (result.stdout ?? "").match(/^HUMANISH_BROWSER_RESOLVED=(\S+)$/m)?.[1];
|
|
912
|
-
const processId = (result.stdout ?? "").match(/^HUMANISH_BROWSER_PID=(\d+)$/m)?.[1];
|
|
913
|
-
const profileDir = (result.stdout ?? "").match(/^HUMANISH_BROWSER_PROFILE_DIR=(\S+)$/m)?.[1];
|
|
914
|
-
const cdpPortRaw = (result.stdout ?? "").match(/^HUMANISH_BROWSER_CDP_PORT=(\d+)$/m)?.[1];
|
|
915
|
-
const cdpPort = cdpPortRaw === undefined ? undefined : Number(cdpPortRaw);
|
|
916
|
-
return {
|
|
917
|
-
family: desktopBrowserFamily(resolved ?? requestedBrowser),
|
|
918
|
-
...(processId === undefined || profileDir === undefined
|
|
919
|
-
? {}
|
|
920
|
-
: { identity: { processId, profileDir, targetUrl, ...(cdpPort === undefined ? {} : { cdpPort }) } }),
|
|
921
|
-
...(browserPreference === undefined
|
|
922
|
-
? {}
|
|
923
|
-
: { evidence: { requested: requestedBrowser, ...(resolved === undefined ? {} : { resolved }) } })
|
|
924
|
-
};
|
|
925
|
-
}
|
|
926
|
-
if (browserPreference === undefined || browserPreference === "default") {
|
|
927
|
-
if (desktop.open) {
|
|
928
|
-
await desktop.open(targetUrl);
|
|
929
|
-
}
|
|
930
|
-
else {
|
|
931
|
-
await desktop.launch("google-chrome", targetUrl);
|
|
932
|
-
}
|
|
933
|
-
return {
|
|
934
|
-
family: desktop.open ? "unknown" : "chromium",
|
|
935
|
-
...(browserPreference === undefined ? {} : { evidence: { requested: requestedBrowser } })
|
|
936
|
-
};
|
|
937
|
-
}
|
|
938
|
-
const launchTarget = requestedBrowser === "chrome" ? "google-chrome"
|
|
939
|
-
: requestedBrowser === "chromium" ? "chromium"
|
|
940
|
-
: requestedBrowser === "firefox" ? "firefox"
|
|
941
|
-
: "google-chrome";
|
|
942
|
-
await desktop.launch(launchTarget, targetUrl);
|
|
943
|
-
return {
|
|
944
|
-
family: desktopBrowserFamily(launchTarget),
|
|
945
|
-
evidence: { requested: requestedBrowser, resolved: launchTarget }
|
|
946
|
-
};
|
|
947
|
-
}
|
|
948
|
-
export function desktopBrowserFamily(value) {
|
|
949
|
-
if (value === "firefox")
|
|
950
|
-
return "firefox";
|
|
951
|
-
if (value === "chrome" || value === "chromium" || value === "google-chrome" || value === "google-chrome-stable" || value === "chromium-browser") {
|
|
952
|
-
return "chromium";
|
|
953
|
-
}
|
|
954
|
-
return "unknown";
|
|
955
|
-
}
|
|
956
|
-
/**
|
|
957
|
-
* The URL / title / page-text / scroll observer behind stopWhen and task criteria. One probe per
|
|
958
|
-
* observation, run on the sandbox's python3 (see chrome-cdp-probe.ts for why not node: #514).
|
|
959
|
-
*
|
|
960
|
-
* "active": follow the participant to whatever tab they are driving now — never pin the state
|
|
961
|
-
* observer to the launch tab (a verification link that opened in a NEW tab left a pinned observer
|
|
962
|
-
* reading the old tab forever).
|
|
963
|
-
*
|
|
964
|
-
* `onUnavailable` fires ONCE, on the first probe that could not read the page, with the reason.
|
|
965
|
-
* The observer still degrades to `{}` for the loop; the callback is how a lane says out loud that
|
|
966
|
-
* url/text criteria are not being measured, instead of letting the funnel report 0/N (#514).
|
|
967
|
-
*/
|
|
968
|
-
export function makeChromeBrowserStateObserver(desktop, requestTimeoutMs, endpoint, targetId, onUnavailable,
|
|
969
|
-
/**
|
|
970
|
-
* Mobile emulation on later tabs (#623): the holder attaches to every page target Chrome opens
|
|
971
|
-
* after the launch page, so a tab the participant opens later should lay out at the phone width
|
|
972
|
-
* too. The first observation on each new target reads that page's OWN report; a target that
|
|
973
|
-
* reports the requested width is recorded through `onCovered`, and one that does not (or cannot
|
|
974
|
-
* be read) fires `onDrift` once, so a phone-labelled lane that spent part of its session at
|
|
975
|
-
* desktop layout says so with the number the page gave.
|
|
976
|
-
*/
|
|
977
|
-
drift) {
|
|
978
|
-
let reported = false;
|
|
979
|
-
let drifted = false;
|
|
980
|
-
const checkedTargets = new Set(drift === undefined ? [] : [drift.emulatedTargetId]);
|
|
981
|
-
const unavailable = (reason) => {
|
|
982
|
-
if (!reported) {
|
|
983
|
-
reported = true;
|
|
984
|
-
onUnavailable?.(reason);
|
|
985
|
-
}
|
|
986
|
-
return {};
|
|
987
|
-
};
|
|
988
|
-
const checkLaterTarget = async (newTargetId) => {
|
|
989
|
-
if (drift === undefined || checkedTargets.has(newTargetId))
|
|
990
|
-
return;
|
|
991
|
-
checkedTargets.add(newTargetId);
|
|
992
|
-
const read = await desktop.commands.run(chromeCdpProbeCommand({ ...endpoint, targetId: newTargetId, prefer: "pinned", mode: "fidelity" }), { requestTimeoutMs, timeoutMs: 5_000 });
|
|
993
|
-
const fidelity = read.exitCode !== undefined && read.exitCode !== 0 ? undefined : parseChromeCdpProbeOutput(read.stdout).fidelity;
|
|
994
|
-
if (fidelity !== undefined && fidelity.innerWidth === drift.expectedWidth) {
|
|
995
|
-
drift.onCovered?.(newTargetId, { innerWidth: fidelity.innerWidth, devicePixelRatio: fidelity.devicePixelRatio, maxTouchPoints: fidelity.maxTouchPoints });
|
|
996
|
-
if (drift.expectTouch === true && fidelity.maxTouchPoints === 0 && !drifted) {
|
|
997
|
-
// The viewport followed; touch did not (yet): the holder reloads a later tab once after its
|
|
998
|
-
// first navigation commits, and this observation may have landed before that reload.
|
|
999
|
-
drifted = true;
|
|
1000
|
-
drift.onDrift(`a later page target reports the ${fidelity.innerWidth} px viewport but navigator.maxTouchPoints 0 on its first observation; touch reaches a document only when it loads under the override`);
|
|
1001
|
-
}
|
|
1002
|
-
return;
|
|
1003
|
-
}
|
|
1004
|
-
if (drifted)
|
|
1005
|
-
return;
|
|
1006
|
-
drifted = true;
|
|
1007
|
-
drift.onDrift(fidelity === undefined
|
|
1008
|
-
? "the participant drove a page target other than the emulated launch tab and that page's own read-back could not be taken; whether it laid out at the phone width is not known"
|
|
1009
|
-
: `the participant drove a page target other than the emulated launch tab and that page reports a ${fidelity.innerWidth} px viewport where ${drift.expectedWidth} px was requested (DPR ${fidelity.devicePixelRatio}); the mobile user agent and touch events are browser-wide, the viewport override was not re-applied to it`);
|
|
1010
|
-
};
|
|
1011
|
-
return async () => {
|
|
1012
|
-
const result = await desktop.commands.run(chromeCdpProbeCommand({ ...endpoint, ...(targetId === undefined ? {} : { targetId }), prefer: "active", mode: "state" }), { requestTimeoutMs, timeoutMs: 5_000 });
|
|
1013
|
-
if (result.exitCode !== undefined && result.exitCode !== 0) {
|
|
1014
|
-
return unavailable(`probe exited ${result.exitCode}: ${tailOf(result.stderr ?? result.stdout ?? "")}`);
|
|
1015
|
-
}
|
|
1016
|
-
const parsed = parseChromeCdpProbeOutput(result.stdout);
|
|
1017
|
-
if (parsed.unavailable !== undefined)
|
|
1018
|
-
return unavailable(parsed.unavailable);
|
|
1019
|
-
if (parsed.targetId !== undefined)
|
|
1020
|
-
await checkLaterTarget(parsed.targetId);
|
|
1021
|
-
return {
|
|
1022
|
-
...(parsed.url === undefined ? {} : { url: parsed.url }),
|
|
1023
|
-
...(parsed.title === undefined ? {} : { title: parsed.title }),
|
|
1024
|
-
...(parsed.text === undefined ? {} : { text: parsed.text }),
|
|
1025
|
-
...(parsed.scrollY === undefined ? {} : { scrollY: parsed.scrollY })
|
|
1026
|
-
};
|
|
1027
|
-
};
|
|
1028
|
-
}
|
|
1029
|
-
/**
|
|
1030
|
-
* Read the running browser's actual outer-window bounds and CSS layout viewport through the
|
|
1031
|
-
* already-enabled local Chrome DevTools endpoint. The returned values come from `window.*` in
|
|
1032
|
-
* the target page; requested E2B resolution is deliberately not an input to this function.
|
|
1033
|
-
* Missing channels report their reason via `onUnavailable`, so the geometry warning can name
|
|
1034
|
-
* the cause (a dead CDP endpoint, no python3) instead of only the symptom. Returns `undefined`
|
|
1035
|
-
* only when neither channel could be measured.
|
|
1036
|
-
* Outer bounds and CSS dimensions are independent channels: a background page can report zero
|
|
1037
|
-
* outer dimensions while still reporting a CSS viewport. Final captures follow the active tab;
|
|
1038
|
-
* launch captures and emulation attribution keep the pinned target.
|
|
1039
|
-
*/
|
|
1040
|
-
export function makeChromeDesktopGeometryObserver(desktop, requestTimeoutMs, endpoint, targetId, onUnavailable, prefer = "pinned") {
|
|
1041
|
-
return async () => {
|
|
1042
|
-
const result = await desktop.commands.run(chromeCdpProbeCommand({ ...endpoint, ...(targetId === undefined ? {} : { targetId }), prefer, mode: "geometry" }), { requestTimeoutMs, timeoutMs: 5_000 });
|
|
1043
|
-
if (result.exitCode !== undefined && result.exitCode !== 0) {
|
|
1044
|
-
onUnavailable?.(`probe exited ${result.exitCode}: ${tailOf(result.stderr ?? result.stdout ?? "")}`);
|
|
1045
|
-
return undefined;
|
|
1046
|
-
}
|
|
1047
|
-
const parsed = parseChromeCdpProbeOutput(result.stdout);
|
|
1048
|
-
if (parsed.unavailable !== undefined) {
|
|
1049
|
-
onUnavailable?.(parsed.unavailable);
|
|
1050
|
-
return undefined;
|
|
1051
|
-
}
|
|
1052
|
-
const browserWindow = isMeasuredRect(parsed.browserWindow) ? { ...parsed.browserWindow, source: "cdp" } : undefined;
|
|
1053
|
-
const viewport = isMeasuredViewport(parsed.viewport) ? { ...parsed.viewport, source: "cdp" } : undefined;
|
|
1054
|
-
if (browserWindow === undefined && viewport === undefined) {
|
|
1055
|
-
onUnavailable?.("the page reported no usable window or viewport dimensions");
|
|
1056
|
-
return undefined;
|
|
1057
|
-
}
|
|
1058
|
-
if (browserWindow === undefined)
|
|
1059
|
-
onUnavailable?.("the page reported no usable outer-window dimensions");
|
|
1060
|
-
if (viewport === undefined)
|
|
1061
|
-
onUnavailable?.("the page reported no usable CSS viewport dimensions");
|
|
1062
|
-
return {
|
|
1063
|
-
...(browserWindow === undefined ? {} : { browserWindow }),
|
|
1064
|
-
...(viewport === undefined ? {} : { viewport }),
|
|
1065
|
-
...(parsed.targetId === undefined ? {} : { targetId: parsed.targetId })
|
|
1066
|
-
};
|
|
1067
|
-
};
|
|
1068
|
-
}
|
|
1069
|
-
/** The user agent a mobile-emulated lane presents unless the lab sets its own. */
|
|
1070
|
-
export const DEFAULT_MOBILE_USER_AGENT = "Mozilla/5.0 (iPhone; CPU iPhone OS 17_0 like Mac OS X) AppleWebKit/605.1.15 (KHTML, like Gecko) Version/17.0 Mobile/15E148 Safari/604.1";
|
|
1071
|
-
/**
|
|
1072
|
-
* Apply mobile emulation (#221) to the lane's launch page and read back what the page reports.
|
|
1073
|
-
* Fails CLOSED: a request that cannot be applied throws, because a desktop run labelled mobile is
|
|
1074
|
-
* the over-trust this feature exists to prevent. A read-back that cannot be taken is a warning
|
|
1075
|
-
* (the emulation was applied; only the proof is missing).
|
|
1076
|
-
*/
|
|
1077
|
-
export async function applyMobileEmulation(desktop, requestTimeoutMs, endpoint, targetId, request) {
|
|
1078
|
-
const command = (mode) => chromeCdpProbeCommand({ ...endpoint, ...(targetId === undefined ? {} : { targetId }), prefer: "pinned", mode, emulation: request });
|
|
1079
|
-
const read = async () => {
|
|
1080
|
-
const result = await desktop.commands.run(command("fidelity"), { requestTimeoutMs, timeoutMs: 15_000 });
|
|
1081
|
-
if (result.exitCode !== undefined && result.exitCode !== 0) {
|
|
1082
|
-
return { unavailable: `probe exited ${result.exitCode}: ${tailOf(result.stderr ?? result.stdout ?? "")}` };
|
|
1083
|
-
}
|
|
1084
|
-
return parseChromeCdpProbeOutput(result.stdout);
|
|
1085
|
-
};
|
|
1086
|
-
// The UA / touch / DPR overrides are bound to the DevTools session that set them and lapse the
|
|
1087
|
-
// moment its socket closes (measured: only the viewport width survived a one-shot apply). So the
|
|
1088
|
-
// applier stays attached for the lane's whole life as a detached process; the sandbox teardown
|
|
1089
|
-
// ends it. Its first stdout line says what was applied.
|
|
1090
|
-
const holderName = `mobile-emulation-${Date.now().toString(36)}`;
|
|
1091
|
-
await startDetachedProcess(desktop, { name: holderName, command: command("hold"), requestTimeoutMs });
|
|
1092
|
-
let announced;
|
|
1093
|
-
for (let attempt = 0; attempt < 30 && announced === undefined; attempt += 1) {
|
|
1094
|
-
await new Promise((resolve) => setTimeout(resolve, 500));
|
|
1095
|
-
const log = await readDetachedLog(desktop, holderName, requestTimeoutMs).catch(() => "");
|
|
1096
|
-
const line = log.split("\n").find((candidate) => candidate.trim().startsWith("{"));
|
|
1097
|
-
if (line !== undefined)
|
|
1098
|
-
announced = parseChromeCdpProbeOutput(line);
|
|
1099
|
-
}
|
|
1100
|
-
if (announced === undefined) {
|
|
1101
|
-
throw new Error("mobile emulation could not be applied: the in-sandbox applier printed nothing within 15 s");
|
|
1102
|
-
}
|
|
1103
|
-
if (announced.unavailable !== undefined) {
|
|
1104
|
-
throw new Error(`mobile emulation could not be applied (${announced.unavailable}); applied before failing: ${(announced.applied ?? []).join(", ") || "nothing"}`);
|
|
1105
|
-
}
|
|
1106
|
-
const applied = announced;
|
|
1107
|
-
// Viewport/touch read-back proves context settings, not gesture equivalence. Two hosted
|
|
1108
|
-
// replicas and a native-X conversion-toggle control reproduced reset click counts (#676).
|
|
1109
|
-
const warnings = request.touch
|
|
1110
|
-
? ["Mobile emulation uses desktop pointer-to-touch conversion, which can differ for repeated taps. Confirm gesture failures with direct or native touch input before attributing them to the app."]
|
|
1111
|
-
: [];
|
|
1112
|
-
// The reload inside the applier takes a moment; the read-back is retried until the page reports
|
|
1113
|
-
// the requested viewport and user agent, so a slow page does not read as "no proof".
|
|
1114
|
-
let readBack = await read();
|
|
1115
|
-
for (let attempt = 0; attempt < 20 && (readBack.fidelity === undefined || readBack.fidelity.innerWidth !== request.width || !readBack.fidelity.userAgent.includes(request.userAgent.slice(0, 24))); attempt += 1) {
|
|
1116
|
-
await new Promise((resolve) => setTimeout(resolve, 500));
|
|
1117
|
-
readBack = await read();
|
|
1118
|
-
}
|
|
1119
|
-
const fidelityRead = readBack;
|
|
1120
|
-
const requested = {
|
|
1121
|
-
width: request.width,
|
|
1122
|
-
height: request.height,
|
|
1123
|
-
deviceScaleFactor: request.deviceScaleFactor,
|
|
1124
|
-
touch: request.touch,
|
|
1125
|
-
userAgent: request.userAgent
|
|
1126
|
-
};
|
|
1127
|
-
const emulatedTargetId = applied.targetId ?? fidelityRead.targetId;
|
|
1128
|
-
if (fidelityRead.fidelity === undefined) {
|
|
1129
|
-
warnings.push(`Mobile emulation was applied but the page's own report could not be read (${fidelityRead.unavailable ?? "no fidelity read"}); desktopGeometry.fidelity carries the request without a resolved block.`);
|
|
1130
|
-
return { fidelity: { tier: "mobile-emulated", requested, applied: applied.applied ?? [] }, warnings, holderName, ...(emulatedTargetId === undefined ? {} : { targetId: emulatedTargetId }) };
|
|
1131
|
-
}
|
|
1132
|
-
const resolved = { ...fidelityRead.fidelity, source: "cdp" };
|
|
1133
|
-
if (resolved.innerWidth !== request.width) {
|
|
1134
|
-
warnings.push(`Mobile emulation requested a ${request.width} px viewport; the page reports ${resolved.innerWidth} px.`);
|
|
1135
|
-
}
|
|
1136
|
-
if (resolved.devicePixelRatio !== request.deviceScaleFactor) {
|
|
1137
|
-
warnings.push(`Mobile emulation requested devicePixelRatio ${request.deviceScaleFactor}; the page reports ${resolved.devicePixelRatio}.`);
|
|
1138
|
-
}
|
|
1139
|
-
if (request.touch && resolved.maxTouchPoints === 0) {
|
|
1140
|
-
warnings.push("Mobile emulation requested touch; the page reports navigator.maxTouchPoints 0.");
|
|
1141
|
-
}
|
|
1142
|
-
if (!resolved.userAgent.includes("Mobile") && !resolved.userAgent.includes("Android") && !resolved.userAgent.includes("iPhone")) {
|
|
1143
|
-
warnings.push("Mobile emulation requested a mobile user agent; the page reports a desktop one.");
|
|
1144
|
-
}
|
|
1145
|
-
return { fidelity: { tier: "mobile-emulated", requested, applied: applied.applied ?? [], resolved }, warnings, holderName, ...(emulatedTargetId === undefined ? {} : { targetId: emulatedTargetId }) };
|
|
1146
|
-
}
|
|
1147
|
-
function isMeasuredRect(value) {
|
|
1148
|
-
if (!value || typeof value !== "object")
|
|
1149
|
-
return false;
|
|
1150
|
-
const record = value;
|
|
1151
|
-
return Number.isFinite(record.x)
|
|
1152
|
-
&& Number.isFinite(record.y)
|
|
1153
|
-
&& isPositiveMeasurement(record.width)
|
|
1154
|
-
&& isPositiveMeasurement(record.height);
|
|
1155
|
-
}
|
|
1156
|
-
function isMeasuredViewport(value) {
|
|
1157
|
-
if (!value || typeof value !== "object")
|
|
1158
|
-
return false;
|
|
1159
|
-
const record = value;
|
|
1160
|
-
return isPositiveMeasurement(record.width)
|
|
1161
|
-
&& isPositiveMeasurement(record.height)
|
|
1162
|
-
&& isPositiveMeasurement(record.deviceScaleFactor);
|
|
1163
|
-
}
|
|
1164
|
-
function isPositiveMeasurement(value) {
|
|
1165
|
-
return typeof value === "number" && Number.isFinite(value) && value > 0;
|
|
1166
|
-
}
|
|
1167
|
-
async function measureBrowserWindowWithXwininfo(desktop, windowId, requestTimeoutMs) {
|
|
1168
|
-
const result = await desktop.commands.run([
|
|
1169
|
-
"set -euo pipefail",
|
|
1170
|
-
`win=${shellSingleQuote(windowId)}`,
|
|
1171
|
-
// Older xdotool builds translate parent-relative offsets twice. With window
|
|
1172
|
-
// decorations that falsely reports a visible client as clipped, triggering
|
|
1173
|
-
// fullscreen and hiding the participant's address bar. Read root-relative
|
|
1174
|
-
// client coordinates directly; never substitute emulated CDP outer bounds.
|
|
1175
|
-
"LC_ALL=C xwininfo -id \"$win\" -stats 2>/dev/null"
|
|
1176
|
-
].join("\n"), { requestTimeoutMs, timeoutMs: 5_000 });
|
|
1177
|
-
if (result.exitCode !== undefined && result.exitCode !== 0)
|
|
1178
|
-
return undefined;
|
|
1179
|
-
return parseXwininfoGeometry(result.stdout ?? "");
|
|
1180
|
-
}
|
|
1181
|
-
/** Root-relative physical client bounds from xwininfo's C-locale stats. */
|
|
1182
|
-
export function parseXwininfoGeometry(output) {
|
|
1183
|
-
const read = (label) => {
|
|
1184
|
-
const matches = [...output.matchAll(new RegExp(`^\\s*${label}:\\s*(-?\\d+)\\s*$`, "gm"))];
|
|
1185
|
-
if (matches.length !== 1)
|
|
1186
|
-
return undefined;
|
|
1187
|
-
const value = Number(matches[0][1]);
|
|
1188
|
-
return Number.isSafeInteger(value) ? value : undefined;
|
|
1189
|
-
};
|
|
1190
|
-
const x = read("Absolute upper-left X");
|
|
1191
|
-
const y = read("Absolute upper-left Y");
|
|
1192
|
-
const width = read("Width");
|
|
1193
|
-
const height = read("Height");
|
|
1194
|
-
const mapStates = [...output.matchAll(/^\s*Map State:\s*(\S+)\s*$/gm)];
|
|
1195
|
-
if (mapStates.length !== 1 || mapStates[0][1] !== "IsViewable")
|
|
1196
|
-
return undefined;
|
|
1197
|
-
if (x === undefined || y === undefined || width === undefined || height === undefined || width <= 0 || height <= 0) {
|
|
1198
|
-
return undefined;
|
|
1199
|
-
}
|
|
1200
|
-
return { x, y, width, height, source: "xwininfo" };
|
|
1201
|
-
}
|
|
1202
|
-
/** Physical X client bounds, never the page's emulated window.outerWidth/Height. */
|
|
1203
|
-
function isBrowserWindowContained(bounds, [width, height]) {
|
|
1204
|
-
return bounds.x >= 0 && bounds.y >= 0
|
|
1205
|
-
&& bounds.x + bounds.width <= width && bounds.y + bounds.height <= height;
|
|
1206
|
-
}
|
|
1207
|
-
/** Bounded repair. Resizing can clear a window-manager maximize state and move the
|
|
1208
|
-
* client origin as decorations return, so remeasure before a second adjustment. */
|
|
1209
|
-
async function fitBrowserWindowWithinDesktop(desktop, windowId, resolution, requestTimeoutMs) {
|
|
1210
|
-
const run = (command) => desktop.commands.run([
|
|
1211
|
-
"set -euo pipefail",
|
|
1212
|
-
`win=${shellSingleQuote(windowId)}`,
|
|
1213
|
-
command
|
|
1214
|
-
].join("\n"), { requestTimeoutMs, timeoutMs: 5_000 }).catch(() => undefined);
|
|
1215
|
-
await run('xdotool windowmove "$win" 0 0');
|
|
1216
|
-
await desktop.wait(250).catch(() => undefined);
|
|
1217
|
-
const moved = await measureBrowserWindowWithXwininfo(desktop, windowId, requestTimeoutMs).catch(() => undefined);
|
|
1218
|
-
if (moved === undefined)
|
|
1219
|
-
return moved;
|
|
1220
|
-
let resized = moved;
|
|
1221
|
-
// Resizing alone cannot fix an offscreen client origin. The window manager
|
|
1222
|
-
// can also center a minimum-width client at a negative x on a narrow screen.
|
|
1223
|
-
for (let attempt = 0; attempt < 2; attempt += 1) {
|
|
1224
|
-
if (isBrowserWindowContained(resized, resolution))
|
|
1225
|
-
return resized;
|
|
1226
|
-
const width = resolution[0] - resized.x;
|
|
1227
|
-
const height = resolution[1] - resized.y;
|
|
1228
|
-
if (resized.x < 0 || resized.y < 0 || width <= 0 || height <= 0)
|
|
1229
|
-
break;
|
|
1230
|
-
await run(`xdotool windowsize "$win" ${width} ${height}`);
|
|
1231
|
-
await desktop.wait(250).catch(() => undefined);
|
|
1232
|
-
const measured = await measureBrowserWindowWithXwininfo(desktop, windowId, requestTimeoutMs).catch(() => undefined);
|
|
1233
|
-
if (measured === undefined)
|
|
1234
|
-
return measured;
|
|
1235
|
-
resized = measured;
|
|
1236
|
-
}
|
|
1237
|
-
if (resized === undefined || isBrowserWindowContained(resized, resolution))
|
|
1238
|
-
return resized;
|
|
1239
|
-
// Chrome's minimum client width can equal the whole desktop. Window-manager
|
|
1240
|
-
// borders then make a decorated window impossible to contain, even after a
|
|
1241
|
-
// successful move/resize. Request fullscreen once and prove the physical result.
|
|
1242
|
-
// xprop/xdotool ship with the desktop template; wmctrl is not required.
|
|
1243
|
-
// Check state first so the fullscreen shortcut cannot toggle an existing state off.
|
|
1244
|
-
await run([
|
|
1245
|
-
'state=$(xprop -id "$win" _NET_WM_STATE)',
|
|
1246
|
-
'case "$state" in',
|
|
1247
|
-
' *_NET_WM_STATE_FULLSCREEN*) ;;',
|
|
1248
|
-
' *) xdotool windowactivate --sync "$win"; xdotool key --clearmodifiers F11 ;;',
|
|
1249
|
-
'esac'
|
|
1250
|
-
].join("\n"));
|
|
1251
|
-
// The fullscreen animation may report its new origin before its final width.
|
|
1252
|
-
// Give the window manager a bounded settling window, keeping missing reads unverified.
|
|
1253
|
-
for (let attempt = 0; attempt < 4; attempt += 1) {
|
|
1254
|
-
await desktop.wait(250).catch(() => undefined);
|
|
1255
|
-
const measured = await measureBrowserWindowWithXwininfo(desktop, windowId, requestTimeoutMs).catch(() => undefined);
|
|
1256
|
-
if (measured === undefined || isBrowserWindowContained(measured, resolution))
|
|
1257
|
-
return measured;
|
|
1258
|
-
resized = measured;
|
|
1259
|
-
}
|
|
1260
|
-
return resized;
|
|
1261
|
-
}
|
|
1262
|
-
/** Shared hosted-browser geometry capture used by per-lane and sequential shared-world routes. */
|
|
1263
|
-
export async function captureDesktopBrowserGeometry(args) {
|
|
1264
|
-
const warnings = [];
|
|
1265
|
-
let browserWindowId = args.browserWindowId;
|
|
1266
|
-
if (browserWindowId === undefined && args.browserFamily !== "unknown") {
|
|
1267
|
-
browserWindowId = await findVisibleBrowserWindowId(args.desktop, args.requestTimeoutMs, args.browserFamily, args.launchIdentity).catch((error) => {
|
|
1268
|
-
warnings.push(`Browser window lookup failed for lane ${args.laneId}: ${redactText(toErrorMessage(error))}`);
|
|
1269
|
-
return undefined;
|
|
1270
|
-
});
|
|
1271
|
-
}
|
|
1272
|
-
let physicalWindow;
|
|
1273
|
-
if (browserWindowId !== undefined) {
|
|
1274
|
-
if (args.resize !== false) {
|
|
1275
|
-
await fillDesktopBrowserWindow(args.desktop, browserWindowId, args.requestedScreen, args.requestTimeoutMs);
|
|
1276
|
-
// Let the window manager apply the resize before querying both X and page layout geometry.
|
|
1277
|
-
await args.desktop.wait(250).catch(() => undefined);
|
|
1278
|
-
}
|
|
1279
|
-
physicalWindow = await measureBrowserWindowWithXwininfo(args.desktop, browserWindowId, args.requestTimeoutMs)
|
|
1280
|
-
.catch(() => undefined);
|
|
1281
|
-
}
|
|
1282
|
-
else {
|
|
1283
|
-
warnings.push(`Browser window bounds could not be measured for lane ${args.laneId}; the live stream will use the full desktop.`);
|
|
1284
|
-
}
|
|
1285
|
-
let unusable;
|
|
1286
|
-
if (physicalWindow !== undefined && !isBrowserWindowContained(physicalWindow, args.requestedScreen)) {
|
|
1287
|
-
const before = physicalWindow;
|
|
1288
|
-
if (args.resize !== false && browserWindowId !== undefined) {
|
|
1289
|
-
physicalWindow = await fitBrowserWindowWithinDesktop(args.desktop, browserWindowId, args.requestedScreen, args.requestTimeoutMs);
|
|
1290
|
-
if (physicalWindow === undefined) {
|
|
1291
|
-
// Keep the last measured bad state; a missing observation cannot prove a successful fix.
|
|
1292
|
-
physicalWindow = before;
|
|
1293
|
-
unusable = `Physical browser containment could not be verified after correction for lane ${args.laneId}; the last measured window was clipped.`;
|
|
1294
|
-
}
|
|
1295
|
-
else if (isBrowserWindowContained(physicalWindow, args.requestedScreen)) {
|
|
1296
|
-
warnings.push(`Browser window clipping corrected for lane ${args.laneId}; physical bounds are ${physicalWindow.width}x${physicalWindow.height} at (${physicalWindow.x}, ${physicalWindow.y}).`);
|
|
1297
|
-
}
|
|
1298
|
-
}
|
|
1299
|
-
if (unusable === undefined && !isBrowserWindowContained(physicalWindow, args.requestedScreen)) {
|
|
1300
|
-
unusable = `Browser window is outside the captured ${args.requestedScreen[0]}x${args.requestedScreen[1]} desktop for lane ${args.laneId}: physical bounds ${physicalWindow.width}x${physicalWindow.height} at (${physicalWindow.x}, ${physicalWindow.y}), right=${physicalWindow.x + physicalWindow.width}, bottom=${physicalWindow.y + physicalWindow.height}.`;
|
|
1301
|
-
}
|
|
1302
|
-
if (unusable !== undefined)
|
|
1303
|
-
warnings.push(unusable);
|
|
1304
|
-
}
|
|
1305
|
-
if (physicalWindow === undefined) {
|
|
1306
|
-
warnings.push(`Physical browser containment is unverified for lane ${args.laneId}; X window bounds could not be measured. Page-reported outer dimensions can be emulated and do not prove physical visibility.`);
|
|
1307
|
-
}
|
|
1308
|
-
let cdpUnavailable;
|
|
1309
|
-
const chromeGeometry = args.browserFamily === "chromium"
|
|
1310
|
-
? await makeChromeDesktopGeometryObserver(args.desktop, args.requestTimeoutMs, {
|
|
1311
|
-
...(args.launchIdentity?.cdpPort === undefined ? {} : { cdpPort: args.launchIdentity.cdpPort }),
|
|
1312
|
-
...(args.launchIdentity?.profileDir === undefined ? {} : { profileDir: args.launchIdentity.profileDir }),
|
|
1313
|
-
targetUrl: args.targetUrl
|
|
1314
|
-
}, args.browserTargetId, (reason) => {
|
|
1315
|
-
cdpUnavailable = reason;
|
|
1316
|
-
}, args.pagePreference ?? "pinned")().catch((error) => {
|
|
1317
|
-
cdpUnavailable = toErrorMessage(error);
|
|
1318
|
-
return undefined;
|
|
1319
|
-
})
|
|
1320
|
-
: undefined;
|
|
1321
|
-
const browserWindow = physicalWindow ?? chromeGeometry?.browserWindow;
|
|
1322
|
-
const viewport = chromeGeometry?.viewport;
|
|
1323
|
-
// The fill check reads the X window when it was measured: under mobile emulation (#221) the
|
|
1324
|
-
// page's window.outerWidth reports the EMULATED screen (414), which is not a fill failure.
|
|
1325
|
-
const fillBounds = physicalWindow;
|
|
1326
|
-
if (!browserWindow) {
|
|
1327
|
-
warnings.push(`Browser outer bounds could not be measured for lane ${args.laneId}.`);
|
|
1328
|
-
}
|
|
1329
|
-
else if (unusable === undefined && fillBounds !== undefined && (fillBounds.x !== 0 || fillBounds.y !== 0 || fillBounds.width !== args.requestedScreen[0] || fillBounds.height !== args.requestedScreen[1])) {
|
|
1330
|
-
warnings.push(`Browser window fill did not reach the requested ${args.requestedScreen[0]}x${args.requestedScreen[1]} screen for lane ${args.laneId}; measured physical bounds are ${fillBounds.width}x${fillBounds.height} at (${fillBounds.x}, ${fillBounds.y}).`);
|
|
1331
|
-
}
|
|
1332
|
-
if (!viewport) {
|
|
1333
|
-
// Name the cause, not only the symptom: the same dead DevTools channel that loses the viewport
|
|
1334
|
-
// loses every url/text observation, and a reader of the bundle should learn that here (#514).
|
|
1335
|
-
const cause = cdpUnavailable === undefined ? "" : ` DevTools probe: ${redactText(cdpUnavailable)}.`;
|
|
1336
|
-
warnings.push(args.browserFamily === "firefox"
|
|
1337
|
-
? `Browser CSS viewport measurement is unavailable for Firefox on lane ${args.laneId}; stream.viewport is omitted instead of reading a different browser's CDP endpoint.`
|
|
1338
|
-
: `Browser CSS viewport could not be measured for lane ${args.laneId}; stream.viewport is omitted instead of copying the requested screen resolution.${cause}`);
|
|
1339
|
-
}
|
|
1340
|
-
return {
|
|
1341
|
-
...(unusable === undefined ? {} : { unusable }),
|
|
1342
|
-
...(browserWindowId === undefined ? {} : { browserWindowId }),
|
|
1343
|
-
...(chromeGeometry?.targetId === undefined ? {} : { browserTargetId: chromeGeometry.targetId }),
|
|
1344
|
-
...(browserWindow === undefined ? {} : { browserWindow }),
|
|
1345
|
-
...(viewport === undefined ? {} : { viewport }),
|
|
1346
|
-
warnings
|
|
1347
|
-
};
|
|
1348
|
-
}
|
|
1349
|
-
function shellSingleQuote(value) {
|
|
1350
|
-
return `'${value.replace(/'/g, "'\\''")}'`;
|
|
1351
|
-
}
|
|
1352
|
-
/**
|
|
1353
|
-
* Prepare a CLI study's runtime and, only when declared, its product (#495, #515).
|
|
1354
|
-
*
|
|
1355
|
-
* The install runs UNKEYED and before the session starts, for the same reason the clone route
|
|
1356
|
-
* provisions its subject first: what is being studied begins when the participant looks at the
|
|
1357
|
-
* screen. Omitting install deliberately studies product installation; Node/npm remain a
|
|
1358
|
-
* harness prerequisite so the participant can follow the product's public npm instructions.
|
|
1359
|
-
*/
|
|
1360
|
-
async function provisionDesktopCli(desktop, args) {
|
|
1361
|
-
const install = args.install;
|
|
1362
|
-
const now = () => Date.now();
|
|
1363
|
-
if (install === undefined || needsNodeRuntime([install])) {
|
|
1364
|
-
const startedAt = now();
|
|
1365
|
-
emitPhaseStarted(args.onPhase, now, "runtime", "providing Node/npm for the desktop CLI study");
|
|
1366
|
-
const bootstrap = await runDetachedStep(desktop, {
|
|
1367
|
-
name: "desktop-cli-runtime-node",
|
|
1368
|
-
command: TERMINAL_NODE_BOOTSTRAP_COMMAND,
|
|
1369
|
-
cwd: "/home/user",
|
|
1370
|
-
timeoutMs: INSTALL_TIMEOUT_MS,
|
|
1371
|
-
requestTimeoutMs: args.requestTimeoutMs
|
|
1372
|
-
});
|
|
1373
|
-
emitPhaseCompleted(args.onPhase, now, startedAt, "runtime", bootstrap.ok, bootstrap.ok
|
|
1374
|
-
? "Node runtime ready"
|
|
1375
|
-
: "Node runtime bootstrap failed");
|
|
1376
|
-
if (!bootstrap.ok) {
|
|
1377
|
-
throw new Error(`desktop-cli runtime bootstrap failed for "${args.product}"`);
|
|
1378
|
-
}
|
|
1379
|
-
}
|
|
1380
|
-
if (install === undefined)
|
|
1381
|
-
return;
|
|
1382
|
-
const startedAt = now();
|
|
1383
|
-
emitPhaseStarted(args.onPhase, now, "install", `installing ${args.product} on the desktop`);
|
|
1384
|
-
const result = await runDetachedStep(desktop, {
|
|
1385
|
-
name: "desktop-cli-install",
|
|
1386
|
-
command: install,
|
|
1387
|
-
cwd: "/home/user",
|
|
1388
|
-
timeoutMs: INSTALL_TIMEOUT_MS,
|
|
1389
|
-
requestTimeoutMs: args.requestTimeoutMs
|
|
1390
|
-
});
|
|
1391
|
-
emitPhaseCompleted(args.onPhase, now, startedAt, "install", result.ok, result.ok
|
|
1392
|
-
? `${args.product} installed`
|
|
1393
|
-
: `installing ${args.product} failed`);
|
|
1394
|
-
if (!result.ok) {
|
|
1395
|
-
// Fail closed: a participant handed a desktop where the product is not installed would produce
|
|
1396
|
-
// a transcript about a missing command, and that finding belongs to the harness, not the tool.
|
|
1397
|
-
// The tail rides along, scrubbed before truncation like every other provisioning failure — a
|
|
1398
|
-
// bare "install failed" is unactionable to whoever wrote the command.
|
|
1399
|
-
throw new Error(args.scrub(`desktop-cli install failed for "${args.product}" (${result.timedOut ? "timed out" : `exit ${result.exitCode ?? "?"}`}): ${tailOf(args.scrub(result.logTail))}`));
|
|
1400
|
-
}
|
|
1401
|
-
}
|
|
1402
|
-
/**
|
|
1403
|
-
* Open a terminal window on the desktop.
|
|
1404
|
-
*
|
|
1405
|
-
* The stock template is XFCE and ships xfce4-terminal (also aliased x-terminal-emulator), verified
|
|
1406
|
-
* live before this route was built. `x-terminal-emulator` is tried first so a template that swaps
|
|
1407
|
-
* the emulator still works; a desktop with neither is a template problem and fails closed rather
|
|
1408
|
-
* than handing a participant an empty screen and calling it a study.
|
|
1409
|
-
*/
|
|
1410
|
-
async function openDesktopTerminal(desktop, requestTimeoutMs, workdir) {
|
|
1411
|
-
const dir = workdir ?? "/home/user";
|
|
1412
|
-
const result = await runDetachedStep(desktop, {
|
|
1413
|
-
name: "desktop-cli-terminal",
|
|
1414
|
-
command: [
|
|
1415
|
-
"for candidate in x-terminal-emulator xfce4-terminal gnome-terminal konsole xterm; do",
|
|
1416
|
-
' if command -v "$candidate" >/dev/null 2>&1; then',
|
|
1417
|
-
// LANG is set on the terminal we open, not globally: the stock image declares no locale, and
|
|
1418
|
-
// a study that measures our own mojibake against an unconfigured template would be measuring
|
|
1419
|
-
// the template. The PRODUCT-side fix (an ASCII fallback when the locale is not UTF-8) is in
|
|
1420
|
-
// src/terminal-encoding.ts, and it is the one that matters for real users.
|
|
1421
|
-
` (cd ${shellSingleQuote(dir)} 2>/dev/null || cd /home/user; DISPLAY=:0 LANG=C.UTF-8 LC_ALL=C.UTF-8 HUMANISH_STUDY_PARTICIPANT=1 nohup "$candidate" >/dev/null 2>&1 &)`,
|
|
1422
|
-
" sleep 3",
|
|
1423
|
-
' echo "humanish: opened $candidate"',
|
|
1424
|
-
" exit 0",
|
|
1425
|
-
" fi",
|
|
1426
|
-
"done",
|
|
1427
|
-
"echo 'humanish: no terminal emulator on this desktop template' >&2",
|
|
1428
|
-
"exit 1"
|
|
1429
|
-
].join("\n"),
|
|
1430
|
-
cwd: "/home/user",
|
|
1431
|
-
timeoutMs: 60_000,
|
|
1432
|
-
requestTimeoutMs
|
|
1433
|
-
});
|
|
1434
|
-
if (!result.ok) {
|
|
1435
|
-
throw new Error("desktop-cli lane could not open a terminal on this desktop template");
|
|
1436
|
-
}
|
|
1437
|
-
}
|
|
1438
|
-
async function startDesktopStream(desktop, browserWindowId) {
|
|
1439
|
-
if (!browserWindowId) {
|
|
1440
|
-
await desktop.stream.start({ requireAuth: true });
|
|
1441
|
-
return;
|
|
1442
|
-
}
|
|
1443
|
-
try {
|
|
1444
|
-
await desktop.stream.start({ requireAuth: true, windowId: browserWindowId });
|
|
1445
|
-
}
|
|
1446
|
-
catch {
|
|
1447
|
-
await desktop.stream.start({ requireAuth: true });
|
|
1448
|
-
}
|
|
1449
|
-
}
|
|
1450
544
|
// "can't" followed by a PERCEPTION verb describes what the screen showed, not an inability to
|
|
1451
545
|
// proceed: "the canvas truncates it so you can't even read the whole thing", "I can't tell from
|
|
1452
546
|
// the screen whether the rename is persisted", "so I could not read its full description". Five
|
|
@@ -1656,118 +750,18 @@ export function resolveSelfReportedFriction(session) {
|
|
|
1656
750
|
return reports.join("\n\n");
|
|
1657
751
|
return undefined;
|
|
1658
752
|
}
|
|
1659
|
-
/**
|
|
1660
|
-
*
|
|
1661
|
-
* resolution), prepareDesktop, verify geometry, (clone+serve+seed the subject per lane), open the
|
|
1662
|
-
* browser, run the session, and ALWAYS tear down THIS lane's sandbox BY ID in a finally. Never
|
|
1663
|
-
* enumerates sandboxes. Extracted from the former single-lane block; at N=1 it writes the exact
|
|
1664
|
-
* same artifacts (actor.json, screenshots/<name>) the bundle has always referenced.
|
|
1665
|
-
*/
|
|
753
|
+
/** Run one participant against a prepared desktop. The adapter owns provisioning, final
|
|
754
|
+
* evidence and cleanup; this runner owns the model, trace and participant outcome. */
|
|
1666
755
|
export async function runCuaLane(spec, deps) {
|
|
1667
|
-
const { config,
|
|
1668
|
-
const desktopCliRoute = deps.desktopCliRoute === true;
|
|
1669
|
-
// The local brain, when there is one. `appServer` / `claudeSession` own a process, so the lane
|
|
1670
|
-
// closes it.
|
|
756
|
+
const { config, env } = deps;
|
|
1671
757
|
let appServer;
|
|
1672
758
|
let claudeSession;
|
|
1673
759
|
let localAgentProvider;
|
|
1674
|
-
const subjectEnvValues = config.subject.envValues ?? {};
|
|
1675
|
-
const targetUrl = spec.targetUrl ?? appUrl;
|
|
1676
|
-
const env = deps.env;
|
|
1677
|
-
// Off-app comms (#297): on an in-sandbox subject route, redirect the app's email-API sends into an
|
|
1678
|
-
// in-sandbox catch (loopback) so its verification mail is CAPTURED, not sent to the internet. Gated
|
|
1679
|
-
// ENTIRELY on config.comms — no comms declared → zero change. The base-URL env is injected at
|
|
1680
|
-
// sandbox-create (below, so the app reads it at boot); the catch is started right after create.
|
|
1681
|
-
const commsEmail = (cloneRoute || localTreeRoute) && config.comms?.email?.kind === "fake" ? config.comms.email : undefined;
|
|
1682
|
-
const commsPort = commsEmail ? (commsEmail.port ?? DEFAULT_SANDBOX_CATCH_PORT) : undefined;
|
|
1683
|
-
// Hoisted so the finally can drain the catch before teardown; `commsArtifactPath` is the written
|
|
1684
|
-
// evidence path folded into the lane outcome.
|
|
1685
|
-
let deployedComms;
|
|
1686
|
-
let commsArtifactPath;
|
|
1687
|
-
let receivingInboxUrl;
|
|
1688
|
-
// injectEnv is absent on an adopter-hosted plane (#328): there is no subject env to inject
|
|
1689
|
-
// because the operator points their own app at their own catch.
|
|
1690
|
-
const commsEnv = commsEmail?.injectEnv !== undefined && commsPort !== undefined
|
|
1691
|
-
? { [commsEmail.injectEnv]: `http://127.0.0.1:${commsPort}` }
|
|
1692
|
-
: {};
|
|
1693
|
-
// SMTP transport: the same idea as injectEnv, but an app that speaks SMTP needs a host and a port
|
|
1694
|
-
// rather than a base URL. The catch accepts any credentials (loopback only), yet many apps refuse
|
|
1695
|
-
// to boot unless the user/password vars exist at all, so those are injected when declared.
|
|
1696
|
-
const commsSmtpPort = commsEmail?.smtp?.port;
|
|
1697
|
-
if (commsEmail?.smtp && commsSmtpPort !== undefined) {
|
|
1698
|
-
commsEnv[commsEmail.smtp.hostEnv] = "127.0.0.1";
|
|
1699
|
-
commsEnv[commsEmail.smtp.portEnv] = String(commsSmtpPort);
|
|
1700
|
-
if (commsEmail.smtp.userEnv)
|
|
1701
|
-
commsEnv[commsEmail.smtp.userEnv] = commsEmail.smtp.user ?? "humanish";
|
|
1702
|
-
if (commsEmail.smtp.passwordEnv)
|
|
1703
|
-
commsEnv[commsEmail.smtp.passwordEnv] = commsEmail.smtp.password ?? "humanish";
|
|
1704
|
-
}
|
|
1705
|
-
// Persona inbox SURFACE (#297 slice B): the loopback URL the persona opens to read captured mail; the
|
|
1706
|
-
// origin-rewrite map (identity on this same-sandbox route, but covers localhost/0.0.0.0 alias skew + an
|
|
1707
|
-
// operator-declared linkOrigin); and a disposable background loop that renders the surface DURING the
|
|
1708
|
-
// session so the inbox is live when the persona checks. The surface uses its OWN FakeInbox + cursor,
|
|
1709
|
-
// independent of the teardown evidence drain (two readers of the append-only NDJSON — no double-count).
|
|
1710
|
-
const commsInboxUrl = commsEmail && commsPort !== undefined ? `http://127.0.0.1:${commsPort}/inbox` : undefined;
|
|
1711
|
-
const commsOriginMap = commsEmail
|
|
1712
|
-
? buildOriginMap({
|
|
1713
|
-
...(config.subject.serve?.url === undefined ? {} : { internalServeUrl: config.subject.serve.url }),
|
|
1714
|
-
reachableBaseUrl: targetUrl,
|
|
1715
|
-
...(commsEmail.linkOrigin === undefined ? {} : { linkOrigin: commsEmail.linkOrigin })
|
|
1716
|
-
})
|
|
1717
|
-
: [];
|
|
1718
|
-
const surfaceRecipients = (commsEmail?.recipients ?? [])
|
|
1719
|
-
.filter((recipient) => recipient.address !== undefined)
|
|
1720
|
-
.map((recipient) => ({ lane: recipient.lane, address: recipient.address }));
|
|
1721
|
-
let surfaceRenderedCount = 0;
|
|
1722
|
-
let surfaceDisposed = false;
|
|
1723
|
-
let releaseSurface = () => { };
|
|
1724
|
-
const surfaceDispose = new Promise((resolve) => { releaseSurface = resolve; });
|
|
1725
|
-
let surfaceLoop;
|
|
1726
760
|
const warnings = [];
|
|
1727
761
|
const screenshots = [];
|
|
1728
762
|
const writeScreenshot = makeLaneWriteScreenshot(deps.artifactRoot, spec, screenshots);
|
|
1729
|
-
const stateStepRecords = [];
|
|
1730
|
-
// Completed-only trail (durationMs/ok are set on completed events, never on started ones):
|
|
1731
|
-
// this is what survives into bundle.events. The default/injected sink below sees EVERY event,
|
|
1732
|
-
// started and completed alike, so an operator watching stderr sees both halves of each phase.
|
|
1733
|
-
const phaseRecords = [];
|
|
1734
|
-
const onSubjectPhase = (event) => {
|
|
1735
|
-
if (event.ok !== undefined) {
|
|
1736
|
-
phaseRecords.push(event);
|
|
1737
|
-
}
|
|
1738
|
-
(deps.hooks.onPhase ?? defaultSubjectPhaseSink)(event, { laneId: spec.laneId, laneCount: deps.laneCount });
|
|
1739
|
-
};
|
|
1740
763
|
let session;
|
|
1741
764
|
let sessionError;
|
|
1742
|
-
let failureCode;
|
|
1743
|
-
let sandboxId;
|
|
1744
|
-
// Host-side E2B desktop billed-span endpoints, measured via the injected clock. Captured right
|
|
1745
|
-
// after create() succeeds and again in the finally after teardown resolves (both the killed and
|
|
1746
|
-
// kept-for-debug paths). This measured span excludes allocation before the acquired handle;
|
|
1747
|
-
// a kept/unconfirmed allocation gets an extra unknown lifetime cost line.
|
|
1748
|
-
let sandboxCreatedAtMs;
|
|
1749
|
-
let sandboxTornDownAtMs;
|
|
1750
|
-
let desktopResources;
|
|
1751
|
-
let killed = false;
|
|
1752
|
-
let streamUrl;
|
|
1753
|
-
let subjectCommit;
|
|
1754
|
-
let desktopBrowser;
|
|
1755
|
-
let launchedBrowserFamily = "unknown";
|
|
1756
|
-
let browserLaunchIdentity;
|
|
1757
|
-
let browserLaunched = false;
|
|
1758
|
-
let initialBrowserGeometry;
|
|
1759
|
-
let appliedFidelity;
|
|
1760
|
-
let emulatedTargetId;
|
|
1761
|
-
let emulationHolderName;
|
|
1762
|
-
let browserWindowId;
|
|
1763
|
-
let browserTargetId;
|
|
1764
|
-
const declaredScreen = declaredScreenForRender(spec.devicePreset, spec.deviceName, spec.resolution);
|
|
1765
|
-
let desktopGeometry = {
|
|
1766
|
-
screen: {
|
|
1767
|
-
requested: { width: spec.resolution[0], height: spec.resolution[1] },
|
|
1768
|
-
...(declaredScreen ? { declared: declaredScreen } : {})
|
|
1769
|
-
}
|
|
1770
|
-
};
|
|
1771
765
|
let provisioned = false;
|
|
1772
766
|
let signaled = false;
|
|
1773
767
|
const signal = (ok) => {
|
|
@@ -1776,641 +770,153 @@ export async function runCuaLane(spec, deps) {
|
|
|
1776
770
|
deps.signalProvisioned(ok);
|
|
1777
771
|
}
|
|
1778
772
|
};
|
|
1779
|
-
|
|
1780
|
-
let desktop;
|
|
773
|
+
const desktopLane = deps.createDesktopLane?.(spec, warnings) ?? createE2BCuaDesktopLane(spec, deps, warnings);
|
|
1781
774
|
try {
|
|
1782
|
-
|
|
1783
|
-
//
|
|
1784
|
-
// the
|
|
1785
|
-
|
|
1786
|
-
|
|
1787
|
-
|
|
1788
|
-
|
|
1789
|
-
|
|
1790
|
-
...
|
|
1791
|
-
|
|
1792
|
-
|
|
1793
|
-
|
|
1794
|
-
laneIndex: String(spec.laneIndex),
|
|
1795
|
-
laneCount: String(deps.laneCount)
|
|
1796
|
-
},
|
|
1797
|
-
// Env placement per the doctrine: the ACTOR's key never enters the sandbox (the model drives
|
|
1798
|
-
// from outside). The SUBJECT's declared env NAMES are provisioned here on the clone route.
|
|
1799
|
-
// Three sources, in precedence order: committed non-secret config (subject.envValues), then
|
|
1800
|
-
// secret values forwarded from the caller's environment (subject.env), then the harness's own
|
|
1801
|
-
// comms wiring, which must win because only it knows the catch's address.
|
|
1802
|
-
...(subjectEnvNames.length > 0 || Object.keys(subjectEnvValues).length > 0 || Object.keys(commsEnv).length > 0
|
|
1803
|
-
? {
|
|
1804
|
-
envs: {
|
|
1805
|
-
...subjectEnvValues,
|
|
1806
|
-
...Object.fromEntries(subjectEnvNames.map((name) => [name, env[name]])),
|
|
1807
|
-
...commsEnv
|
|
1808
|
-
}
|
|
1809
|
-
}
|
|
1810
|
-
: {}),
|
|
1811
|
-
resolution: spec.resolution,
|
|
1812
|
-
dpi: 96,
|
|
1813
|
-
lifecycle: { onTimeout: "kill" }
|
|
1814
|
-
}, config.execution?.desktop?.template, {
|
|
1815
|
-
// The default loader reclaims an acquired handle before retrying failed desktop startup.
|
|
1816
|
-
// Its error names the cleanup outcome; pre-construction allocation failures remain unowned.
|
|
1817
|
-
onRetry: (reason) => {
|
|
1818
|
-
const named = redactText(deps.scrubKnownValues(reason));
|
|
1819
|
-
warnings.push(`Sandbox create for lane ${spec.laneId} retried once after a transient provider error (${named}).`);
|
|
1820
|
-
onSubjectPhase({ at: new Date(deps.now()).toISOString(), type: "cua-lab.sandbox.create.retry", message: `sandbox create retried once (${named})` });
|
|
1821
|
-
}
|
|
1822
|
-
});
|
|
1823
|
-
sandboxId = desktop.sandboxId;
|
|
1824
|
-
// #358 salvage: journal the id to disk before any work — an interrupted run reclaims by
|
|
1825
|
-
// exact recorded id (`humanish reclaim`), never by enumerating the account.
|
|
1826
|
-
await appendSandboxReceipt(deps.artifactRoot, { at: new Date(deps.now()).toISOString(), laneId: spec.laneId, sandboxId, timeoutMs: deps.perLaneSandboxMs });
|
|
1827
|
-
// The billed span starts the instant the sandbox exists.
|
|
1828
|
-
sandboxCreatedAtMs = deps.now();
|
|
1829
|
-
desktopResources = await observeDesktopResources(desktop);
|
|
1830
|
-
if ("reason" in desktopResources) {
|
|
1831
|
-
warnings.push(`Desktop resource size unavailable (${desktopResources.reason}); compute cost remains unpriced.`);
|
|
1832
|
-
}
|
|
1833
|
-
if (deps.hooks.prepareDesktop) {
|
|
1834
|
-
await deps.hooks.prepareDesktop(desktop, { laneId: spec.laneId, laneIndex: spec.laneIndex, laneCount: deps.laneCount });
|
|
1835
|
-
}
|
|
1836
|
-
// Start the in-sandbox email catch BEFORE the subject serve, so the app's send-API base URL (injected
|
|
1837
|
-
// into its env at create) resolves the moment it boots. A comms-declared lab that can't stand the
|
|
1838
|
-
// catch up is a setup failure (fail closed) rather than silently sending real mail.
|
|
1839
|
-
if (deps.receiving) {
|
|
1840
|
-
const surface = await deployReceivingInbox(desktop, { leaseId: spec.streamId, requestTimeoutMs: Math.min(deps.requestTimeoutMs, 30_000) });
|
|
1841
|
-
receivingInboxUrl = surface.url;
|
|
1842
|
-
const email = config.comms?.email;
|
|
1843
|
-
try {
|
|
1844
|
-
await deps.receiving.attach(spec.laneId, {
|
|
1845
|
-
surface,
|
|
1846
|
-
allowedOrigins: [...new Set([new URL(targetUrl).origin, ...(email?.allowedOrigins ?? [])])],
|
|
1847
|
-
originMap: buildOriginMap({
|
|
1848
|
-
...(config.subject.serve?.url === undefined ? {} : { internalServeUrl: config.subject.serve.url }),
|
|
1849
|
-
reachableBaseUrl: targetUrl,
|
|
1850
|
-
...(email?.linkOrigin === undefined ? {} : { linkOrigin: email.linkOrigin })
|
|
1851
|
-
})
|
|
1852
|
-
});
|
|
1853
|
-
commsArtifactPath = "comms/receiving.json";
|
|
1854
|
-
}
|
|
1855
|
-
catch (error) {
|
|
1856
|
-
await surface.stop().catch(() => { });
|
|
1857
|
-
throw error;
|
|
1858
|
-
}
|
|
1859
|
-
}
|
|
1860
|
-
if (commsEmail && commsPort !== undefined) {
|
|
1861
|
-
deployedComms = await deployCommsCatch(desktop, {
|
|
1862
|
-
port: commsPort,
|
|
1863
|
-
...(commsSmtpPort === undefined ? {} : { smtpPort: commsSmtpPort }),
|
|
1864
|
-
requestTimeoutMs: deps.requestTimeoutMs
|
|
775
|
+
await desktopLane.prepare();
|
|
776
|
+
// Start the brain BEFORE the first screenshot: the app-server handshake is ~500ms, and it
|
|
777
|
+
// is paid here, while the sandbox is still settling, rather than inside turn one.
|
|
778
|
+
if (deps.hooks.buildProvider) {
|
|
779
|
+
localAgentProvider = await deps.hooks.buildProvider({ config, actor: deps.descriptor, lane: spec });
|
|
780
|
+
}
|
|
781
|
+
else if (deps.localAgent === "codex") {
|
|
782
|
+
appServer = await startAppServerSession({
|
|
783
|
+
...(spec.reasoningEffort === undefined ? {} : { reasoningEffort: spec.reasoningEffort }),
|
|
784
|
+
...(config.actors[0]?.model === undefined ? {} : { model: config.actors[0].model }),
|
|
785
|
+
// The persona lives on the THREAD, so it is stated once instead of re-sent every turn.
|
|
786
|
+
baseInstructions: spec.instructions
|
|
1865
787
|
});
|
|
1866
|
-
|
|
1867
|
-
|
|
1868
|
-
|
|
1869
|
-
//
|
|
1870
|
-
//
|
|
1871
|
-
//
|
|
1872
|
-
|
|
1873
|
-
|
|
1874
|
-
|
|
1875
|
-
|
|
1876
|
-
|
|
1877
|
-
|
|
1878
|
-
|
|
1879
|
-
|
|
1880
|
-
|
|
1881
|
-
|
|
1882
|
-
|
|
1883
|
-
const refreshed = await refreshInboxSurface({
|
|
1884
|
-
desktop,
|
|
1885
|
-
deployed: deployedRef,
|
|
1886
|
-
recipients: surfaceRecipients,
|
|
1887
|
-
sinceCount: surfaceRenderedCount,
|
|
1888
|
-
originMap: commsOriginMap,
|
|
1889
|
-
requestTimeoutMs: deps.requestTimeoutMs
|
|
1890
|
-
});
|
|
1891
|
-
if (refreshed.rendered)
|
|
1892
|
-
surfaceRenderedCount = refreshed.count;
|
|
1893
|
-
}
|
|
1894
|
-
catch {
|
|
1895
|
-
// Never throw into the render loop; the teardown drain + by-id teardown must still run.
|
|
1896
|
-
}
|
|
1897
|
-
if (surfaceDisposed)
|
|
1898
|
-
break;
|
|
1899
|
-
await new Promise((resolve) => {
|
|
1900
|
-
const timer = setTimeout(resolve, INBOX_SURFACE_CADENCE_MS);
|
|
1901
|
-
void surfaceDispose.then(() => { clearTimeout(timer); resolve(); });
|
|
1902
|
-
});
|
|
1903
|
-
if (surfaceDisposed)
|
|
1904
|
-
break;
|
|
1905
|
-
}
|
|
1906
|
-
})();
|
|
1907
|
-
}
|
|
1908
|
-
// Per-lane geometry assertion (fail-closed) — the device claim is verified in-sandbox.
|
|
1909
|
-
const screenGeometry = await inspectDesktopScreenGeometry({
|
|
1910
|
-
desktop,
|
|
1911
|
-
laneId: spec.laneId,
|
|
1912
|
-
requestedScreen: spec.resolution,
|
|
1913
|
-
requestTimeoutMs: deps.requestTimeoutMs
|
|
1914
|
-
});
|
|
1915
|
-
if (screenGeometry.verified) {
|
|
1916
|
-
desktopGeometry = {
|
|
1917
|
-
...desktopGeometry,
|
|
1918
|
-
screen: { ...desktopGeometry.screen, verified: screenGeometry.verified }
|
|
1919
|
-
};
|
|
1920
|
-
}
|
|
1921
|
-
if (screenGeometry.warning) {
|
|
1922
|
-
warnings.push(screenGeometry.warning);
|
|
1923
|
-
desktopGeometry = { ...desktopGeometry, warnings: [screenGeometry.warning] };
|
|
1924
|
-
}
|
|
1925
|
-
if (screenGeometry.error && deps.screenMismatchPolicy !== "record-evidence") {
|
|
1926
|
-
sessionError = screenGeometry.error;
|
|
1927
|
-
failureCode = "HUMANISH_CUA_LAB_DEVICE_GEOMETRY";
|
|
1928
|
-
}
|
|
1929
|
-
else {
|
|
1930
|
-
if (screenGeometry.error && screenGeometry.verified) {
|
|
1931
|
-
// record-evidence policy: the bundle keeps requested vs verified as separate facts and
|
|
1932
|
-
// discloses the divergence instead of failing this lane's world mid-flight.
|
|
1933
|
-
const mismatchWarning = deps.scrubKnownValues(`Lane ${spec.laneId} requested a ${spec.resolution[0]}x${spec.resolution[1]} screen but xdpyinfo reports ${screenGeometry.verified.width}x${screenGeometry.verified.height}; recording requested vs verified separately instead of failing the lane closed.`);
|
|
1934
|
-
warnings.push(mismatchWarning);
|
|
1935
|
-
desktopGeometry = {
|
|
1936
|
-
...desktopGeometry,
|
|
1937
|
-
warnings: [...(desktopGeometry.warnings ?? []), mismatchWarning]
|
|
1938
|
-
};
|
|
1939
|
-
}
|
|
1940
|
-
if (desktopCliRoute) {
|
|
1941
|
-
// Prepare the runtime and any declared product install, UNKEYED. With install omitted,
|
|
1942
|
-
// the participant discovers and installs the product from its public surfaces.
|
|
1943
|
-
await provisionDesktopCli(desktop, {
|
|
1944
|
-
product: config.subject.product?.name ?? "",
|
|
1945
|
-
...(config.subject.product?.install === undefined ? {} : { install: config.subject.product.install }),
|
|
1946
|
-
requestTimeoutMs: deps.requestTimeoutMs,
|
|
1947
|
-
scrub: deps.scrubKnownValues,
|
|
1948
|
-
onPhase: onSubjectPhase
|
|
1949
|
-
});
|
|
1950
|
-
}
|
|
1951
|
-
if (cloneRoute && serve && subjectRepo) {
|
|
1952
|
-
subjectCommit = await provisionCloneSubject(desktop, {
|
|
1953
|
-
repo: subjectRepo,
|
|
1954
|
-
depth: config.subject.clone?.depth ?? 1,
|
|
1955
|
-
serve,
|
|
1956
|
-
...(config.subject.state === undefined ? {} : { state: config.subject.state }),
|
|
1957
|
-
hasGithubToken: deps.hasGithubToken,
|
|
1958
|
-
requestTimeoutMs: deps.requestTimeoutMs,
|
|
1959
|
-
scrub: deps.scrubKnownValues,
|
|
1960
|
-
onCommit: (commit) => {
|
|
1961
|
-
subjectCommit = commit;
|
|
1962
|
-
},
|
|
1963
|
-
onStateStep: (record) => {
|
|
1964
|
-
stateStepRecords.push(record);
|
|
1965
|
-
},
|
|
1966
|
-
onPhase: onSubjectPhase,
|
|
1967
|
-
...(deps.hooks.detachedTimers ?? {})
|
|
1968
|
-
});
|
|
1969
|
-
}
|
|
1970
|
-
else if (localTreeRoute && serve && deps.localTreeArchiveBuffer) {
|
|
1971
|
-
await provisionLocalTreeSubject(desktop, {
|
|
1972
|
-
archiveBuffer: deps.localTreeArchiveBuffer,
|
|
1973
|
-
serve,
|
|
1974
|
-
...(config.subject.state === undefined ? {} : { state: config.subject.state }),
|
|
1975
|
-
requestTimeoutMs: deps.requestTimeoutMs,
|
|
1976
|
-
scrub: deps.scrubKnownValues,
|
|
1977
|
-
onStateStep: (record) => {
|
|
1978
|
-
stateStepRecords.push(record);
|
|
1979
|
-
},
|
|
1980
|
-
onPhase: onSubjectPhase,
|
|
1981
|
-
...(deps.hooks.detachedTimers ?? {})
|
|
788
|
+
localAgentProvider = appServer.provider;
|
|
789
|
+
}
|
|
790
|
+
else if (deps.localAgent === "claude") {
|
|
791
|
+
// One session for the whole run, like the codex thread above (#520). The one-shot
|
|
792
|
+
// provider (createLocalAgentProvider) spawned `claude -p` per turn, and every turn
|
|
793
|
+
// started with no memory of the last. HUMANISH_LOCAL_AGENT_ONE_SHOT=1 keeps that path
|
|
794
|
+
// reachable as a MEASUREMENT switch: MemTrapBench (2026-08) reports memory frameworks
|
|
795
|
+
// degrading agent performance by 10-40% on some tasks, so "remembers" has to be measured
|
|
796
|
+
// against "does not" on the same lab, not assumed. The trace records which one ran.
|
|
797
|
+
const oneShot = env.HUMANISH_LOCAL_AGENT_ONE_SHOT !== undefined
|
|
798
|
+
&& env.HUMANISH_LOCAL_AGENT_ONE_SHOT !== ""
|
|
799
|
+
&& env.HUMANISH_LOCAL_AGENT_ONE_SHOT !== "0";
|
|
800
|
+
if (oneShot) {
|
|
801
|
+
localAgentProvider = createLocalAgentProvider({
|
|
802
|
+
agent: "claude",
|
|
803
|
+
...(spec.reasoningEffort === undefined ? {} : { reasoningEffort: spec.reasoningEffort }),
|
|
804
|
+
...(config.actors[0]?.model === undefined ? {} : { model: config.actors[0].model })
|
|
1982
805
|
});
|
|
1983
806
|
}
|
|
1984
|
-
if (!desktopCliRoute) {
|
|
1985
|
-
const requestedFidelity = config.execution?.desktop?.fidelity;
|
|
1986
|
-
// A declared camera (#509) is in place before the browser starts: the feed is generated or
|
|
1987
|
-
// uploaded first, and a feed that cannot be produced fails the lane closed here.
|
|
1988
|
-
const requestedMedia = config.execution?.desktop?.media;
|
|
1989
|
-
const mediaEvidence = requestedMedia === undefined
|
|
1990
|
-
? undefined
|
|
1991
|
-
: await prepareDesktopMedia(desktop, requestedMedia, config.policies?.mediaPermission ?? "prompt", deps.labCwd, deps.requestTimeoutMs);
|
|
1992
|
-
const browserLaunch = await openDesktopBrowserTarget(desktop, targetUrl, deps.requestTimeoutMs, config.execution?.desktop?.browser, [
|
|
1993
|
-
...(requestedFidelity?.mobileEmulation && spec.devicePreset.isMobile
|
|
1994
|
-
? [
|
|
1995
|
-
`--user-agent=${requestedFidelity.userAgent ?? DEFAULT_MOBILE_USER_AGENT}`,
|
|
1996
|
-
...(requestedFidelity.touch === false ? [] : ["--touch-events=enabled"])
|
|
1997
|
-
]
|
|
1998
|
-
: []),
|
|
1999
|
-
...(mediaEvidence?.flags ?? [])
|
|
2000
|
-
]);
|
|
2001
|
-
desktopBrowser = mediaEvidence === undefined
|
|
2002
|
-
? browserLaunch.evidence
|
|
2003
|
-
: { requested: config.execution?.desktop?.browser ?? "default", ...(browserLaunch.evidence ?? {}), media: mediaEvidence };
|
|
2004
|
-
if (mediaEvidence !== undefined && browserLaunch.family !== "chromium") {
|
|
2005
|
-
throw new Error(`execution.desktop.media needs Chrome or Chromium on lane ${spec.laneId} (the fake-device flags are Chromium's); the launched browser family is ${browserLaunch.family}. Set execution.desktop.browser: chrome.`);
|
|
2006
|
-
}
|
|
2007
|
-
launchedBrowserFamily = browserLaunch.family;
|
|
2008
|
-
browserLaunchIdentity = browserLaunch.identity;
|
|
2009
|
-
browserLaunched = true;
|
|
2010
|
-
await desktop.wait(BROWSER_SETTLE_MS).catch(() => undefined);
|
|
2011
|
-
// Mobile fidelity beyond viewport size (#221): applied to the launch page before the
|
|
2012
|
-
// geometry capture and the participant's first observation, OUTSIDE the stream/geometry
|
|
2013
|
-
// try below (whose catch degrades to a warning): a request that cannot be applied fails
|
|
2014
|
-
// the lane closed with the reason.
|
|
2015
|
-
// Only lanes on a mobile preset are emulated: a run-wide flag must not hand a desktop or
|
|
2016
|
-
// tablet lane an iPhone user agent (the first live proof did exactly that to the desktop
|
|
2017
|
-
// newcomer beside the phone lane). Those lanes carry no fidelity block, which is honest.
|
|
2018
|
-
const fidelityRequest = config.execution?.desktop?.fidelity;
|
|
2019
|
-
if (fidelityRequest?.mobileEmulation && spec.devicePreset.isMobile) {
|
|
2020
|
-
if (launchedBrowserFamily !== "chromium") {
|
|
2021
|
-
throw new Error(`execution.desktop.fidelity.mobileEmulation needs Chrome or Chromium on lane ${spec.laneId}; the launched browser family is ${launchedBrowserFamily}. Set execution.desktop.browser: chrome.`);
|
|
2022
|
-
}
|
|
2023
|
-
const applied = await applyMobileEmulation(desktop, deps.requestTimeoutMs, {
|
|
2024
|
-
...(browserLaunchIdentity?.cdpPort === undefined ? {} : { cdpPort: browserLaunchIdentity.cdpPort }),
|
|
2025
|
-
...(browserLaunchIdentity?.profileDir === undefined ? {} : { profileDir: browserLaunchIdentity.profileDir }),
|
|
2026
|
-
targetUrl
|
|
2027
|
-
}, browserTargetId, {
|
|
2028
|
-
width: spec.devicePreset.width,
|
|
2029
|
-
height: spec.devicePreset.height,
|
|
2030
|
-
deviceScaleFactor: fidelityRequest.deviceScaleFactor ?? spec.devicePreset.deviceScaleFactor,
|
|
2031
|
-
touch: fidelityRequest.touch ?? true,
|
|
2032
|
-
userAgent: fidelityRequest.userAgent ?? DEFAULT_MOBILE_USER_AGENT
|
|
2033
|
-
});
|
|
2034
|
-
appliedFidelity = applied.fidelity;
|
|
2035
|
-
emulatedTargetId = applied.targetId;
|
|
2036
|
-
emulationHolderName = applied.holderName;
|
|
2037
|
-
warnings.push(...applied.warnings);
|
|
2038
|
-
}
|
|
2039
|
-
}
|
|
2040
807
|
else {
|
|
2041
|
-
|
|
2042
|
-
// participant arrives at a desktop with the thing they were asked to use already in front
|
|
2043
|
-
// of them. They can still open another from the dock — that is the point of a desktop.
|
|
2044
|
-
await openDesktopTerminal(desktop, deps.requestTimeoutMs, config.subject.product?.workdir);
|
|
2045
|
-
await desktop.wait(BROWSER_SETTLE_MS).catch(() => undefined);
|
|
2046
|
-
}
|
|
2047
|
-
// Start the brain BEFORE the first screenshot: the app-server handshake is ~500ms, and it
|
|
2048
|
-
// is paid here, while the sandbox is still settling, rather than inside turn one.
|
|
2049
|
-
if (deps.localAgent === "codex") {
|
|
2050
|
-
appServer = await startAppServerSession({
|
|
808
|
+
claudeSession = await startClaudeSession({
|
|
2051
809
|
...(spec.reasoningEffort === undefined ? {} : { reasoningEffort: spec.reasoningEffort }),
|
|
2052
|
-
...(config.actors[0]?.model === undefined ? {} : { model: config.actors[0].model })
|
|
2053
|
-
// The persona lives on the THREAD, so it is stated once instead of re-sent every turn.
|
|
2054
|
-
baseInstructions: spec.instructions
|
|
2055
|
-
});
|
|
2056
|
-
localAgentProvider = appServer.provider;
|
|
2057
|
-
}
|
|
2058
|
-
else if (deps.localAgent === "claude") {
|
|
2059
|
-
// One session for the whole run, like the codex thread above (#520). The one-shot
|
|
2060
|
-
// provider (createLocalAgentProvider) spawned `claude -p` per turn, and every turn
|
|
2061
|
-
// started with no memory of the last. HUMANISH_LOCAL_AGENT_ONE_SHOT=1 keeps that path
|
|
2062
|
-
// reachable as a MEASUREMENT switch: MemTrapBench (2026-08) reports memory frameworks
|
|
2063
|
-
// degrading agent performance by 10-40% on some tasks, so "remembers" has to be measured
|
|
2064
|
-
// against "does not" on the same lab, not assumed. The trace records which one ran.
|
|
2065
|
-
const oneShot = env.HUMANISH_LOCAL_AGENT_ONE_SHOT !== undefined
|
|
2066
|
-
&& env.HUMANISH_LOCAL_AGENT_ONE_SHOT !== ""
|
|
2067
|
-
&& env.HUMANISH_LOCAL_AGENT_ONE_SHOT !== "0";
|
|
2068
|
-
if (oneShot) {
|
|
2069
|
-
localAgentProvider = createLocalAgentProvider({
|
|
2070
|
-
agent: "claude",
|
|
2071
|
-
...(spec.reasoningEffort === undefined ? {} : { reasoningEffort: spec.reasoningEffort }),
|
|
2072
|
-
...(config.actors[0]?.model === undefined ? {} : { model: config.actors[0].model })
|
|
2073
|
-
});
|
|
2074
|
-
}
|
|
2075
|
-
else {
|
|
2076
|
-
claudeSession = await startClaudeSession({
|
|
2077
|
-
...(spec.reasoningEffort === undefined ? {} : { reasoningEffort: spec.reasoningEffort }),
|
|
2078
|
-
...(config.actors[0]?.model === undefined ? {} : { model: config.actors[0].model })
|
|
2079
|
-
});
|
|
2080
|
-
localAgentProvider = claudeSession.provider;
|
|
2081
|
-
}
|
|
2082
|
-
}
|
|
2083
|
-
// World is ready: release the pipeline gate so the remaining lanes may start.
|
|
2084
|
-
provisioned = true;
|
|
2085
|
-
signal(true);
|
|
2086
|
-
try {
|
|
2087
|
-
// No browser means no browser geometry, and none is invented: the CSS-viewport facts a
|
|
2088
|
-
// browser reports have no counterpart in a terminal window, and an empty record shaped like
|
|
2089
|
-
// a measurement would read as one. The screen geometry above is still verified.
|
|
2090
|
-
if (!desktopCliRoute) {
|
|
2091
|
-
const browserGeometry = await captureDesktopBrowserGeometry({
|
|
2092
|
-
desktop,
|
|
2093
|
-
browserFamily: launchedBrowserFamily,
|
|
2094
|
-
...(browserLaunchIdentity === undefined ? {} : { launchIdentity: browserLaunchIdentity }),
|
|
2095
|
-
laneId: spec.laneId,
|
|
2096
|
-
targetUrl,
|
|
2097
|
-
requestedScreen: spec.resolution,
|
|
2098
|
-
requestTimeoutMs: deps.requestTimeoutMs
|
|
2099
|
-
});
|
|
2100
|
-
initialBrowserGeometry = browserGeometry;
|
|
2101
|
-
browserWindowId = browserGeometry.browserWindowId;
|
|
2102
|
-
browserTargetId = browserGeometry.browserTargetId;
|
|
2103
|
-
}
|
|
2104
|
-
// The WHOLE desktop, not one window: a person studying a terminal app opens other windows,
|
|
2105
|
-
// and a stream bound to the first one would quietly stop being evidence.
|
|
2106
|
-
await startDesktopStream(desktop, browserWindowId);
|
|
2107
|
-
const candidateStreamUrl = desktop.stream.getUrl({
|
|
2108
|
-
authKey: desktop.stream.getAuthKey(),
|
|
2109
|
-
autoConnect: true,
|
|
2110
|
-
viewOnly: true,
|
|
2111
|
-
resize: "scale"
|
|
810
|
+
...(config.actors[0]?.model === undefined ? {} : { model: config.actors[0].model })
|
|
2112
811
|
});
|
|
2113
|
-
|
|
2114
|
-
streamUrl = candidateStreamUrl;
|
|
2115
|
-
await deps.hooks.onRuntimeStreamReady?.({
|
|
2116
|
-
laneId: spec.laneId,
|
|
2117
|
-
sandboxId: desktop.sandboxId,
|
|
2118
|
-
simId: spec.simId,
|
|
2119
|
-
streamId: spec.streamId,
|
|
2120
|
-
url: streamUrl
|
|
2121
|
-
});
|
|
2122
|
-
}
|
|
2123
|
-
else {
|
|
2124
|
-
warnings.push("Live desktop stream started but did not return a usable watch URL; Observer will fall back to screenshots.");
|
|
2125
|
-
}
|
|
812
|
+
localAgentProvider = claudeSession.provider;
|
|
2126
813
|
}
|
|
2127
|
-
catch (error) {
|
|
2128
|
-
warnings.push(`Live desktop stream unavailable (run continues; evidence still captured): ${redactText(deps.scrubKnownValues(toErrorMessage(error)))}`);
|
|
2129
|
-
}
|
|
2130
|
-
// This is outside the stream's best-effort catch: unusable geometry is a harness failure,
|
|
2131
|
-
// never a participant finding about missing controls. Both per-lane and concurrent seats
|
|
2132
|
-
// use this route; sequential seats enforce the same capture result in shared-world-lab.
|
|
2133
|
-
if (initialBrowserGeometry?.unusable !== undefined) {
|
|
2134
|
-
failureCode = "HUMANISH_CUA_LAB_DEVICE_GEOMETRY";
|
|
2135
|
-
throw new Error(`${failureCode}: ${initialBrowserGeometry.unusable} Participant actions were not started.`);
|
|
2136
|
-
}
|
|
2137
|
-
// The FAIL-CLOSED spend cap (execution.caps.maxUsd) is wired into the loop as maxUsd + an
|
|
2138
|
-
// injected pure per-turn estimator keyed on the resolved model. Preflight already refused a
|
|
2139
|
-
// cap on an unpriced model, so the estimate is measurable whenever a cap is in force. The
|
|
2140
|
-
// model id here matches provider.version (openai-responses-cu resolves the default when unset).
|
|
2141
|
-
const capModelId = config.actors[0]?.model ?? DEFAULT_OPENAI_CU_MODEL;
|
|
2142
|
-
const maxUsd = config.execution?.caps?.maxUsd;
|
|
2143
|
-
const sessionOptions = {
|
|
2144
|
-
// Tell the persona where its inbox is — but only when comms is live AND this lane has a declared
|
|
2145
|
-
// recipient it can actually receive mail into (else it would stall on an inbox that stays
|
|
2146
|
-
// empty). Two comms planes, mutually exclusive by parse: the in-sandbox catch humanish
|
|
2147
|
-
// deployed, or the adopter-hosted one (#380).
|
|
2148
|
-
instructions: deps.receiving && receivingInboxUrl
|
|
2149
|
-
? withInboxMission(spec, receivingInboxUrl, deps.receiving.address(spec.laneId), true).instructions
|
|
2150
|
-
: commsEmail && commsInboxUrl && deployedComms?.ready && laneHasInboxRecipient(commsEmail, spec.laneId)
|
|
2151
|
-
? withInboxMission(spec, commsInboxUrl, inboxRecipientFor(commsEmail, spec.laneId)?.address).instructions
|
|
2152
|
-
: deps.externalComms && laneHasInboxRecipient(deps.externalComms.email, spec.laneId)
|
|
2153
|
-
? withInboxMission(spec, deps.externalComms.inboxUrl, inboxRecipientFor(deps.externalComms.email, spec.laneId)?.address).instructions
|
|
2154
|
-
: spec.instructions,
|
|
2155
|
-
persona: spec.persona,
|
|
2156
|
-
timeoutMs: deps.timeoutMs,
|
|
2157
|
-
// The brain is either a keyed API client or a CLI the operator is already signed in to.
|
|
2158
|
-
// Everything below this line — loop, executor, trace, affordances — is identical either
|
|
2159
|
-
// way, which is what makes a local-agent run comparable to an API one.
|
|
2160
|
-
...(localAgentProvider === undefined ? {} : { provider: localAgentProvider }),
|
|
2161
|
-
openai: {
|
|
2162
|
-
apiKey: deps.openaiApiKey,
|
|
2163
|
-
...(config.actors[0]?.model ? { model: config.actors[0].model } : {}),
|
|
2164
|
-
// Per-LANE, not per-actor: two lanes at different efforts is the control this exists for.
|
|
2165
|
-
...(spec.reasoningEffort === undefined ? {} : { reasoningEffort: spec.reasoningEffort }),
|
|
2166
|
-
...(spec.maxOutputTokens === undefined ? {} : { maxOutputTokens: spec.maxOutputTokens })
|
|
2167
|
-
},
|
|
2168
|
-
...(maxUsd === undefined
|
|
2169
|
-
? {}
|
|
2170
|
-
: {
|
|
2171
|
-
maxUsd,
|
|
2172
|
-
estimateTurnCostUsd: (usage) => estimateActorCost(usage, capModelId).estimatedCostUsd
|
|
2173
|
-
}),
|
|
2174
|
-
desktop: desktop,
|
|
2175
|
-
...(launchedBrowserFamily === "chromium"
|
|
2176
|
-
? {
|
|
2177
|
-
executorOptions: {
|
|
2178
|
-
observeBrowserState: makeChromeBrowserStateObserver(desktop, deps.requestTimeoutMs, {
|
|
2179
|
-
...(browserLaunchIdentity?.cdpPort === undefined ? {} : { cdpPort: browserLaunchIdentity.cdpPort }),
|
|
2180
|
-
...(browserLaunchIdentity?.profileDir === undefined ? {} : { profileDir: browserLaunchIdentity.profileDir }),
|
|
2181
|
-
targetUrl
|
|
2182
|
-
}, browserTargetId,
|
|
2183
|
-
// Once per lane: a dark observation channel is a gap in the instrument, and the
|
|
2184
|
-
// funnel's NEVER MEASURED count needs this line to explain itself (#514).
|
|
2185
|
-
(reason) => {
|
|
2186
|
-
warnings.push(`Browser-state observer unavailable for lane ${spec.laneId} (${redactText(deps.scrubKnownValues(reason))}); ` +
|
|
2187
|
-
"urlIncludes/urlPathEquals/textIncludes stop conditions and task criteria are NOT being measured this session.");
|
|
2188
|
-
}, emulatedTargetId === undefined
|
|
2189
|
-
? undefined
|
|
2190
|
-
: {
|
|
2191
|
-
emulatedTargetId,
|
|
2192
|
-
expectedWidth: spec.devicePreset.width,
|
|
2193
|
-
expectTouch: appliedFidelity?.requested.touch === true,
|
|
2194
|
-
onDrift: (reason) => {
|
|
2195
|
-
warnings.push(`Mobile emulation drift on lane ${spec.laneId}: ${reason} (#623).`);
|
|
2196
|
-
},
|
|
2197
|
-
onCovered: (coveredTargetId, read) => {
|
|
2198
|
-
// A later tab the page itself reported at the phone width: evidence that
|
|
2199
|
-
// the emulation followed the participant (#623), kept on the bundle.
|
|
2200
|
-
if (appliedFidelity === undefined)
|
|
2201
|
-
return;
|
|
2202
|
-
appliedFidelity = {
|
|
2203
|
-
...appliedFidelity,
|
|
2204
|
-
laterTargets: [...(appliedFidelity.laterTargets ?? []), { targetId: coveredTargetId, ...read }]
|
|
2205
|
-
};
|
|
2206
|
-
}
|
|
2207
|
-
})
|
|
2208
|
-
}
|
|
2209
|
-
}
|
|
2210
|
-
: {}),
|
|
2211
|
-
redactScreenshots: deps.redactScreenshots,
|
|
2212
|
-
scrubText: deps.scrubKnownValues,
|
|
2213
|
-
writeScreenshot,
|
|
2214
|
-
...(spec.idleSteps === undefined ? {} : { idleSteps: spec.idleSteps }),
|
|
2215
|
-
...(spec.noProgressSteps === undefined ? {} : { noProgressSteps: spec.noProgressSteps }),
|
|
2216
|
-
...(spec.stopWhen === undefined ? {} : { stopWhen: spec.stopWhen }),
|
|
2217
|
-
...(spec.dwell === undefined ? {} : { dwell: spec.dwell }),
|
|
2218
|
-
...(spec.tasks === undefined ? {} : { tasks: spec.tasks }),
|
|
2219
|
-
// The STUDY budget (#299): this lane notes its own running estimate on the shared ledger
|
|
2220
|
-
// and stops when the RUN total crosses the cap — independent of the per-lane maxUsd above.
|
|
2221
|
-
...(deps.runBudget === undefined
|
|
2222
|
-
? {}
|
|
2223
|
-
: {
|
|
2224
|
-
overRunBudget: (usage) => {
|
|
2225
|
-
const estimate = estimateActorCost(usage, capModelId).estimatedCostUsd;
|
|
2226
|
-
const totalUsd = deps.runBudget.note(spec.laneId, estimate);
|
|
2227
|
-
return totalUsd > deps.runBudget.maxTotalUsd
|
|
2228
|
-
? `study budget reached: the run's estimated model spend $${round6(totalUsd)} crossed execution.caps.maxTotalUsd=$${deps.runBudget.maxTotalUsd}; this lane stops here and sibling lanes stop at their next turn`
|
|
2229
|
-
: null;
|
|
2230
|
-
}
|
|
2231
|
-
}),
|
|
2232
|
-
...(deps.onObservedUrl === undefined ? {} : { onObservedUrl: deps.onObservedUrl }),
|
|
2233
|
-
...(deps.onMessage === undefined ? {} : { onMessage: deps.onMessage }),
|
|
2234
|
-
...(deps.onScreenshot === undefined ? {} : { onScreenshot: deps.onScreenshot }),
|
|
2235
|
-
...(deps.onTrace === undefined
|
|
2236
|
-
? {}
|
|
2237
|
-
: {
|
|
2238
|
-
// Forwards the RUNNING usage as well: the lane is where both are known, and usage
|
|
2239
|
-
// without it never reaches the flush — which is how the live cost stayed unknown.
|
|
2240
|
-
onTrace: (items, usage) => deps.onTrace?.(spec.laneId, items, usage)
|
|
2241
|
-
})
|
|
2242
|
-
};
|
|
2243
|
-
session = await deps.runSession(sessionOptions);
|
|
2244
814
|
}
|
|
815
|
+
// World is ready: release the pipeline gate so the remaining lanes may start.
|
|
816
|
+
provisioned = true;
|
|
817
|
+
signal(true);
|
|
818
|
+
const ready = await desktopLane.openSession();
|
|
819
|
+
// The FAIL-CLOSED spend cap (execution.caps.maxUsd) is wired into the loop as maxUsd + an
|
|
820
|
+
// injected pure per-turn estimator keyed on the resolved model. Preflight already refused a
|
|
821
|
+
// cap on an unpriced model, so the estimate is measurable whenever a cap is in force. The
|
|
822
|
+
// model id here matches provider.version (openai-responses-cu resolves the default when unset).
|
|
823
|
+
const capModelId = config.actors[0]?.model ?? DEFAULT_OPENAI_CU_MODEL;
|
|
824
|
+
const maxUsd = config.execution?.caps?.maxUsd;
|
|
825
|
+
const sessionOptions = {
|
|
826
|
+
instructions: ready.inbox
|
|
827
|
+
? withInboxMission(spec, ready.inbox.url, ready.inbox.address, ready.inbox.receiving).instructions
|
|
828
|
+
: spec.instructions,
|
|
829
|
+
persona: spec.persona,
|
|
830
|
+
timeoutMs: deps.timeoutMs,
|
|
831
|
+
// The brain is either a keyed API client or a CLI the operator is already signed in to.
|
|
832
|
+
// Everything below this line — loop, executor, trace, affordances — is identical either
|
|
833
|
+
// way, which is what makes a local-agent run comparable to an API one.
|
|
834
|
+
...(localAgentProvider === undefined ? {} : { provider: localAgentProvider }),
|
|
835
|
+
openai: {
|
|
836
|
+
apiKey: deps.openaiApiKey,
|
|
837
|
+
...(config.actors[0]?.model ? { model: config.actors[0].model } : {}),
|
|
838
|
+
// Per-LANE, not per-actor: two lanes at different efforts is the control this exists for.
|
|
839
|
+
...(spec.reasoningEffort === undefined ? {} : { reasoningEffort: spec.reasoningEffort }),
|
|
840
|
+
...(spec.maxOutputTokens === undefined ? {} : { maxOutputTokens: spec.maxOutputTokens })
|
|
841
|
+
},
|
|
842
|
+
...(maxUsd === undefined
|
|
843
|
+
? {}
|
|
844
|
+
: {
|
|
845
|
+
maxUsd,
|
|
846
|
+
estimateTurnCostUsd: (usage) => estimateActorCost(usage, capModelId).estimatedCostUsd
|
|
847
|
+
}),
|
|
848
|
+
executor: ready.executor,
|
|
849
|
+
redactScreenshots: deps.redactScreenshots,
|
|
850
|
+
scrubText: deps.scrubKnownValues,
|
|
851
|
+
writeScreenshot,
|
|
852
|
+
...(spec.idleSteps === undefined ? {} : { idleSteps: spec.idleSteps }),
|
|
853
|
+
...(spec.noProgressSteps === undefined ? {} : { noProgressSteps: spec.noProgressSteps }),
|
|
854
|
+
...(spec.stopWhen === undefined ? {} : { stopWhen: spec.stopWhen }),
|
|
855
|
+
...(spec.dwell === undefined ? {} : { dwell: spec.dwell }),
|
|
856
|
+
...(spec.tasks === undefined ? {} : { tasks: spec.tasks }),
|
|
857
|
+
// The STUDY budget (#299): this lane notes its own running estimate on the shared ledger
|
|
858
|
+
// and stops when the RUN total crosses the cap — independent of the per-lane maxUsd above.
|
|
859
|
+
...(deps.runBudget === undefined
|
|
860
|
+
? {}
|
|
861
|
+
: {
|
|
862
|
+
overRunBudget: (usage) => {
|
|
863
|
+
const estimate = estimateActorCost(usage, capModelId).estimatedCostUsd;
|
|
864
|
+
const totalUsd = deps.runBudget.note(spec.laneId, estimate);
|
|
865
|
+
return totalUsd > deps.runBudget.maxTotalUsd
|
|
866
|
+
? `study budget reached: the run's estimated model spend $${round6(totalUsd)} crossed execution.caps.maxTotalUsd=$${deps.runBudget.maxTotalUsd}; this lane stops here and sibling lanes stop at their next turn`
|
|
867
|
+
: null;
|
|
868
|
+
}
|
|
869
|
+
}),
|
|
870
|
+
...(deps.onObservedUrl === undefined ? {} : { onObservedUrl: deps.onObservedUrl }),
|
|
871
|
+
...(deps.onMessage === undefined ? {} : { onMessage: deps.onMessage }),
|
|
872
|
+
...(deps.onScreenshot === undefined ? {} : { onScreenshot: deps.onScreenshot }),
|
|
873
|
+
...(deps.onTrace === undefined
|
|
874
|
+
? {}
|
|
875
|
+
: {
|
|
876
|
+
// Forwards the RUNNING usage as well: the lane is where both are known, and usage
|
|
877
|
+
// without it never reaches the flush — which is how the live cost stayed unknown.
|
|
878
|
+
onTrace: (items, usage, metadata) => deps.onTrace?.(spec.laneId, items, usage, metadata)
|
|
879
|
+
})
|
|
880
|
+
};
|
|
881
|
+
session = await deps.runSession(sessionOptions);
|
|
2245
882
|
}
|
|
2246
883
|
catch (error) {
|
|
2247
884
|
sessionError = redactText(deps.scrubKnownValues(toErrorMessage(error)));
|
|
2248
885
|
}
|
|
2249
886
|
finally {
|
|
2250
|
-
|
|
2251
|
-
|
|
2252
|
-
appServer?.close();
|
|
2253
|
-
await claudeSession?.close();
|
|
2254
|
-
// Stop the mid-run inbox-surface loop FIRST — before the teardown evidence drain below — so the two
|
|
2255
|
-
// `cat`s never overlap and the final surface state is deterministic. A surface failure can never
|
|
2256
|
-
// block teardown (the loop body is fully try/caught and this await is on its already-caught promise).
|
|
2257
|
-
surfaceDisposed = true;
|
|
2258
|
-
releaseSurface();
|
|
2259
|
-
if (surfaceLoop)
|
|
2260
|
-
await surfaceLoop.catch(() => undefined);
|
|
2261
|
-
if (!provisioned) {
|
|
2262
|
-
signal(false);
|
|
887
|
+
try {
|
|
888
|
+
await localAgentProvider?.close?.();
|
|
2263
889
|
}
|
|
2264
|
-
|
|
2265
|
-
|
|
2266
|
-
|
|
2267
|
-
|
|
2268
|
-
|
|
2269
|
-
|
|
2270
|
-
|
|
2271
|
-
|
|
2272
|
-
|
|
2273
|
-
|
|
2274
|
-
|
|
2275
|
-
|
|
2276
|
-
|
|
2277
|
-
|
|
2278
|
-
|
|
2279
|
-
|
|
2280
|
-
|
|
2281
|
-
|
|
2282
|
-
|
|
2283
|
-
|
|
2284
|
-
|
|
2285
|
-
|
|
2286
|
-
? finalGeometry
|
|
2287
|
-
: initialBrowserGeometry ?? finalGeometry;
|
|
2288
|
-
const geometryWarnings = [...new Set([...(initialBrowserGeometry?.warnings ?? []), ...chosenGeometry.warnings].map((warning) => deps.scrubKnownValues(warning)))];
|
|
2289
|
-
warnings.push(...geometryWarnings);
|
|
2290
|
-
// The emulation holder's own log, after its announce line: which later targets it
|
|
2291
|
-
// attached to, what it sent, and any reply that came back as an error (#623). Read while
|
|
2292
|
-
// the sandbox is alive; the first live proof had no way to say what the holder did.
|
|
2293
|
-
if (appliedFidelity !== undefined && emulationHolderName !== undefined) {
|
|
2294
|
-
const holderLog = await readDetachedLog(desktop, emulationHolderName, deps.requestTimeoutMs).catch(() => "");
|
|
2295
|
-
const lines = holderLog.split("\n").map((line) => line.trim()).filter((line) => line.startsWith("{")).slice(1, 51);
|
|
2296
|
-
if (lines.length > 0)
|
|
2297
|
-
appliedFidelity = { ...appliedFidelity, holderLog: lines.map((line) => deps.scrubKnownValues(line)) };
|
|
2298
|
-
}
|
|
2299
|
-
desktopGeometry = {
|
|
2300
|
-
screen: desktopGeometry.screen,
|
|
2301
|
-
...(chosenGeometry.browserWindow === undefined ? {} : { browserWindow: chosenGeometry.browserWindow }),
|
|
2302
|
-
...(chosenGeometry.viewport === undefined ? {} : { viewport: chosenGeometry.viewport }),
|
|
2303
|
-
...(appliedFidelity === undefined ? {} : { fidelity: appliedFidelity }),
|
|
2304
|
-
...((desktopGeometry.warnings?.length ?? 0) + geometryWarnings.length === 0
|
|
2305
|
-
? {}
|
|
2306
|
-
: { warnings: [...(desktopGeometry.warnings ?? []), ...geometryWarnings] })
|
|
2307
|
-
};
|
|
2308
|
-
}
|
|
2309
|
-
if (deps.receiving) {
|
|
2310
|
-
try {
|
|
2311
|
-
await deps.receiving.finishParticipant(spec.laneId);
|
|
2312
|
-
}
|
|
2313
|
-
catch {
|
|
2314
|
-
warnings.push("Real email finalization is incomplete. Inspect communication cleanup with humanish comms recover.");
|
|
2315
|
-
}
|
|
2316
|
-
}
|
|
2317
|
-
// Off-app comms evidence (#297): before this lane's sandbox is torn down, drain everything the
|
|
2318
|
-
// in-sandbox catch captured, route it into a host fake inbox addressed to the declared
|
|
2319
|
-
// recipients, and write the digest-only thread artifact. Wrapped so a drain failure NEVER
|
|
2320
|
-
// breaks teardown — the sandbox must still be killed either way. Runs only for a ready catch.
|
|
2321
|
-
if (commsEmail && deployedComms?.ready) {
|
|
2322
|
-
try {
|
|
2323
|
-
const commsChannel = new FakeInbox();
|
|
2324
|
-
const commsInboxes = [];
|
|
2325
|
-
for (const recipient of commsEmail.recipients ?? []) {
|
|
2326
|
-
if (recipient.address !== undefined) {
|
|
2327
|
-
commsInboxes.push(await commsChannel.provisionAddress(recipient.lane, recipient.address));
|
|
2328
|
-
}
|
|
2329
|
-
}
|
|
2330
|
-
const collected = await collectCommsThread({
|
|
2331
|
-
desktop,
|
|
2332
|
-
deployed: deployedComms,
|
|
2333
|
-
channel: commsChannel,
|
|
2334
|
-
inboxes: commsInboxes,
|
|
2335
|
-
requestTimeoutMs: deps.requestTimeoutMs
|
|
2336
|
-
});
|
|
2337
|
-
if (collected.artifact) {
|
|
2338
|
-
const path = deps.laneCount === 1 ? "comms/thread.json" : `comms/${spec.streamId}.thread.json`;
|
|
2339
|
-
await writeContainedOutputFile(deps.artifactRoot, path, `${JSON.stringify(collected.artifact, null, 2)}\n`, "utf8");
|
|
2340
|
-
commsArtifactPath = path;
|
|
2341
|
-
}
|
|
2342
|
-
else if (collected.captured > 0) {
|
|
2343
|
-
// Captured mail that matched no declared recipient must not vanish silently (invariant 6:
|
|
2344
|
-
// honest signals): tell the operator to declare comms.email.recipients[].address to match
|
|
2345
|
-
// the address the app actually sends to (e.g. the one the persona surface will sign up with).
|
|
2346
|
-
warnings.push(`Comms catch captured ${collected.captured} email send(s) but none matched a declared recipient inbox — no comms evidence written. Declare comms.email.recipients[].address to match the address the app sends to.`);
|
|
2347
|
-
}
|
|
2348
|
-
else {
|
|
2349
|
-
// Zero captures is the silent-broken shape (#351): the app never posted to the catch at
|
|
2350
|
-
// all, so the personas stared at an empty inbox. Most common cause: the app does not
|
|
2351
|
-
// actually read the declared injectEnv var for its email API base URL.
|
|
2352
|
-
const transportHint = commsEmail.smtp
|
|
2353
|
-
? `Verify the app reads ${commsEmail.smtp.hostEnv}/${commsEmail.smtp.portEnv} for its SMTP host and port`
|
|
2354
|
-
: `Verify the app reads ${commsEmail.injectEnv} for its email API base URL (an SDK that ignores it sends real mail or throws)`;
|
|
2355
|
-
warnings.push(`Comms catch captured ZERO email sends — the app never delivered mail through the catch. ${transportHint} and that the flow reached an email step.`);
|
|
2356
|
-
}
|
|
2357
|
-
}
|
|
2358
|
-
catch (error) {
|
|
2359
|
-
warnings.push(`Comms evidence collection failed (run continues; sandbox still torn down): ${redactText(deps.scrubKnownValues(toErrorMessage(error)))}`);
|
|
2360
|
-
}
|
|
2361
|
-
}
|
|
2362
|
-
const failed = sessionError !== undefined || session === undefined;
|
|
2363
|
-
// Each route's own keep flag gates its own lane only: a clone.keep can never leak into
|
|
2364
|
-
// a local-tree lane's teardown decision, and vice versa.
|
|
2365
|
-
const keepReason = cloneRoute && config.subject.clone?.keep === true
|
|
2366
|
-
? "subject.clone.keep"
|
|
2367
|
-
: localTreeRoute && config.subject.localTree?.keep === true
|
|
2368
|
-
? "subject.localTree.keep"
|
|
2369
|
-
: undefined;
|
|
2370
|
-
const keepForDebug = keepReason !== undefined && failed;
|
|
2371
|
-
if (keepForDebug) {
|
|
2372
|
-
warnings.push(`Sandbox ${desktop.sandboxId} kept for debugging (${keepReason} on failure); reclaim it via E2B or it will be killed on its server-side timeout.`);
|
|
2373
|
-
}
|
|
2374
|
-
else if (typeof desktopModule.Sandbox.kill === "function") {
|
|
2375
|
-
try {
|
|
2376
|
-
await desktopModule.Sandbox.kill(desktop.sandboxId, { requestTimeoutMs: 60_000 });
|
|
2377
|
-
killed = true;
|
|
2378
|
-
}
|
|
2379
|
-
catch (error) {
|
|
2380
|
-
warnings.push(`Sandbox teardown failed (server-side kill-on-timeout will reclaim it): ${redactText(deps.scrubKnownValues(toErrorMessage(error)))}`);
|
|
2381
|
-
}
|
|
2382
|
-
}
|
|
2383
|
-
else {
|
|
2384
|
-
warnings.push("Installed @e2b/desktop SDK does not expose Sandbox.kill; server-side kill-on-timeout will reclaim the sandbox.");
|
|
2385
|
-
}
|
|
2386
|
-
// Close the observed span. A kept or unconfirmed sandbox can still accrue compute cost;
|
|
2387
|
-
// the summary records that remaining lifetime as unknown instead of calling this complete.
|
|
2388
|
-
sandboxTornDownAtMs = deps.now();
|
|
2389
|
-
// The lane's live stream is now a dead page whichever teardown path ran (killed, kept, or
|
|
2390
|
-
// kill-failed-awaiting-TTL) — tell the watch overlay so the tile falls back to recorded
|
|
2391
|
-
// evidence instead of "sandbox not found" (#357). Guarded: a viewer callback must never
|
|
2392
|
-
// break teardown.
|
|
2393
|
-
if (streamUrl !== undefined) {
|
|
2394
|
-
try {
|
|
2395
|
-
await deps.hooks.onRuntimeStreamEnded?.({ laneId: spec.laneId, simId: spec.simId, streamId: spec.streamId });
|
|
2396
|
-
}
|
|
2397
|
-
catch {
|
|
2398
|
-
// viewer-side only; nothing to record
|
|
2399
|
-
}
|
|
2400
|
-
}
|
|
890
|
+
catch {
|
|
891
|
+
warnings.push("Model provider cleanup is unconfirmed.");
|
|
892
|
+
sessionError ??= "Model provider cleanup is unconfirmed.";
|
|
893
|
+
}
|
|
894
|
+
try {
|
|
895
|
+
appServer?.close();
|
|
896
|
+
}
|
|
897
|
+
catch {
|
|
898
|
+
warnings.push('Codex session cleanup failed; desktop cleanup will still run.');
|
|
899
|
+
}
|
|
900
|
+
try {
|
|
901
|
+
await claudeSession?.close();
|
|
902
|
+
}
|
|
903
|
+
catch {
|
|
904
|
+
warnings.push('Claude session cleanup failed; desktop cleanup will still run.');
|
|
905
|
+
}
|
|
906
|
+
try {
|
|
907
|
+
if (!provisioned)
|
|
908
|
+
signal(false);
|
|
909
|
+
}
|
|
910
|
+
finally {
|
|
911
|
+
await desktopLane.finalize({ failed: sessionError !== undefined || session === undefined });
|
|
2401
912
|
}
|
|
2402
913
|
}
|
|
2403
|
-
// Host-side approximation of the E2B desktop's billed lifetime; feeds the desktop-minute cost
|
|
2404
|
-
// estimate. Never negative.
|
|
2405
|
-
const desktopDurationMs = sandboxCreatedAtMs !== undefined && sandboxTornDownAtMs !== undefined
|
|
2406
|
-
? Math.max(0, sandboxTornDownAtMs - sandboxCreatedAtMs)
|
|
2407
|
-
: undefined;
|
|
2408
914
|
if (session) {
|
|
2409
915
|
// Per-lane model-token cost ESTIMATE, attached to the trace before it is persisted (the model
|
|
2410
916
|
// id is authoritative here — provider.version). Kept at the lab boundary so the pure loop
|
|
2411
917
|
// never depends on the operator rate table. estimateActorCost declares absent (null) for an
|
|
2412
918
|
// unknown rate / missing usage rather than guessing.
|
|
2413
|
-
session.trace.estimatedCost =
|
|
919
|
+
session.trace.estimatedCost = estimateActorCostForExecution(session.trace.tokenUsage, session.trace.ids.model, session.trace.executionProfile);
|
|
2414
920
|
await writeContainedOutputFile(deps.artifactRoot, spec.traceArtifactPath, `${JSON.stringify(session.trace, null, 2)}\n`, "utf8");
|
|
2415
921
|
if (session.trace.redaction.screenshots === "raw") {
|
|
2416
922
|
warnings.push("Screenshots are full-fidelity (raw) for local use — the bundle stays in gitignored .humanish and nothing scans these pixels; review them before sharing anywhere. Set policies.redactScreenshots: true to blur a share-as-is bundle.");
|
|
@@ -2435,24 +941,13 @@ export async function runCuaLane(spec, deps) {
|
|
|
2435
941
|
spec,
|
|
2436
942
|
...(session ? { session } : {}),
|
|
2437
943
|
...(sessionError === undefined ? {} : { sessionError }),
|
|
2438
|
-
...(
|
|
2439
|
-
...(desktopDurationMs === undefined ? {} : { desktopDurationMs }),
|
|
2440
|
-
...(desktopResources === undefined ? {} : { desktopResources }),
|
|
2441
|
-
killed,
|
|
2442
|
-
streamUrlPresent: streamUrl !== undefined,
|
|
944
|
+
...desktopLane.snapshot(),
|
|
2443
945
|
screenshots,
|
|
2444
|
-
...(subjectCommit === undefined ? {} : { subjectCommit }),
|
|
2445
|
-
...(desktopBrowser === undefined ? {} : { desktopBrowser }),
|
|
2446
|
-
desktopGeometry,
|
|
2447
|
-
stateStepRecords,
|
|
2448
|
-
phaseRecords,
|
|
2449
946
|
warnings,
|
|
2450
947
|
noEngagement,
|
|
2451
948
|
selfReportedBlocker,
|
|
2452
949
|
reportedFriction,
|
|
2453
950
|
harnessError,
|
|
2454
|
-
...(failureCode === undefined ? {} : { failureCode }),
|
|
2455
|
-
...(commsArtifactPath === undefined ? {} : { commsArtifactPath })
|
|
2456
951
|
};
|
|
2457
952
|
}
|
|
2458
953
|
/** Run the single IN-PROCESS lane (a custom executor + provider; NO E2B). Always one lane. */
|
|
@@ -2462,9 +957,10 @@ async function runInProcessLane(spec, deps) {
|
|
|
2462
957
|
const writeScreenshot = makeLaneWriteScreenshot(deps.artifactRoot, spec, screenshots);
|
|
2463
958
|
let session;
|
|
2464
959
|
let sessionError;
|
|
960
|
+
let provider;
|
|
2465
961
|
try {
|
|
2466
962
|
const executor = await deps.hooks.buildExecutor({ config: deps.config, actor: deps.descriptor, appUrl: deps.appUrl });
|
|
2467
|
-
|
|
963
|
+
provider = await deps.hooks.buildProvider({ config: deps.config, actor: deps.descriptor, lane: spec });
|
|
2468
964
|
const sessionOptions = {
|
|
2469
965
|
instructions: spec.instructions,
|
|
2470
966
|
persona: spec.persona,
|
|
@@ -2474,6 +970,7 @@ async function runInProcessLane(spec, deps) {
|
|
|
2474
970
|
redactScreenshots: deps.redactScreenshots,
|
|
2475
971
|
scrubText: deps.scrubKnownValues,
|
|
2476
972
|
writeScreenshot,
|
|
973
|
+
...(deps.onTrace === undefined ? {} : { onTrace: (items, usage, metadata) => deps.onTrace?.(spec.laneId, items, usage, metadata) }),
|
|
2477
974
|
...(spec.stopWhen === undefined ? {} : { stopWhen: spec.stopWhen }),
|
|
2478
975
|
...(spec.dwell === undefined ? {} : { dwell: spec.dwell }),
|
|
2479
976
|
...(spec.tasks === undefined ? {} : { tasks: spec.tasks })
|
|
@@ -2483,6 +980,14 @@ async function runInProcessLane(spec, deps) {
|
|
|
2483
980
|
catch (error) {
|
|
2484
981
|
sessionError = redactText(deps.scrubKnownValues(toErrorMessage(error)));
|
|
2485
982
|
}
|
|
983
|
+
finally {
|
|
984
|
+
try {
|
|
985
|
+
await provider?.close?.();
|
|
986
|
+
}
|
|
987
|
+
catch {
|
|
988
|
+
sessionError ??= "Model provider cleanup is unconfirmed.";
|
|
989
|
+
}
|
|
990
|
+
}
|
|
2486
991
|
if (session) {
|
|
2487
992
|
await writeContainedOutputFile(deps.artifactRoot, spec.traceArtifactPath, `${JSON.stringify(session.trace, null, 2)}\n`, "utf8");
|
|
2488
993
|
if (session.trace.redaction.screenshots === "raw") {
|
|
@@ -2918,6 +1423,9 @@ async function runCuaActorLabInScope(options) {
|
|
|
2918
1423
|
if (localAppSubject && !inProcessRoute) {
|
|
2919
1424
|
return fail("HUMANISH_CUA_LAB_LOCAL_APP_NO_EXECUTOR", "subject.source: local-app requires a library caller to supply cuaHooks.buildExecutor + buildProvider; there is no built-in driver for an in-process JS contract. (Drive the app via runLab(..., { cuaHooks: { buildExecutor, buildProvider } }).)", descriptor.id);
|
|
2920
1425
|
}
|
|
1426
|
+
if (config.subject.source === "app-url" && config.execution?.target === "local" && !hooks.createDesktopLane) {
|
|
1427
|
+
return fail("HUMANISH_CUA_LAB_LOCAL_APP_NO_EXECUTOR", "Local browser studies require a configured local desktop runtime.", descriptor.id);
|
|
1428
|
+
}
|
|
2921
1429
|
// Re-enforce the fan-out cross-validation (library API surface): lanes XOR count/laneFocus,
|
|
2922
1430
|
// device XOR raw resolution, cap, unique ids, allowPublicTargets+N>1, clone.fanout.
|
|
2923
1431
|
const fanoutReason = cuaLaneValidationReason(config);
|
|
@@ -3010,8 +1518,8 @@ async function runCuaActorLabInScope(options) {
|
|
|
3010
1518
|
// the local-agent route uses a CLI the operator has already signed in to.
|
|
3011
1519
|
if (!dryRun && !inProcessRoute) {
|
|
3012
1520
|
const missingKeys = [
|
|
3013
|
-
...(openaiApiKey || localAgentRoute ? [] : ["OPENAI_API_KEY"]),
|
|
3014
|
-
...(e2bApiKey ? [] : ["E2B_API_KEY"])
|
|
1521
|
+
...(openaiApiKey || localAgentRoute || hooks.buildProvider ? [] : ["OPENAI_API_KEY"]),
|
|
1522
|
+
...(e2bApiKey || hooks.createDesktopLane ? [] : ["E2B_API_KEY"])
|
|
3015
1523
|
];
|
|
3016
1524
|
if (missingKeys.length > 0) {
|
|
3017
1525
|
// The moment someone new actually hits the wall. If a signed-in coding agent is sitting
|
|
@@ -3028,7 +1536,7 @@ async function runCuaActorLabInScope(options) {
|
|
|
3028
1536
|
: "";
|
|
3029
1537
|
return fail("HUMANISH_CUA_LAB_KEYS_MISSING", `Live computer-use labs need ${missingKeys.join(" and ")} in the environment (values are never persisted). ${describeMissingKeys(missingKeys, env)}${suggestion}`, descriptor.id);
|
|
3030
1538
|
}
|
|
3031
|
-
if (localAgentRoute) {
|
|
1539
|
+
if (localAgentRoute && !hooks.buildProvider) {
|
|
3032
1540
|
// Refuse HERE, before a sandbox exists. "codex is not installed" discovered after the
|
|
3033
1541
|
// machine is paid for is the same information delivered at the worst possible moment.
|
|
3034
1542
|
const available = await detectLocalAgents({ env });
|
|
@@ -3125,7 +1633,8 @@ async function runCuaActorLabInScope(options) {
|
|
|
3125
1633
|
let flushLiveTrace;
|
|
3126
1634
|
let stopLiveFlush;
|
|
3127
1635
|
const deps = {
|
|
3128
|
-
|
|
1636
|
+
...(hooks.createDesktopLane ? { createDesktopLane: hooks.createDesktopLane } : {}),
|
|
1637
|
+
onTrace: (laneId, items, usage, metadata) => flushLiveTrace?.(laneId, items, usage, metadata),
|
|
3129
1638
|
config,
|
|
3130
1639
|
descriptor,
|
|
3131
1640
|
appUrl,
|
|
@@ -3241,6 +1750,7 @@ async function runCuaActorLabInScope(options) {
|
|
|
3241
1750
|
// Running token usage per lane, so a run in flight can price itself instead of reporting the
|
|
3242
1751
|
// cost as unknown until the moment it ends.
|
|
3243
1752
|
const liveUsageByStream = new Map();
|
|
1753
|
+
const liveMetadataByStream = new Map();
|
|
3244
1754
|
// The rate the running usage prices at. Usage without its model is not a cost, so both travel
|
|
3245
1755
|
// together or neither does.
|
|
3246
1756
|
const modelForLiveCost = config.actors[0]?.model ?? DEFAULT_OPENAI_CU_MODEL;
|
|
@@ -3279,6 +1789,7 @@ async function runCuaActorLabInScope(options) {
|
|
|
3279
1789
|
// The model too: usage without the rate it prices at is not a cost.
|
|
3280
1790
|
ids: { model: modelForLiveCost }
|
|
3281
1791
|
}),
|
|
1792
|
+
...liveMetadataByStream.get(stream.id),
|
|
3282
1793
|
items: [...liveItems]
|
|
3283
1794
|
}
|
|
3284
1795
|
};
|
|
@@ -3309,7 +1820,7 @@ async function runCuaActorLabInScope(options) {
|
|
|
3309
1820
|
flushTimer.unref?.();
|
|
3310
1821
|
}
|
|
3311
1822
|
};
|
|
3312
|
-
flushLiveTrace = (laneId, items, usage) => {
|
|
1823
|
+
flushLiveTrace = (laneId, items, usage, metadata) => {
|
|
3313
1824
|
// An empty snapshot (the initial observation on a frameless route) carries no
|
|
3314
1825
|
// evidence worth a disk write; the first real item triggers the first flush.
|
|
3315
1826
|
if (items.length === 0)
|
|
@@ -3318,6 +1829,8 @@ async function runCuaActorLabInScope(options) {
|
|
|
3318
1829
|
if (streamId === undefined)
|
|
3319
1830
|
return;
|
|
3320
1831
|
liveItemsByStream.set(streamId, items.slice());
|
|
1832
|
+
if (metadata !== undefined)
|
|
1833
|
+
liveMetadataByStream.set(streamId, metadata);
|
|
3321
1834
|
if (usage !== undefined)
|
|
3322
1835
|
liveUsageByStream.set(streamId, usage);
|
|
3323
1836
|
flushDirty = true;
|
|
@@ -3736,6 +2249,7 @@ function buildSingleLaneBundle(args) {
|
|
|
3736
2249
|
persona: spec.persona,
|
|
3737
2250
|
resolution: spec.resolution,
|
|
3738
2251
|
desktopRoute: !args.inProcessRoute,
|
|
2252
|
+
feedbackSubstrate: args.inProcessRoute ? "local-filesystem" : args.config.execution?.target === "local" ? "local-desktop" : "e2b-desktop",
|
|
3739
2253
|
...(outcome?.desktopGeometry === undefined ? {} : { desktopGeometry: outcome.desktopGeometry }),
|
|
3740
2254
|
isMobile: spec.devicePreset.isMobile,
|
|
3741
2255
|
runId: args.runId,
|
|
@@ -3779,300 +2293,6 @@ function buildSingleLaneBundle(args) {
|
|
|
3779
2293
|
phaseEvents: outcome?.phaseRecords ?? []
|
|
3780
2294
|
});
|
|
3781
2295
|
}
|
|
3782
|
-
/**
|
|
3783
|
-
* Shared post-populate provisioning pipeline (clone AND local-tree routes): (install) ->
|
|
3784
|
-
* state(before-build) -> (build) -> state(before-start) -> detached start -> readiness probe ->
|
|
3785
|
-
* state(after-ready). Both provisioning routes populate SUBJECT_DIR by different means (git
|
|
3786
|
-
* clone vs. upload+extract) and then run this identical pipeline unchanged.
|
|
3787
|
-
*
|
|
3788
|
-
* State steps run through the same detached primitive as serve steps (author-trusted, the
|
|
3789
|
-
* "serve commands are author-trusted" corollary) under the reserved `subject-state-<name>`
|
|
3790
|
-
* label prefix, so a step name can never collide with subject-clone/subject-extract/install/
|
|
3791
|
-
* build/start. after-ready steps complete BEFORE the caller opens the browser: the actor never
|
|
3792
|
-
* drives a half-seeded subject and seeding never eats the session budget.
|
|
3793
|
-
*/
|
|
3794
|
-
async function runSubjectServePipeline(desktop, args) {
|
|
3795
|
-
const timers = {
|
|
3796
|
-
...(args.now === undefined ? {} : { now: args.now }),
|
|
3797
|
-
...(args.sleep === undefined ? {} : { sleep: args.sleep })
|
|
3798
|
-
};
|
|
3799
|
-
const now = args.now ?? Date.now;
|
|
3800
|
-
const refresh = args.onPhaseComplete ?? (() => Promise.resolve());
|
|
3801
|
-
const stateSteps = args.state?.seed ?? [];
|
|
3802
|
-
const runStateSteps = async (when) => {
|
|
3803
|
-
const steps = stateSteps.filter((step) => (step.when ?? "before-start") === when);
|
|
3804
|
-
if (steps.length === 0) {
|
|
3805
|
-
// No declared steps for this group: no boundary to report (avoids empty-group noise on
|
|
3806
|
-
// every run, since before-build/before-start/after-ready are always called).
|
|
3807
|
-
return;
|
|
3808
|
-
}
|
|
3809
|
-
const groupStartedAt = now();
|
|
3810
|
-
emitPhaseStarted(args.onPhase, now, `state.${when}`, `running subject state seed steps (${when})`);
|
|
3811
|
-
for (const step of steps) {
|
|
3812
|
-
const stepTimeoutMs = step.timeoutMs ?? DEFAULT_STATE_STEP_TIMEOUT_MS;
|
|
3813
|
-
const startedAt = now();
|
|
3814
|
-
const result = await runDetachedStep(desktop, {
|
|
3815
|
-
name: `subject-state-${step.name}`,
|
|
3816
|
-
command: step.command,
|
|
3817
|
-
cwd: SUBJECT_DIR,
|
|
3818
|
-
timeoutMs: stepTimeoutMs,
|
|
3819
|
-
requestTimeoutMs: args.requestTimeoutMs,
|
|
3820
|
-
...timers
|
|
3821
|
-
});
|
|
3822
|
-
args.onStateStep?.({
|
|
3823
|
-
name: step.name,
|
|
3824
|
-
when,
|
|
3825
|
-
// Digest only (sha256-16): the command text never persists: the lab YAML in the
|
|
3826
|
-
// consumer's repo is the plaintext source of truth.
|
|
3827
|
-
commandDigest: commandDigestOf(step.command),
|
|
3828
|
-
ok: result.ok,
|
|
3829
|
-
...(result.exitCode === undefined ? {} : { exitCode: result.exitCode }),
|
|
3830
|
-
...(result.timedOut ? { timedOut: true } : {}),
|
|
3831
|
-
durationMs: Math.max(0, now() - startedAt)
|
|
3832
|
-
});
|
|
3833
|
-
if (!result.ok) {
|
|
3834
|
-
emitPhaseCompleted(args.onPhase, now, groupStartedAt, `state.${when}`, false, `subject state seed steps failed (${when})`);
|
|
3835
|
-
// Fail closed with the existing scrub-before-truncate tail chain: literal scrub of
|
|
3836
|
-
// every provisioned value PRE-truncation, then pattern redaction + cap in tailOf.
|
|
3837
|
-
throw new Error(`subject state step "${step.name}" ${result.timedOut ? `timed out after ${stepTimeoutMs}ms` : `failed (exit ${result.exitCode})`}: ${tailOf(args.scrub(result.logTail))}`);
|
|
3838
|
-
}
|
|
3839
|
-
}
|
|
3840
|
-
emitPhaseCompleted(args.onPhase, now, groupStartedAt, `state.${when}`, true, `subject state seed steps complete (${when})`);
|
|
3841
|
-
};
|
|
3842
|
-
// Provide the runtime the pipeline needs before running it (#371). The stock desktop template
|
|
3843
|
-
// ships python3 and curl but no Node, so an `npm install` here used to die at exit 127 after the
|
|
3844
|
-
// sandbox was already paid for. Probe-first, so a template that ships its own Node pays nothing.
|
|
3845
|
-
const serveCommands = [args.serve.install, args.serve.build, args.serve.start];
|
|
3846
|
-
if (needsNodeRuntime(serveCommands)) {
|
|
3847
|
-
const runtimeStartedAt = now();
|
|
3848
|
-
emitPhaseStarted(args.onPhase, now, "runtime", "providing the Node runtime the serve pipeline needs");
|
|
3849
|
-
const bootstrap = await runProvisioningStepWithOneRetry(desktop, {
|
|
3850
|
-
name: "subject-runtime-node",
|
|
3851
|
-
command: nodeBootstrapCommand(),
|
|
3852
|
-
cwd: SUBJECT_DIR,
|
|
3853
|
-
timeoutMs: args.serve.installTimeoutMs ?? INSTALL_TIMEOUT_MS,
|
|
3854
|
-
requestTimeoutMs: args.requestTimeoutMs,
|
|
3855
|
-
timers,
|
|
3856
|
-
retryPhase: "runtime-retry",
|
|
3857
|
-
retryMessage: "Node runtime bootstrap",
|
|
3858
|
-
onPhase: args.onPhase,
|
|
3859
|
-
now
|
|
3860
|
-
});
|
|
3861
|
-
let ok = bootstrap.ok;
|
|
3862
|
-
const corepack = ok ? corepackCommandFor(serveCommands) : undefined;
|
|
3863
|
-
if (corepack) {
|
|
3864
|
-
const pm = await runDetachedStep(desktop, {
|
|
3865
|
-
name: "subject-runtime-pm",
|
|
3866
|
-
command: corepack,
|
|
3867
|
-
cwd: SUBJECT_DIR,
|
|
3868
|
-
timeoutMs: args.serve.installTimeoutMs ?? INSTALL_TIMEOUT_MS,
|
|
3869
|
-
requestTimeoutMs: args.requestTimeoutMs,
|
|
3870
|
-
...timers
|
|
3871
|
-
});
|
|
3872
|
-
ok = pm.ok;
|
|
3873
|
-
}
|
|
3874
|
-
emitPhaseCompleted(args.onPhase, now, runtimeStartedAt, "runtime", ok, ok ? "Node runtime ready" : "could not provide a Node runtime");
|
|
3875
|
-
if (!ok) {
|
|
3876
|
-
throw new Error(`the subject's serve pipeline needs a Node runtime and this desktop template has none, and bootstrapping one failed${bootstrap.attempts === 2 ? " twice" : ""}: ${tailOf(args.scrub(bootstrap.logTail))}. Use execution.desktop.template with an image that ships Node, or change serve.install to a runtime the template provides.`);
|
|
3877
|
-
}
|
|
3878
|
-
}
|
|
3879
|
-
if (args.serve.install) {
|
|
3880
|
-
const installStartedAt = now();
|
|
3881
|
-
emitPhaseStarted(args.onPhase, now, "install", "installing subject dependencies");
|
|
3882
|
-
const install = await runProvisioningStepWithOneRetry(desktop, {
|
|
3883
|
-
name: "subject-install",
|
|
3884
|
-
command: args.serve.install,
|
|
3885
|
-
cwd: SUBJECT_DIR,
|
|
3886
|
-
timeoutMs: args.serve.installTimeoutMs ?? INSTALL_TIMEOUT_MS,
|
|
3887
|
-
requestTimeoutMs: args.requestTimeoutMs,
|
|
3888
|
-
timers,
|
|
3889
|
-
retryPhase: "install-retry",
|
|
3890
|
-
retryMessage: "subject install",
|
|
3891
|
-
onPhase: args.onPhase,
|
|
3892
|
-
now
|
|
3893
|
-
});
|
|
3894
|
-
emitPhaseCompleted(args.onPhase, now, installStartedAt, "install", install.ok, install.ok
|
|
3895
|
-
? install.attempts === 2
|
|
3896
|
-
? "subject dependencies installed (on the second attempt)"
|
|
3897
|
-
: "subject dependencies installed"
|
|
3898
|
-
: install.attempts === 2
|
|
3899
|
-
? "subject install failed twice"
|
|
3900
|
-
: "subject install failed");
|
|
3901
|
-
if (!install.ok) {
|
|
3902
|
-
// Lead with the line a person can act on; npm's own trace follows it (#602).
|
|
3903
|
-
const headline = install.timedOut
|
|
3904
|
-
? `subject install timed out after ${args.serve.installTimeoutMs ?? INSTALL_TIMEOUT_MS}ms`
|
|
3905
|
-
: install.attempts === 2
|
|
3906
|
-
? `subject install failed twice (exit ${install.firstExitCode ?? "null"}, then exit ${install.exitCode ?? "null"}); the sandbox could not complete serve.install`
|
|
3907
|
-
: `subject install failed (exit ${install.exitCode ?? "null"})`;
|
|
3908
|
-
throw new Error(`${headline}: ${tailOf(args.scrub(install.logTail))}`);
|
|
3909
|
-
}
|
|
3910
|
-
await refresh();
|
|
3911
|
-
}
|
|
3912
|
-
// before-build: after install, before build (builds that read seeded state, e.g. SSG).
|
|
3913
|
-
// When no build is declared this simply precedes start: equivalent to before-start.
|
|
3914
|
-
await runStateSteps("before-build");
|
|
3915
|
-
await refresh();
|
|
3916
|
-
if (args.serve.build) {
|
|
3917
|
-
const buildStartedAt = now();
|
|
3918
|
-
emitPhaseStarted(args.onPhase, now, "build", "building subject");
|
|
3919
|
-
const build = await runDetachedStep(desktop, {
|
|
3920
|
-
name: "subject-build",
|
|
3921
|
-
command: args.serve.build,
|
|
3922
|
-
cwd: SUBJECT_DIR,
|
|
3923
|
-
timeoutMs: args.serve.buildTimeoutMs ?? BUILD_TIMEOUT_MS,
|
|
3924
|
-
requestTimeoutMs: args.requestTimeoutMs,
|
|
3925
|
-
...timers
|
|
3926
|
-
});
|
|
3927
|
-
emitPhaseCompleted(args.onPhase, now, buildStartedAt, "build", build.ok, build.ok ? "subject build complete" : "subject build failed");
|
|
3928
|
-
if (!build.ok) {
|
|
3929
|
-
throw new Error(`subject build ${build.timedOut ? "timed out" : `failed (exit ${build.exitCode})`}: ${tailOf(args.scrub(build.logTail))}`);
|
|
3930
|
-
}
|
|
3931
|
-
await refresh();
|
|
3932
|
-
}
|
|
3933
|
-
// before-start (the default phase): migrations, SQL/file fixtures, an in-sandbox DB server
|
|
3934
|
-
// (`sudo service postgresql start && pg_isready` is a bounded step; the daemon it forks is
|
|
3935
|
-
// reclaimed by the sandbox lifecycle like everything else).
|
|
3936
|
-
await runStateSteps("before-start");
|
|
3937
|
-
await refresh();
|
|
3938
|
-
await startDetachedProcess(desktop, {
|
|
3939
|
-
name: "subject-start",
|
|
3940
|
-
command: args.serve.start,
|
|
3941
|
-
cwd: SUBJECT_DIR,
|
|
3942
|
-
requestTimeoutMs: args.requestTimeoutMs
|
|
3943
|
-
});
|
|
3944
|
-
// Fire-and-forget: startDetachedProcess never waits for the long-lived server to exit, so
|
|
3945
|
-
// there is no matching completed event here (no ok/durationMs to report yet); readiness is
|
|
3946
|
-
// the next boundary.
|
|
3947
|
-
args.onPhase?.({ at: isoNow(now), type: "cua-lab.subject.serve.started", message: "subject server launched (detached)" });
|
|
3948
|
-
const readyStartedAt = now();
|
|
3949
|
-
emitPhaseStarted(args.onPhase, now, "ready", "waiting for subject to become ready");
|
|
3950
|
-
const ready = await probeUrl(desktop, args.serve.url, {
|
|
3951
|
-
timeoutMs: args.serve.readyTimeoutMs ?? DEFAULT_READY_TIMEOUT_MS,
|
|
3952
|
-
requestTimeoutMs: args.requestTimeoutMs,
|
|
3953
|
-
...timers
|
|
3954
|
-
});
|
|
3955
|
-
emitPhaseCompleted(args.onPhase, now, readyStartedAt, "ready", ready, ready ? "subject is ready" : "subject did not become ready in time");
|
|
3956
|
-
if (!ready) {
|
|
3957
|
-
const startLog = await readDetachedLog(desktop, "subject-start", args.requestTimeoutMs).catch(() => "");
|
|
3958
|
-
throw new Error(`subject did not answer at ${args.serve.url} within ${args.serve.readyTimeoutMs ?? DEFAULT_READY_TIMEOUT_MS}ms; server log tail: ${tailOf(args.scrub(startLog))}`);
|
|
3959
|
-
}
|
|
3960
|
-
// after-ready: fixture loading through the RUNNING app (loopback curl from in-sandbox:
|
|
3961
|
-
// steps are author-trusted provisioning, not actors, so no new URL policy surface). These
|
|
3962
|
-
// complete before the caller opens the browser and the session timer starts.
|
|
3963
|
-
await runStateSteps("after-ready");
|
|
3964
|
-
await refresh();
|
|
3965
|
-
}
|
|
3966
|
-
/**
|
|
3967
|
-
* Provision a clone subject inside the sandbox: clone → the shared serve pipeline
|
|
3968
|
-
* (install → state(before-build) → build → state(before-start) → start → readiness
|
|
3969
|
-
* probe → state(after-ready)). Returns the latest subject HEAD after successful
|
|
3970
|
-
* provisioning. Throws (with a capped log tail for the caller to redact) on any failing step:
|
|
3971
|
-
* the lab persists that as a failed-evidence bundle.
|
|
3972
|
-
*
|
|
3973
|
-
* Auth: when GITHUB_TOKEN is among the declared subject env names, the clone authenticates
|
|
3974
|
-
* via an Authorization header computed IN-SANDBOX from the provisioned env: the token never
|
|
3975
|
-
* appears in the script text, the process argv beyond the transient git call, the clone URL,
|
|
3976
|
-
* or .git/config.
|
|
3977
|
-
*/
|
|
3978
|
-
export async function provisionCloneSubject(desktop, args) {
|
|
3979
|
-
const timers = {
|
|
3980
|
-
...(args.now === undefined ? {} : { now: args.now }),
|
|
3981
|
-
...(args.sleep === undefined ? {} : { sleep: args.sleep })
|
|
3982
|
-
};
|
|
3983
|
-
const now = args.now ?? Date.now;
|
|
3984
|
-
let latestCommit;
|
|
3985
|
-
const refreshCommit = async () => {
|
|
3986
|
-
const head = await desktop.commands.run(`git -C ${SUBJECT_DIR} rev-parse HEAD 2>/dev/null || true`, { requestTimeoutMs: args.requestTimeoutMs });
|
|
3987
|
-
const commit = (head.stdout ?? "").trim() || undefined;
|
|
3988
|
-
if (commit) {
|
|
3989
|
-
latestCommit = commit;
|
|
3990
|
-
args.onCommit?.(commit);
|
|
3991
|
-
}
|
|
3992
|
-
};
|
|
3993
|
-
const cloneCommand = args.hasGithubToken
|
|
3994
|
-
? `auth=$(printf 'x-access-token:%s' "$GITHUB_TOKEN" | base64 -w0) && git -c http.extraHeader="Authorization: Basic $auth" clone --depth ${args.depth} https://github.com/${args.repo}.git ${SUBJECT_DIR}`
|
|
3995
|
-
: `git clone --depth ${args.depth} https://github.com/${args.repo}.git ${SUBJECT_DIR}`;
|
|
3996
|
-
const cloneStartedAt = now();
|
|
3997
|
-
emitPhaseStarted(args.onPhase, now, "clone", "cloning subject repository");
|
|
3998
|
-
const clone = await runDetachedStep(desktop, {
|
|
3999
|
-
name: "subject-clone",
|
|
4000
|
-
command: cloneCommand,
|
|
4001
|
-
timeoutMs: CLONE_TIMEOUT_MS,
|
|
4002
|
-
requestTimeoutMs: args.requestTimeoutMs,
|
|
4003
|
-
...timers
|
|
4004
|
-
});
|
|
4005
|
-
emitPhaseCompleted(args.onPhase, now, cloneStartedAt, "clone", clone.ok, clone.ok ? "subject repository cloned" : "subject clone failed");
|
|
4006
|
-
if (!clone.ok) {
|
|
4007
|
-
throw new Error(`subject clone ${clone.timedOut ? "timed out" : `failed (exit ${clone.exitCode})`}: ${tailOf(args.scrub(clone.logTail))}`);
|
|
4008
|
-
}
|
|
4009
|
-
await refreshCommit();
|
|
4010
|
-
await runSubjectServePipeline(desktop, {
|
|
4011
|
-
serve: args.serve,
|
|
4012
|
-
...(args.state === undefined ? {} : { state: args.state }),
|
|
4013
|
-
requestTimeoutMs: args.requestTimeoutMs,
|
|
4014
|
-
scrub: args.scrub,
|
|
4015
|
-
...(args.onStateStep === undefined ? {} : { onStateStep: args.onStateStep }),
|
|
4016
|
-
...(args.onPhase === undefined ? {} : { onPhase: args.onPhase }),
|
|
4017
|
-
onPhaseComplete: refreshCommit,
|
|
4018
|
-
...timers
|
|
4019
|
-
});
|
|
4020
|
-
return latestCommit;
|
|
4021
|
-
}
|
|
4022
|
-
/**
|
|
4023
|
-
* Provision a local-tree subject inside the sandbox: upload the once-per-run packed archive
|
|
4024
|
-
* (identical bytes across every fan-out lane) → extract it into SUBJECT_DIR → the
|
|
4025
|
-
* same shared serve pipeline provisionCloneSubject uses. Unlike the clone route there is no
|
|
4026
|
-
* in-sandbox git refresh: the archive excludes .git entirely (see source-archive.ts), so
|
|
4027
|
-
* subject identity is the host-side LocalTreeArchive captured at pack time, never anything
|
|
4028
|
-
* resolved in-sandbox.
|
|
4029
|
-
*/
|
|
4030
|
-
export async function provisionLocalTreeSubject(desktop, args) {
|
|
4031
|
-
const timers = {
|
|
4032
|
-
...(args.now === undefined ? {} : { now: args.now }),
|
|
4033
|
-
...(args.sleep === undefined ? {} : { sleep: args.sleep })
|
|
4034
|
-
};
|
|
4035
|
-
const now = args.now ?? Date.now;
|
|
4036
|
-
const uploadStartedAt = now();
|
|
4037
|
-
emitPhaseStarted(args.onPhase, now, "upload", "uploading packed local-tree archive");
|
|
4038
|
-
try {
|
|
4039
|
-
await withOneRetryOnTransientE2BError(() => desktop.files.write(LOCAL_TREE_REMOTE_ARCHIVE_PATH, args.archiveBuffer, {
|
|
4040
|
-
requestTimeoutMs: args.requestTimeoutMs,
|
|
4041
|
-
useOctetStream: true
|
|
4042
|
-
}), {
|
|
4043
|
-
onRetry: (reason) => emitPhaseStarted(args.onPhase, now, "upload-retry", `local-tree archive upload retried once (${tailOf(args.scrub(reason))})`),
|
|
4044
|
-
...(args.sleep === undefined ? {} : { sleep: args.sleep })
|
|
4045
|
-
});
|
|
4046
|
-
}
|
|
4047
|
-
catch (error) {
|
|
4048
|
-
emitPhaseCompleted(args.onPhase, now, uploadStartedAt, "upload", false, "local-tree archive upload failed");
|
|
4049
|
-
throw new Error(`subject-upload failed: ${tailOf(args.scrub(toErrorMessage(error)))}`);
|
|
4050
|
-
}
|
|
4051
|
-
emitPhaseCompleted(args.onPhase, now, uploadStartedAt, "upload", true, "local-tree archive uploaded");
|
|
4052
|
-
const extractCommand = `rm -rf ${SUBJECT_DIR} && mkdir -p ${SUBJECT_DIR} && tar -xzf ${LOCAL_TREE_REMOTE_ARCHIVE_PATH} -C ${SUBJECT_DIR} && rm -f ${LOCAL_TREE_REMOTE_ARCHIVE_PATH}`;
|
|
4053
|
-
const extractStartedAt = now();
|
|
4054
|
-
emitPhaseStarted(args.onPhase, now, "extract", "extracting local-tree archive");
|
|
4055
|
-
const extract = await runDetachedStep(desktop, {
|
|
4056
|
-
name: "subject-extract",
|
|
4057
|
-
command: extractCommand,
|
|
4058
|
-
timeoutMs: CLONE_TIMEOUT_MS,
|
|
4059
|
-
requestTimeoutMs: args.requestTimeoutMs,
|
|
4060
|
-
...timers
|
|
4061
|
-
});
|
|
4062
|
-
emitPhaseCompleted(args.onPhase, now, extractStartedAt, "extract", extract.ok, extract.ok ? "local-tree archive extracted" : "local-tree archive extraction failed");
|
|
4063
|
-
if (!extract.ok) {
|
|
4064
|
-
throw new Error(`subject extract ${extract.timedOut ? "timed out" : `failed (exit ${extract.exitCode})`}: ${tailOf(args.scrub(extract.logTail))}`);
|
|
4065
|
-
}
|
|
4066
|
-
await runSubjectServePipeline(desktop, {
|
|
4067
|
-
serve: args.serve,
|
|
4068
|
-
...(args.state === undefined ? {} : { state: args.state }),
|
|
4069
|
-
requestTimeoutMs: args.requestTimeoutMs,
|
|
4070
|
-
scrub: args.scrub,
|
|
4071
|
-
...(args.onStateStep === undefined ? {} : { onStateStep: args.onStateStep }),
|
|
4072
|
-
...(args.onPhase === undefined ? {} : { onPhase: args.onPhase }),
|
|
4073
|
-
...timers
|
|
4074
|
-
});
|
|
4075
|
-
}
|
|
4076
2296
|
/**
|
|
4077
2297
|
* Default local-tree packing implementation: createLocalTreeArchive(root, opts) on the host,
|
|
4078
2298
|
* then a single read of the produced archive file into an ArrayBuffer for upload. The DI seam
|
|
@@ -4092,10 +2312,6 @@ export async function defaultPackLocalTree(args) {
|
|
|
4092
2312
|
await rm(path.dirname(archive.archivePath), { recursive: true, force: true }).catch(() => undefined);
|
|
4093
2313
|
return { archive, buffer };
|
|
4094
2314
|
}
|
|
4095
|
-
/** sha256 hex of the exact command string, first 16 chars (the promptDigest convention). */
|
|
4096
|
-
export function commandDigestOf(command) {
|
|
4097
|
-
return digestText(command, 16);
|
|
4098
|
-
}
|
|
4099
2315
|
/**
|
|
4100
2316
|
* Resolve the bundle's state marker from the declaration and what actually ran.
|
|
4101
2317
|
* Precedence: external declared → "unpinned" (seed records, if any, stay attached — a
|
|
@@ -4151,10 +2367,6 @@ function describeSubjectState(state, dryRun) {
|
|
|
4151
2367
|
return "external-public (operator-declared, operator-owned public deployment; neither provisioned nor seeded)";
|
|
4152
2368
|
}
|
|
4153
2369
|
}
|
|
4154
|
-
// The in-sandbox `tail -c` upstream is a fundamental log-tail limit we cannot redact past.
|
|
4155
|
-
function tailOf(log) {
|
|
4156
|
-
return redactedTail(log, ERROR_TAIL_CHARS);
|
|
4157
|
-
}
|
|
4158
2370
|
export function buildCuaCostSummary(args) {
|
|
4159
2371
|
const breakdown = [];
|
|
4160
2372
|
let sumInput = 0;
|
|
@@ -4188,7 +2400,8 @@ export function buildCuaCostSummary(args) {
|
|
|
4188
2400
|
ratesAsOf: null
|
|
4189
2401
|
});
|
|
4190
2402
|
}
|
|
4191
|
-
const est = lane.trace.
|
|
2403
|
+
const est = lane.trace.executionProfile?.billing === "account-unknown"
|
|
2404
|
+
? estimateActorCostForExecution(usage, lane.trace.ids.model, lane.trace.executionProfile) : lane.trace.estimatedCost;
|
|
4192
2405
|
if (!est) {
|
|
4193
2406
|
continue;
|
|
4194
2407
|
}
|
|
@@ -4264,7 +2477,7 @@ export function buildCuaCostSummary(args) {
|
|
|
4264
2477
|
}
|
|
4265
2478
|
const estimatedTotalUsd = anyKnown ? round6(knownSum) : null;
|
|
4266
2479
|
const estimateNote = estimatedTotalUsd === null
|
|
4267
|
-
? `No priced spend lines this run — every cost line is DECLARED ABSENT (unknown rate / no usage / no duration); nothing is guessed. Add a rate to src/pricing.ts to estimate this model
|
|
2480
|
+
? `No priced spend lines this run — every cost line is DECLARED ABSENT (unknown rate / no usage / no duration); nothing is guessed. ${args.lanes.some(lane => lane.trace.executionProfile?.billing === "account-unknown") ? "Account billing remains unknown; API prices do not measure account spend." : "Add a rate to src/pricing.ts to estimate this model."}`
|
|
4268
2481
|
: `Estimated ${estimatedTotalUsd} USD total${anyNull ? " (LOWER BOUND — some lines unmeasured/unpriced)" : ""}${placeholder ? "; includes PLACEHOLDER rate(s) — confirm before trusting the magnitude" : ""}. Every figure is an ESTIMATE (rates as of ${minRatesAsOf} — the OLDEST contributing rate, since an aggregate is only as fresh as its stalest input), a rate-table multiply, NOT an authoritative provider charge.`;
|
|
4269
2482
|
const note = estimateNote + ((args.desktops?.length ?? 0) > 0
|
|
4270
2483
|
? args.desktops.some(usage => usage.observation !== undefined && "resources" in usage.observation)
|
|
@@ -4279,7 +2492,14 @@ export function buildCuaCostSummary(args) {
|
|
|
4279
2492
|
fullyEstimated: !anyNull,
|
|
4280
2493
|
placeholder,
|
|
4281
2494
|
breakdown,
|
|
4282
|
-
tokenUsage:
|
|
2495
|
+
tokenUsage: args.lanes.some(lane => lane.trace.executionProfile?.billing === "account-unknown") ? {
|
|
2496
|
+
...(args.lanes.some(lane => lane.trace.tokenUsage?.input !== undefined) ? { input: sumInput } : {}),
|
|
2497
|
+
...(args.lanes.some(lane => lane.trace.tokenUsage?.output !== undefined) ? { output: sumOutput } : {}),
|
|
2498
|
+
...(args.lanes.every(lane => lane.trace.tokenUsage?.input !== undefined && lane.trace.tokenUsage?.output !== undefined &&
|
|
2499
|
+
lane.trace.interactionUsageIncomplete !== true && lane.trace.debrief?.usageReported !== false &&
|
|
2500
|
+
(lane.trace.executionProfile === undefined || lane.trace.providerRequests?.every(r => r.usageComplete) === true))
|
|
2501
|
+
? { total: sumInput + sumOutput } : {})
|
|
2502
|
+
} : { input: sumInput, output: sumOutput, total: sumInput + sumOutput },
|
|
4283
2503
|
desktopMinutes: args.desktops === undefined ? args.desktopMinutes ?? null
|
|
4284
2504
|
: args.desktops.some(usage => usage.minutes !== undefined)
|
|
4285
2505
|
? round6(args.desktops.reduce((sum, usage) => sum + (usage.minutes ?? 0), 0)) : null,
|
|
@@ -4674,7 +2894,7 @@ export function buildCuaBundle(args) {
|
|
|
4674
2894
|
scenarioId: `cua-${args.labId}`,
|
|
4675
2895
|
adapterId: args.labId,
|
|
4676
2896
|
goal: redactText(args.mission),
|
|
4677
|
-
substrate: args.desktopRoute === false ? "local-filesystem" : "e2b-desktop",
|
|
2897
|
+
substrate: args.feedbackSubstrate ?? (args.desktopRoute === false ? "local-filesystem" : "e2b-desktop"),
|
|
4678
2898
|
lanes: [{
|
|
4679
2899
|
laneId: args.laneId ?? "lane-01",
|
|
4680
2900
|
streamId: "stream-001",
|
|
@@ -5170,7 +3390,7 @@ export function buildCuaFanoutBundle(args) {
|
|
|
5170
3390
|
scenarioId: `cua-${config.id}`,
|
|
5171
3391
|
adapterId: config.id,
|
|
5172
3392
|
goal: redactText(specs[0].evidenceInstructions ?? specs[0].instructions),
|
|
5173
|
-
substrate: "e2b-desktop",
|
|
3393
|
+
substrate: config.execution?.target === "local" ? "local-desktop" : "e2b-desktop",
|
|
5174
3394
|
lanes: specs.map((spec, index) => {
|
|
5175
3395
|
const outcome = outcomes?.[index];
|
|
5176
3396
|
return {
|