humanish 0.87.0 → 0.88.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (55) hide show
  1. package/README.md +12 -2
  2. package/dist/actor-contract.d.ts +4 -1
  3. package/dist/actor-contract.js.map +1 -1
  4. package/dist/actor-goal-source.d.ts +5 -0
  5. package/dist/actor-goal-source.js +32 -0
  6. package/dist/actor-goal-source.js.map +1 -0
  7. package/dist/actor-stop-cause.d.ts +2 -0
  8. package/dist/actor-stop-cause.js +7 -2
  9. package/dist/actor-stop-cause.js.map +1 -1
  10. package/dist/computer-use.d.ts +3 -0
  11. package/dist/computer-use.js +30 -4
  12. package/dist/computer-use.js.map +1 -1
  13. package/dist/cua-actor-lab.d.ts +10 -3
  14. package/dist/cua-actor-lab.js +35 -10
  15. package/dist/cua-actor-lab.js.map +1 -1
  16. package/dist/cua-admission-limit.d.ts +12 -0
  17. package/dist/cua-admission-limit.js +20 -0
  18. package/dist/cua-admission-limit.js.map +1 -0
  19. package/dist/cua-diagnostics.d.ts +38 -0
  20. package/dist/cua-diagnostics.js +68 -0
  21. package/dist/cua-diagnostics.js.map +1 -0
  22. package/dist/e2b-desktop-executor.js +4 -2
  23. package/dist/e2b-desktop-executor.js.map +1 -1
  24. package/dist/feedback.js +5 -3
  25. package/dist/feedback.js.map +1 -1
  26. package/dist/index.d.ts +2 -1
  27. package/dist/index.js +2 -1
  28. package/dist/index.js.map +1 -1
  29. package/dist/observer-data.js +11 -4
  30. package/dist/observer-data.js.map +1 -1
  31. package/dist/openai-responses-cu.js +13 -2
  32. package/dist/openai-responses-cu.js.map +1 -1
  33. package/dist/program.d.ts +2 -0
  34. package/dist/program.js +9 -4
  35. package/dist/program.js.map +1 -1
  36. package/dist/run.d.ts +18 -5
  37. package/dist/run.js +70 -2
  38. package/dist/run.js.map +1 -1
  39. package/dist/stats.js +1 -1
  40. package/dist/stats.js.map +1 -1
  41. package/dist/telemetry.d.ts +3 -0
  42. package/dist/telemetry.js +15 -1
  43. package/dist/telemetry.js.map +1 -1
  44. package/docs/architecture/examples/state-driven-local-app/README.md +75 -0
  45. package/docs/architecture/examples/state-driven-local-app/app.mjs +48 -0
  46. package/docs/architecture/examples/state-driven-local-app/runner.mjs +104 -0
  47. package/docs/architecture/state-driven-executor.md +19 -26
  48. package/docs/contracts/adapter-admission.md +54 -0
  49. package/docs/contracts/run-bundle.md +8 -1
  50. package/docs/contracts/schemas.md +16 -2
  51. package/docs/goals/current.md +9 -4
  52. package/docs/ramp/README.md +11 -1
  53. package/docs/release/0.88.0-study-diagnostics.md +43 -0
  54. package/docs/release/0.88.1-completion-evidence-and-local-app.md +79 -0
  55. package/package.json +1 -1
@@ -25,6 +25,7 @@ import { randomBytes } from "node:crypto";
25
25
  import { describeMissingKeys } from "./key-resolution.js";
26
26
  import { readFile, realpath, rm } from "node:fs/promises";
27
27
  import path from "node:path";
28
+ import { cuaLaneDiagnostics, summarizeCuaDiagnostics } from "./cua-diagnostics.js";
28
29
  import { feedbackProofCommands } from "./feedback-proof.js";
29
30
  import { runDesktopCommandOrThrow, toErrorMessage } from "./command-failure.js";
30
31
  import { pathToFileURL } from "node:url";
@@ -56,10 +57,11 @@ import { renderTaskPrompt } from "./tasks.js";
56
57
  import { attachObserverRuntimeStreamUrls, renderObserver } from "./observer.js";
57
58
  import { containsSensitive, digestText, redactedTail, redactText } from "./redaction.js";
58
59
  import { participantAssignment } from "./participant-assignment.js";
60
+ import { actorEnding } from "./actor-stop-cause.js";
59
61
  import { assertPreparedSelectedOutputDirectory, assertSafeOutputPathSegment, prepareContainedOutputDirectory, prepareSelectedOutputDirectory, writeContainedOutputFile, writePreparedRunLatestPointer } from "./selected-output-paths.js";
60
62
  import { prepareRunArtifactPaths, validatePreparedRunArtifactPaths } from "./run-paths.js";
61
63
  import { createLocalTreeArchive } from "./source-archive.js";
62
- import { buildRunSource, loadRunBundle, PUBLIC_TARGET_CWD, REVIEW_SCHEMA, RUN_BUNDLE_SCHEMA, aggregateTaskFunnels, formatParticipantOutcomes, formatStudyTaskFunnel, tallyParticipantOutcomes } from "./run.js";
64
+ import { buildRunSource, loadRunBundle, PUBLIC_TARGET_CWD, REVIEW_SCHEMA, RUN_BUNDLE_SCHEMA, aggregateTaskFunnels, formatParticipantOutcomes, formatStudyTaskFunnel, tallyParticipantOutcomes, withCuaReviewProvenance } from "./run.js";
63
65
  import { estimateActorCost, estimateDesktopCost, estimateAllocatedDesktopCost, MODEL_RATES, round6 } from "./pricing.js";
64
66
  import { observeDesktopResources } from "./e2b-desktop-resources.js";
65
67
  export const CUA_ACTOR_LAB_SCHEMA = "humanish.cua-lab-result.v2";
@@ -2539,13 +2541,14 @@ function toLaneResult(spec, outcome, subject, dryRun) {
2539
2541
  subject
2540
2542
  };
2541
2543
  if (!outcome || dryRun) {
2542
- return { ...base, status: "contract_proof_only", ok: dryRun };
2544
+ return { ...base, status: "contract_proof_only", ok: dryRun, diagnostics: cuaLaneDiagnostics({ dryRun }) };
2543
2545
  }
2544
2546
  if (outcome.skippedReason !== undefined) {
2545
2547
  return {
2546
2548
  ...base,
2547
2549
  status: "blocked",
2548
2550
  ok: false,
2551
+ diagnostics: cuaLaneDiagnostics({ dryRun, skipped: true }),
2549
2552
  skippedReason: outcome.skippedReason,
2550
2553
  error: { code: "HUMANISH_CUA_LAB_FAILED", message: outcome.skippedReason }
2551
2554
  };
@@ -2557,11 +2560,16 @@ function toLaneResult(spec, outcome, subject, dryRun) {
2557
2560
  ...base,
2558
2561
  status,
2559
2562
  ok: laneOk,
2563
+ diagnostics: cuaLaneDiagnostics({
2564
+ dryRun, executionError: outcome.sessionError !== undefined, noEngagement: outcome.noEngagement,
2565
+ ...(session ? { session: { status: session.status, completionReason: session.completionReason, ...(session.trace.stopCause === undefined ? {} : { stopCause: session.trace.stopCause }) } } : {})
2566
+ }),
2560
2567
  ...(session
2561
2568
  ? {
2562
2569
  session: {
2563
2570
  status: session.status,
2564
2571
  completionReason: session.completionReason,
2572
+ ...(session.trace.stopCause === undefined ? {} : { stopCause: session.trace.stopCause }),
2565
2573
  reason: session.reason,
2566
2574
  screenshots: outcome.screenshots.length
2567
2575
  }
@@ -3487,6 +3495,7 @@ async function runCuaActorLabInScope(options) {
3487
3495
  session: {
3488
3496
  status: firstOutcome.session.status,
3489
3497
  completionReason: firstOutcome.session.completionReason,
3498
+ ...(firstOutcome.session.trace.stopCause === undefined ? {} : { stopCause: firstOutcome.session.trace.stopCause }),
3490
3499
  reason: firstOutcome.session.reason,
3491
3500
  screenshots: firstOutcome.screenshots.length
3492
3501
  }
@@ -3498,6 +3507,7 @@ async function runCuaActorLabInScope(options) {
3498
3507
  subject: aggregateSubject,
3499
3508
  plan,
3500
3509
  lanes: laneResults,
3510
+ diagnostics: summarizeCuaDiagnostics({ dryRun, evidenceInvalid: !observer.ok, lanes: laneResults }),
3501
3511
  laneSummary,
3502
3512
  ...(rerunLineage === undefined ? {} : { rerun: rerunLineage }),
3503
3513
  observer,
@@ -4031,8 +4041,19 @@ export function buildCuaCostSummary(args) {
4031
4041
  sumInput += usage.input ?? 0;
4032
4042
  sumOutput += usage.output ?? 0;
4033
4043
  }
4034
- // An attempted closing request can fail after provider work without reporting usage.
4035
- // Keep the known interaction estimate and make the additional unknown explicit.
4044
+ // A stalled/ambiguous interaction can remain unreported after a later successful retry.
4045
+ // Keep known token estimates and make the additional unknown explicit.
4046
+ if (lane.trace.interactionUsageIncomplete === true) {
4047
+ breakdown.push({
4048
+ kind: "model-tokens",
4049
+ ...(lane.laneId === undefined ? {} : { laneId: lane.laneId }),
4050
+ ...(lane.trace.providerVersion === undefined ? {} : { modelId: lane.trace.providerVersion }),
4051
+ estimatedCostUsd: null,
4052
+ reason: "interaction_usage_unreported",
4053
+ ratesAsOf: null
4054
+ });
4055
+ }
4056
+ // An attempted closing request has its own accounting boundary.
4036
4057
  if (lane.trace.debrief?.usageReported === false) {
4037
4058
  breakdown.push({
4038
4059
  kind: "model-tokens",
@@ -4444,7 +4465,7 @@ export function buildCuaBundle(args) {
4444
4465
  : args.credibility?.noEngagement === true
4445
4466
  ? "Not counted as a pass: the participant took no actions and said nothing."
4446
4467
  : "Not counted as a pass: the participant's final message described a blocker.";
4447
- const review = {
4468
+ const review = withCuaReviewProvenance({
4448
4469
  schema: REVIEW_SCHEMA,
4449
4470
  verdict: args.inProgress === true
4450
4471
  ? "contract_proof_only"
@@ -4465,7 +4486,7 @@ export function buildCuaBundle(args) {
4465
4486
  : args.inProgress === true
4466
4487
  ? ["Live desktop session is still running."]
4467
4488
  : ["Live desktop session not yet run (dry-run contract only)."]
4468
- };
4489
+ }, [stream]);
4469
4490
  return {
4470
4491
  schema: RUN_BUNDLE_SCHEMA,
4471
4492
  runId: args.runId,
@@ -4897,7 +4918,11 @@ export function buildCuaFanoutBundle(args) {
4897
4918
  .map((outcome) => outcome?.session?.trace.taskFunnel)
4898
4919
  .filter((funnel) => funnel !== undefined);
4899
4920
  const studyTasks = args.inProgress === true ? undefined : aggregateTaskFunnels(participantFunnels);
4900
- const review = {
4921
+ const participantEndings = terminalOutcomes.map((outcome) => {
4922
+ const ending = actorEnding(outcome.session.trace);
4923
+ return { status: outcome.session.status, ...(ending === undefined ? {} : { label: ending.label }) };
4924
+ });
4925
+ const review = withCuaReviewProvenance({
4901
4926
  schema: REVIEW_SCHEMA,
4902
4927
  verdict,
4903
4928
  ...(participants === undefined ? {} : { participants }),
@@ -4906,7 +4931,7 @@ export function buildCuaFanoutBundle(args) {
4906
4931
  ? `Live computer-use fan-out is running (${specs.length} per-lane worlds); terminal lane evidence has not been written yet.`
4907
4932
  : args.dryRun
4908
4933
  ? `${args.rerun ? `Rerun contract from ${args.rerun.sourceRunId}: ` : ""}Dry-run fan-out contract: ${specs.length} per-lane-world lanes composed for ${args.descriptor.id} against ${args.appUrl}; no desktops launched, $0 spend.`
4909
- : `${args.rerun ? `Rerun from ${args.rerun.sourceRunId}: ` : ""}Computer-use fan-out (${specs.length} per-lane worlds): ${passedLanes}/${specs.length} lane(s) reached a terminal, engaged verdict${participants ? ` — ${formatParticipantOutcomes(participants)}` : ""}${studyTasks ? `; tasks: ${formatStudyTaskFunnel(studyTasks)}` : ""}.`,
4934
+ : `${args.rerun ? `Rerun from ${args.rerun.sourceRunId}: ` : ""}Computer-use fan-out (${specs.length} per-lane worlds): ${passedLanes}/${specs.length} lane(s) reached a terminal, engaged verdict${participants ? ` — ${formatParticipantOutcomes(participants, participantEndings)}` : ""}${studyTasks ? `; tasks: ${formatStudyTaskFunnel(studyTasks)}` : ""}.`,
4910
4935
  gaps: args.inProgress === true
4911
4936
  ? ["Live fan-out session is still running."]
4912
4937
  : args.dryRun
@@ -4921,7 +4946,7 @@ export function buildCuaFanoutBundle(args) {
4921
4946
  || outcome.session === undefined
4922
4947
  || outcome.session.status !== "passed")
4923
4948
  .map(({ spec, outcome }) => `${spec.laneId}: ${outcome?.skippedReason ?? outcome?.sessionError ?? outcome?.session?.reason ?? "did not pass"}`)
4924
- };
4949
+ }, streams);
4925
4950
  const anyRaw = (outcomes ?? []).some((outcome) => outcome.session?.trace.redaction.screenshots === "raw");
4926
4951
  const ranLive = (outcomes ?? []).some((outcome) => outcome.session !== undefined || outcome.sessionError !== undefined);
4927
4952
  const configuredBrowser = config.execution?.desktop?.browser;
@@ -5105,7 +5130,7 @@ function renderCuaReviewMarkdown(bundle) {
5105
5130
  "",
5106
5131
  `- run: ${bundle.runId}`,
5107
5132
  `- mode: ${bundle.mode}`,
5108
- `- verdict: ${bundle.review.verdict}`,
5133
+ `- run gate: ${bundle.review.verdict}`,
5109
5134
  `- summary: ${bundle.review.summary}`,
5110
5135
  ...(provenance ? [`- subject: ${provenance.message}`] : []),
5111
5136
  ...(trace