humanish 0.89.1 → 0.91.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (99) hide show
  1. package/README.md +20 -1
  2. package/dist/automatic-analysis-completion.d.ts +25 -0
  3. package/dist/automatic-analysis-completion.js +56 -0
  4. package/dist/automatic-analysis-completion.js.map +1 -0
  5. package/dist/automatic-analysis-config.d.ts +25 -0
  6. package/dist/automatic-analysis-config.js +44 -0
  7. package/dist/automatic-analysis-config.js.map +1 -0
  8. package/dist/automatic-study-analysis.d.ts +15 -0
  9. package/dist/automatic-study-analysis.js +187 -0
  10. package/dist/automatic-study-analysis.js.map +1 -0
  11. package/dist/concurrent-shared-world-lab.d.ts +4 -2
  12. package/dist/concurrent-shared-world-lab.js +10 -3
  13. package/dist/concurrent-shared-world-lab.js.map +1 -1
  14. package/dist/cua-actor-lab.d.ts +4 -2
  15. package/dist/cua-actor-lab.js +10 -5
  16. package/dist/cua-actor-lab.js.map +1 -1
  17. package/dist/e2b-terminal-lab.d.ts +4 -2
  18. package/dist/e2b-terminal-lab.js +15 -3
  19. package/dist/e2b-terminal-lab.js.map +1 -1
  20. package/dist/export-bundle.js +2 -0
  21. package/dist/export-bundle.js.map +1 -1
  22. package/dist/index.d.ts +6 -1
  23. package/dist/index.js +2 -0
  24. package/dist/index.js.map +1 -1
  25. package/dist/lab-config.d.ts +5 -0
  26. package/dist/lab-config.js +21 -2
  27. package/dist/lab-config.js.map +1 -1
  28. package/dist/lab-engine.d.ts +2 -0
  29. package/dist/lab-engine.js +12 -3
  30. package/dist/lab-engine.js.map +1 -1
  31. package/dist/lab-preflight.d.ts +3 -0
  32. package/dist/lab-preflight.js +3 -0
  33. package/dist/lab-preflight.js.map +1 -1
  34. package/dist/lab-summary.d.ts +4 -0
  35. package/dist/lab-summary.js +6 -1
  36. package/dist/lab-summary.js.map +1 -1
  37. package/dist/observer-app.html +8 -8
  38. package/dist/observer.d.ts +3 -0
  39. package/dist/observer.js +23 -6
  40. package/dist/observer.js.map +1 -1
  41. package/dist/oss-lab.d.ts +1 -1
  42. package/dist/oss-lab.js.map +1 -1
  43. package/dist/oss-meta-lab.d.ts +1 -1
  44. package/dist/oss-meta-lab.js.map +1 -1
  45. package/dist/program.d.ts +5 -0
  46. package/dist/program.js +49 -8
  47. package/dist/program.js.map +1 -1
  48. package/dist/run-detail.d.ts +2 -0
  49. package/dist/run-detail.js +3 -0
  50. package/dist/run-detail.js.map +1 -1
  51. package/dist/run.d.ts +1 -1
  52. package/dist/run.js.map +1 -1
  53. package/dist/scripted-browser-lab.d.ts +4 -2
  54. package/dist/scripted-browser-lab.js +14 -9
  55. package/dist/scripted-browser-lab.js.map +1 -1
  56. package/dist/shared-world-lab.d.ts +4 -2
  57. package/dist/shared-world-lab.js +10 -3
  58. package/dist/shared-world-lab.js.map +1 -1
  59. package/dist/study-analysis-engine.d.ts +7 -2
  60. package/dist/study-analysis-engine.js +26 -6
  61. package/dist/study-analysis-engine.js.map +1 -1
  62. package/dist/study-analysis-evidence.js +273 -59
  63. package/dist/study-analysis-evidence.js.map +1 -1
  64. package/dist/study-analysis-job.d.ts +88 -0
  65. package/dist/study-analysis-job.js +197 -0
  66. package/dist/study-analysis-job.js.map +1 -0
  67. package/dist/study-analysis-service.d.ts +10 -1
  68. package/dist/study-analysis-service.js +31 -8
  69. package/dist/study-analysis-service.js.map +1 -1
  70. package/dist/study-analysis-sharing.js +13 -5
  71. package/dist/study-analysis-sharing.js.map +1 -1
  72. package/dist/study-analysis-store.d.ts +4 -0
  73. package/dist/study-analysis-store.js +40 -1
  74. package/dist/study-analysis-store.js.map +1 -1
  75. package/dist/study-analysis-validation.d.ts +135 -1
  76. package/dist/study-analysis-validation.js +62 -26
  77. package/dist/study-analysis-validation.js.map +1 -1
  78. package/dist/study-analysis.d.ts +14 -0
  79. package/dist/study-analysis.js.map +1 -1
  80. package/dist/terminal-participant-activity.d.ts +6 -0
  81. package/dist/terminal-participant-activity.js +34 -0
  82. package/dist/terminal-participant-activity.js.map +1 -0
  83. package/dist/tui-actions.d.ts +1 -1
  84. package/dist/tui-actions.js +15 -1
  85. package/dist/tui-actions.js.map +1 -1
  86. package/dist/tui-app.js +116 -116
  87. package/dist/tui-contract.d.ts +3 -2
  88. package/dist/tui-contract.js.map +1 -1
  89. package/docs/architecture/examples/state-driven-local-app/README.md +5 -0
  90. package/docs/architecture/examples/state-driven-local-app/runner.mjs +1 -0
  91. package/docs/contracts/schemas.md +1 -1
  92. package/docs/contracts/study-analysis.md +47 -3
  93. package/docs/goals/current.md +4 -4
  94. package/docs/principles/invariants-and-defaults.md +7 -3
  95. package/docs/product/automatic-analysis.md +81 -0
  96. package/docs/ramp/README.md +11 -2
  97. package/docs/release/0.90.0-automatic-analysis.md +21 -0
  98. package/docs/release/0.91.0-analysis-quality-and-defaults.md +33 -0
  99. package/package.json +1 -1
@@ -51,8 +51,9 @@ export interface TuiCapabilities {
51
51
  openObserver(cwd: string, observerPath: string): Promise<TuiActionResult>;
52
52
  /** Stop the sandboxes an interrupted run left behind, keeping its evidence. */
53
53
  reclaimRun(cwd: string, runId: string): Promise<ReclaimResult>;
54
- /** End a run that is still going. Stops the PROCESS; `reclaimRun` stops what it provisioned. */
55
- stopRun(cwd: string, runId: string): Promise<TuiActionResult>;
54
+ /** End a run that is still going. "analysis" is marker-only regardless of the current status;
55
+ * it MUST NOT probe or signal a process. The default "run" intent stops the participant process. */
56
+ stopRun(cwd: string, runId: string, intent?: "run" | "analysis"): Promise<TuiActionResult>;
56
57
  /**
57
58
  * Set this directory up as a humanish project. The surface's only WRITING action outside of
58
59
  * starting runs — offered because "cd somewhere else and run init" is a dead end shown to
@@ -1 +1 @@
1
- {"version":3,"file":"tui-contract.js","sourceRoot":"","sources":["../src/tui-contract.ts"],"names":[],"mappings":"AAAA,2DAA2D;AAC3D,EAAE;AACF,6FAA6F;AAC7F,mGAAmG;AACnG,+EAA+E;AAC/E,EAAE;AACF,gGAAgG;AAChG,+FAA+F;AAC/F,gGAAgG;AAChG,mGAAmG;AACnG,6DAA6D;AAC7D,EAAE;AACF,mGAAmG;AACnG,mBAAmB;AA8FnB,mFAAmF;AACnF,MAAM,CAAC,MAAM,kBAAkB,GAAG,EAAE,CAAC;AAErC;;;GAGG;AACH,MAAM,UAAU,eAAe,CAAC,gBAAwB,OAAO,CAAC,OAAO;IACrE,MAAM,KAAK,GAAG,MAAM,CAAC,QAAQ,CAAC,aAAa,CAAC,OAAO,CAAC,IAAI,EAAE,EAAE,CAAC,CAAC,KAAK,CAAC,GAAG,CAAC,CAAC,CAAC,CAAC,IAAI,EAAE,EAAE,EAAE,CAAC,CAAC;IACvF,OAAO,MAAM,CAAC,QAAQ,CAAC,KAAK,CAAC,IAAI,KAAK,IAAI,kBAAkB,CAAC;AAC/D,CAAC;AAED;;;GAGG;AACH,MAAM,UAAU,YAAY,CAAC,OAAe;IAC1C,OAAO,IAAI,GAAG,CAAC,cAAc,EAAE,OAAO,CAAC,CAAC;AAC1C,CAAC;AAED;;;;;;;;;;GAUG;AACH,MAAM,UAAU,sBAAsB,CAAC,KAKtC;IACC,IAAI,CAAC,KAAK,CAAC,SAAS,EAAE,CAAC;QACrB,OAAO,+BAA+B,kBAAkB,cAAc,KAAK,CAAC,WAAW,mCAAmC,CAAC;IAC7H,CAAC;IACD,IAAI,CAAC,KAAK,CAAC,aAAa,EAAE,CAAC;QACzB,OAAO,8GAA8G,CAAC;IACxH,CAAC;IACD,OAAO,KAAK,CAAC,WAAW;QACtB,CAAC,CAAC,yEAAyE;QAC3E,CAAC,CAAC,kJAAkJ,CAAC;AACzJ,CAAC"}
1
+ {"version":3,"file":"tui-contract.js","sourceRoot":"","sources":["../src/tui-contract.ts"],"names":[],"mappings":"AAAA,2DAA2D;AAC3D,EAAE;AACF,6FAA6F;AAC7F,mGAAmG;AACnG,+EAA+E;AAC/E,EAAE;AACF,gGAAgG;AAChG,+FAA+F;AAC/F,gGAAgG;AAChG,mGAAmG;AACnG,6DAA6D;AAC7D,EAAE;AACF,mGAAmG;AACnG,mBAAmB;AA+FnB,mFAAmF;AACnF,MAAM,CAAC,MAAM,kBAAkB,GAAG,EAAE,CAAC;AAErC;;;GAGG;AACH,MAAM,UAAU,eAAe,CAAC,gBAAwB,OAAO,CAAC,OAAO;IACrE,MAAM,KAAK,GAAG,MAAM,CAAC,QAAQ,CAAC,aAAa,CAAC,OAAO,CAAC,IAAI,EAAE,EAAE,CAAC,CAAC,KAAK,CAAC,GAAG,CAAC,CAAC,CAAC,CAAC,IAAI,EAAE,EAAE,EAAE,CAAC,CAAC;IACvF,OAAO,MAAM,CAAC,QAAQ,CAAC,KAAK,CAAC,IAAI,KAAK,IAAI,kBAAkB,CAAC;AAC/D,CAAC;AAED;;;GAGG;AACH,MAAM,UAAU,YAAY,CAAC,OAAe;IAC1C,OAAO,IAAI,GAAG,CAAC,cAAc,EAAE,OAAO,CAAC,CAAC;AAC1C,CAAC;AAED;;;;;;;;;;GAUG;AACH,MAAM,UAAU,sBAAsB,CAAC,KAKtC;IACC,IAAI,CAAC,KAAK,CAAC,SAAS,EAAE,CAAC;QACrB,OAAO,+BAA+B,kBAAkB,cAAc,KAAK,CAAC,WAAW,mCAAmC,CAAC;IAC7H,CAAC;IACD,IAAI,CAAC,KAAK,CAAC,aAAa,EAAE,CAAC;QACzB,OAAO,8GAA8G,CAAC;IACxH,CAAC;IACD,OAAO,KAAK,CAAC,WAAW;QACtB,CAAC,CAAC,yEAAyE;QAC3E,CAAC,CAAC,kJAAkJ,CAAC;AACzJ,CAAC"}
@@ -73,3 +73,8 @@ npx tsc --allowJs --checkJs --noEmit --strict --skipLibCheck --types node --targ
73
73
  See the [state-driven executor guide](../../state-driven-executor.md) for progress
74
74
  projection limits, runtime-only state, unpinned local-app provenance, and the
75
75
  fail-closed guards on this route.
76
+
77
+ The example explicitly sets `review: { analysis: false }` to keep post-run
78
+ analysis free of provider requests too. Supported live recordings otherwise
79
+ use the separate default analysis budget; a deterministic participant provider
80
+ does not replace the analysis provider.
@@ -68,6 +68,7 @@ try {
68
68
  // This registry id selects the CUA loop. buildProvider supplies the actual provider.
69
69
  actors: [{ type: "openai-computer-use", persona: "pixel-pat", mission: "Greet the app." }],
70
70
  scenario: { mode: "live" },
71
+ review: { analysis: false }, // Keep this deterministic example free of provider requests.
71
72
  execution: { timeoutMs: 15_000 }
72
73
  });
73
74
  if (!parsed.ok) throw new Error(parsed.error.message);
@@ -3,7 +3,7 @@
3
3
  Date: 2026-06-02 (current-state note updated 2026-07-14)
4
4
 
5
5
  Status: reference map for the major contracts shipped through source version
6
- `0.89.1`; it is not an exhaustive inventory of command/result envelopes. Exported types,
6
+ `0.91.0`; it is not an exhaustive inventory of command/result envelopes. Exported types,
7
7
  schema constants, parsers, and validators in `src/` are authoritative. Rows
8
8
  marked "reserved" name layering intent only — no code emits or validates them
9
9
  yet. Do not emit a reserved schema.
@@ -1,8 +1,11 @@
1
1
  # Study analysis
2
2
 
3
- Study analysis is an optional interpretation of retained participant evidence.
3
+ Study analysis is an independent interpretation of retained participant evidence.
4
4
  It is separate from the participant's account, recorded outcome, and the run's
5
5
  deterministic review verdict. Opening an Observer never starts a provider request.
6
+ Supported live runs request analysis on completion by default, with a separate
7
+ $3 admission estimate limit. Set `review.analysis: false` to disable that request;
8
+ see [automatic analysis](../product/automatic-analysis.md).
6
9
 
7
10
  ## Invocation
8
11
 
@@ -59,18 +62,50 @@ stopped, an operator can remove the empty lock directory and retry.
59
62
 
60
63
  The packet currently admits up to 16 participants, 800 evidence items, 40 PNG
61
64
  captures, 160 KiB of text and 20 MiB of images. Individual source files, image
62
- dimensions and result sizes have separate limits. Selection follows retained
63
- source order; it is not a statistically representative sample. Selection
65
+ dimensions and result sizes have separate limits. Count and text budgets are
66
+ distributed across included participants, with unused capacity from short
67
+ sessions available to longer ones. Capture selection prioritizes session endings
68
+ and beginnings, context around recorded failures, and spread across each whole
69
+ session. Failure priority uses structured source status, not application-specific
70
+ keywords or image interpretation. Unflagged visual errors may still be omitted.
71
+ Bounded reads and the total image byte limit can reduce coverage further.
72
+ Selected entries retain their original source order, frame and event identities;
73
+ this is not a statistically representative sample. Selection
64
74
  omissions and unreadable or invalid capture files make declared coverage
65
75
  incomplete. Coverage records file availability and selection; it does not
66
76
  certify visual legibility, correct interpretation, or exhaustive issue discovery.
67
77
 
78
+ New packets declare `captureVersion: 2`, bound into their input digest. They
79
+ include captures attached to `screenshot` and scripted `ui_action` events,
80
+ preserving the original action IDs and evidence basis. Error notices that refer
81
+ to an earlier capture retain that context without creating another frame.
82
+ Scripted lanes use their recorded `ui.intent` goal when no participant assignment
83
+ exists. Missing assignments and declared captures without supported trace
84
+ references are explicit omissions. Artifacts without `captureVersion` continue
85
+ to validate against the original capture mapping; previously saved selections
86
+ are not recomputed or rewritten.
87
+
68
88
  The standard review covers session summary, apparent intent, observed outcome,
69
89
  friction, dead ends, recovery, and participant feedback. Findings are ordered by
70
90
  observed task impact, replication among exposed participants, and recovery.
71
91
  Impact and confidence remain separate. There is no numeric frustration or
72
92
  universal priority score.
73
93
 
94
+ The standard review also accounts for material participant concerns before
95
+ ranking findings. Reported uncertainty can be useful even when the interface is
96
+ correct or study setup may explain it. Claims distinguish that experience from
97
+ an established product defect, preserve consequential recoveries, and check
98
+ participant accounts against the actual assignment and captured state.
99
+
100
+ New results include `concernReviews`: evidence-linked observations with a
101
+ `finding`, `context`, or `unsupported` disposition and a concise reason. A finding
102
+ disposition references an existing local finding ID; other dispositions use null.
103
+ The Observer's **Concerns considered** disclosure exposes these decisions and
104
+ their original evidence. There is no required finding count or inventory of every
105
+ thought. Older reports may omit this field and remain readable. The disclosure
106
+ is model-generated assessment, not an independent completeness audit or a human
107
+ reviewer annotation.
108
+
74
109
  Every observation cites packet-local evidence IDs. The model cannot choose a
75
110
  filesystem path or fetch another resource. Validation checks participant
76
111
  membership, unique counts, quote fidelity, evidence type and reference
@@ -80,6 +115,8 @@ is correct or every consequential issue was found.
80
115
 
81
116
  Elapsed replay time starts at the first retained capture. It is not a video
82
117
  offset. Nonvisual events retain event identity without invented frame offsets.
118
+ Scripted captures without recorded timestamps keep null analysis times; any
119
+ uniform playback pacing is an estimate, not an observed duration.
83
120
 
84
121
  ## Durable records
85
122
 
@@ -94,8 +131,15 @@ status, usage and validated findings:
94
131
  analysis/<analysis>/corrections/<correction>/correction.json
95
132
  analysis-attempts/<analysis>/receipt.json
96
133
  observer/study-analysis.json
134
+ analysis-automatic/job.json # post-run lifecycle; never a retry instruction
97
135
  ```
98
136
 
137
+ The optional automatic job is separate from the immutable analysis. Its view
138
+ binds terminal state to the exact execution receipt and report. A stale or
139
+ unverifiable job remains unknown; reading or exporting it never dispatches.
140
+ Automatic job metadata is omitted from shared derivatives. See
141
+ [automatic analysis](../product/automatic-analysis.md).
142
+
99
143
  Version and correction directories are claimed exclusively; publication is
100
144
  atomic. Source evidence is not rewritten. Minimal execution receipts retain
101
145
  model, budget, status and known usage even if source changes prevent report
@@ -1,9 +1,9 @@
1
1
  # Current Goals
2
2
 
3
- Status date: 2026-09-15. Release baseline: `0.89.1`.
3
+ Status date: 2026-09-15. Release baseline: `0.91.0`.
4
4
 
5
5
  This page guides work on current merged source. Published behavior is described
6
- in the [release notes](../release/0.89.1-analysis-finished-notice.md).
6
+ in the [release notes](../release/0.91.0-analysis-quality-and-defaults.md).
7
7
  The [September 9 history](https://github.com/danielgwilson/humanish/blob/main/docs/goals/current-history-2026-09-09.md)
8
8
  preserves the former status log; its queues do not supersede this page.
9
9
 
@@ -88,7 +88,7 @@ requires decision-equivalent retained evidence and a real deletion branch.
88
88
  No first-party deletion branch has met that gate. Public demonstrations do not
89
89
  substitute for it.
90
90
 
91
- ## Current Program Truth (source `0.89.1`)
91
+ ## Current Program Truth (source `0.91.0`)
92
92
 
93
93
  | Surface | Available in merged source | Remaining boundary |
94
94
  | --- | --- | --- |
@@ -99,7 +99,7 @@ substitute for it.
99
99
  | Shared state | Sequential and concurrent single-origin shared-world studies with retained evidence | Multi-origin implementation remains gated; concurrent state change does not establish per-action causation |
100
100
  | Observer | Live/recorded views, participant assignments, action-specific links, saved moments, zoom, comparison and phone-width review | Sparse captures cannot prove every action's effect; visual comparison alone is not a controlled experiment |
101
101
  | Review and feedback | Verification grades, feedback drafts, portable HTML, redacted bundle derivatives and computer-use completion-source labels | Sharing requires the appropriate grade; participant reports and condition matches still need task adjudication |
102
- | Study findings | Explicit `analyze`, bounded evidence selection, versioned findings, exact source links and append-only corrections within the Observer study shell | Model interpretation needs review; bounded selection and source truncation limit coverage; opening Observer never dispatches analysis |
102
+ | Study findings | Default post-run analysis on supported live routes with a separate disclosed $3 admission estimate limit and opt-out; explicit `analyze`, fairer evidence selection, concern review and versioned findings with exact source links | Model interpretation needs review; bounded selection and source truncation limit coverage; opening Observer never dispatches analysis |
103
103
  | TUI and serving | Detached starts, run stopping, reclamation, Observer attachment, loopback serving and run library | Stopping a process does not itself prove sandbox cleanup; TUI views over CLI `stats`/`export` remain follow-ups |
104
104
  | Off-app communication | In-sandbox email/SMS catch and digest-only thread evidence | This does not establish real-provider delivery |
105
105
  | Mobile and media | Hosted viewport/emulation, desktop geometry checks, bounded dwell and declared camera feed | Physical-device and touch fidelity remain unproven; unsupported microphone declarations are rejected |
@@ -34,9 +34,12 @@ certify (see the conformance suite).
34
34
  terminal) is only ever pointed at a URL the harness itself issued or validated under a
35
35
  declared policy (loopback entry, provisioned subject, declared external target). Never an
36
36
  arbitrary URL from unvalidated input.
37
- 3. **Live spend is explicit.** No configuration default, omission, or fallback may cause
38
- provider or sandbox spend. Spend requires an affirmative declaration (`scenario.mode:
39
- live`, an env opt-in gate for spend-bearing tests).
37
+ 3. **Live spend requires an explicit live invocation.** No omission or fallback may
38
+ turn a dry run, preview, reader or unsupported route into provider or sandbox
39
+ spend. An explicitly live supported study includes the disclosed default
40
+ post-run analysis budget unless `review.analysis: false` disables it. Analysis
41
+ has a separate admission estimate limit, not a provider billing cap or part of
42
+ the actor budget. Spend-bearing tests retain their explicit env gates.
40
43
  4. **Evidence verifies fail-closed.** A run bundle that cannot pass verification (schema,
41
44
  redaction status, artifact presence, public-safety scan) is a failed run, even when the
42
45
  session "worked." The gate applies to the harness's own error reports.
@@ -82,6 +85,7 @@ silently drifting from one is not.
82
85
  | Default | Why it is the default | Legitimate override |
83
86
  |---|---|---|
84
87
  | Dry-run | Spend safety (invariant 3 sets the floor; dry-run keeps the floor far away) | `scenario.mode: live` |
88
+ | Post-run analysis on supported live studies | Findings accompany the recording; a separate $3 admission estimate limit is visible before execution | `review.analysis: false` disables it; an explicit mapping sets another analysis budget. Dry-run and unsupported routes never dispatch; missing default credentials produce a recorded skip |
85
89
  | Per-lane worlds | Isolation, attribution, reproducibility | `subject.topology: shared-world` — N seats against ONE provisioned, mutable plane for scenarios that ARE about interaction between roles (#164). `execution.concurrency: 1` (an explicit choice) = SEQUENTIAL turns (one sandbox); higher = CONCURRENT — and since #350 an omitted concurrency fills to the seat count, so every declared seat runs live at once by default (one getHost-exposed subject sandbox + N actor sandboxes driving it at once, synthetic-subject only). The bundle declares the weaker `attributionClass: shared-world` + a verify-enforced `attributionLimits` ceiling (the concurrent set drops `sequential-only` and adds `best-effort-causal-attribution` etc.), so the looser per-role attribution is honest, not hidden. |
86
90
  | External key placement | Smallest blast radius: when the keyed process (e.g. a computer-use provider loop) runs outside the sandbox, its key never enters | In-sandbox placement when the keyed process runs inside (an agent harness under test); declared per actor type, with a spend budget |
87
91
  | Loopback entry URLs | Public-safety: never drive third-party sites unbidden | `policies.allowPublicTargets` for an owner-declared deployment/preview (a Vercel preview of your own app). Multi-lane public/preview fan-out needs explicit `actors[0].lanes[].target` for every lane, so the adapter-owned topology is declared rather than inferred. Provisioned clone subjects always serve in-sandbox on loopback |
@@ -0,0 +1,81 @@
1
+ # Automatic study analysis
2
+
3
+ Supported live studies automatically request analysis after each recording finishes. Findings remain
4
+ separate from participant feedback and the recorded study verdict.
5
+
6
+ The default is `gpt-6-astra` with high reasoning effort, a separate $3 admission
7
+ estimate limit, a 300-second timeout and 16,384 output tokens. To customize it:
8
+
9
+ ```yaml
10
+ review:
11
+ analysis:
12
+ maxCostUsd: 3
13
+ # Optional; these values match manual analysis defaults.
14
+ model: gpt-6-astra
15
+ timeoutMs: 300000
16
+ maxOutputTokens: 16384
17
+ # question: Where did participants need to recover?
18
+ ```
19
+
20
+ Omitting `review.analysis` uses these defaults. Set `review.analysis: false` to
21
+ run participants without the additional analysis request. An explicit analysis
22
+ mapping requires `maxCostUsd`. This limits an admission estimate, not the
23
+ provider's final bill, and is separate from participant spending limits. Analysis
24
+ sends selected retained text and captures to OpenAI using `OPENAI_API_KEY`.
25
+ Analysis runs in the Humanish runner using its credentials. This setting adds no
26
+ credential channel to the target application; each participant backend retains
27
+ its existing authentication boundary. Review the separate analysis budget before running a manifest live; an actor's
28
+ zero-dollar cap does not cap post-run analysis. The bundled first-contact
29
+ zero-spend product fixture explicitly disables analysis.
30
+
31
+ The same configuration works through `humanish run <lab>`, `lab run <lab>`,
32
+ `watch <lab>`, and TUI live starts. Direct library calls to the five recording
33
+ producers honor it too. Supported routes are computer-use, scripted-browser,
34
+ terminal-product, sequential shared-world and concurrent shared-world. Synthetic,
35
+ smoke and meta routes never enable analysis by default and reject an explicit
36
+ analysis mapping before execution. `false` is accepted on every route. Dry runs show analysis
37
+ as skipped, without reading analysis credentials or making a provider request.
38
+
39
+ Participant execution finishes and its recording is finalized before analysis is
40
+ queued. A participant who was blocked or interrupted can still have useful
41
+ retained evidence; analysis requires a verified live recording, not a successful
42
+ participant outcome. An active, missing or invalid recording is not analyzed.
43
+ Default analysis also skips recordings containing only setup or failure records
44
+ with no retained participant activity. A desktop startup failure does not start
45
+ an analysis request. The original failure remains visible.
46
+
47
+ CLI live starts disclose the separate admission estimate limit before execution.
48
+ `humanish lab preflight <lab> --json` and the TUI lab screen also expose the
49
+ resolved budget without dispatching analysis. Library callers can inspect
50
+ `resolveAutomaticAnalysis` or `automaticAnalysisBudget` before running.
51
+
52
+ The command waits for analysis and reports its separate state. A TUI-launched
53
+ runner continues after the TUI closes; reopening the TUI or Observer reads the
54
+ existing job and does not start another request. Concurrent or repeated automatic
55
+ invocations cannot silently retry a paid attempt. If a process disappears while
56
+ an attempt is in flight, its state can be unknown rather than falsely complete.
57
+ Use manual `humanish analyze --run <exact-run-id> --max-cost 3` for an intentional
58
+ follow-up after inspecting the existing attempt and its accounting.
59
+
60
+ Stopping participant execution does not start a fresh automatic analysis. A
61
+ recorded harness cancellation is skipped; ordinary time limits and participant
62
+ abandonment remain eligible evidence.
63
+
64
+ During analysis, Ctrl-C asks the request to cancel. The TUI's **Cancel analysis**
65
+ action writes a cancellation request for that recording; it does not signal the
66
+ finished participant process. Cancelling cannot undo provider work already
67
+ accepted. Known usage is retained; missing usage remains unknown.
68
+
69
+ The CLI's JSON keeps `runOk` for the original backend result, `automaticAnalysis`
70
+ for post-run analysis, and `ok` for the overall request. Failed, cancelled or
71
+ unknown analysis produces exit code 2 without discarding the recording. A missing
72
+ `OPENAI_API_KEY` skips default analysis and preserves a successful run exit;
73
+ `automaticAnalysisTrigger: "default"` distinguishes that case in JSON. A missing
74
+ key for an explicitly configured analysis remains a failed overall request.
75
+ Either skip is retained for review and does not retry automatically. Partial
76
+ findings remain visibly partial; a valid partial result can succeed, while a
77
+ partial result with an analysis error still fails the command. Recorded task
78
+ outcomes and the deterministic review verdict are never rewritten by analysis.
79
+
80
+ See the [analysis contract](../contracts/study-analysis.md) for selection limits,
81
+ evidence validation, actual usage, corrections and share-safe export behavior.
@@ -2,7 +2,7 @@
2
2
 
3
3
  Status: public-safe contributor and agent ramp.
4
4
 
5
- Package/source version in this tree: `0.89.1` (2026-09-15). The Observer is phone-usable as a stated requirement (observer/AGENTS.md); interactive primitives start from Base UI. The Observer renderer is the observer/ workspace artifact only; the legacy string-concat renderer was deleted at cutover (#426), and rollback is a version pin to 0.42.0. The containment boundary introduced in
5
+ Package/source version in this tree: `0.91.0` (2026-09-15). The Observer is phone-usable as a stated requirement (observer/AGENTS.md); interactive primitives start from Base UI. The Observer renderer is the observer/ workspace artifact only; the legacy string-concat renderer was deleted at cutover (#426), and rollback is a version pin to 0.42.0. The containment boundary introduced in
6
6
  `0.15.1` remains in force: managed run and output paths bind to validated
7
7
  physical filesystem identities, and stored provider IDs are evidence, not
8
8
  cleanup authority. The bundled OSS meta-lab is dry-run only until
@@ -47,6 +47,15 @@ If a change does not improve one of those loops, it probably belongs elsewhere.
47
47
 
48
48
  ## Current State
49
49
 
50
+ The [0.91.0 release note](../release/0.91.0-analysis-quality-and-defaults.md)
51
+ describes automatic analysis by default on supported live recordings, its
52
+ separate disclosed budget and opt-out, fairer evidence selection, and
53
+ evidence-linked review of material concerns and exclusions.
54
+
55
+ The [0.90.0 release note](../release/0.90.0-automatic-analysis.md) describes
56
+ opt-in analysis after live runs, truthful job states, cancellation, and scripted
57
+ captures and assignments in findings and playback.
58
+
50
59
  The [0.89.1 release note](../release/0.89.1-analysis-finished-notice.md)
51
60
  clarifies that an analysis with limitations has finished.
52
61
 
@@ -95,7 +104,7 @@ Implemented:
95
104
 
96
105
  - `commander` CLI with stable command help;
97
106
  - `init`, `doctor`, `run`, `watch`, `verify`, `review`, `runs`, `analyze`, and `feedback`;
98
- - opt-in study analysis with bounded provider admission, immutable findings,
107
+ - study analysis with bounded provider admission, a live-run default and opt-out, immutable findings,
99
108
  source-bound corrections and evidence-linked Observer review;
100
109
  - synthetic run bundles;
101
110
  - public-safety verification with machine-readable `shareSafety.status`
@@ -0,0 +1,21 @@
1
+ # 0.90.0 — Findings after live studies
2
+
3
+ Labs can opt into independent analysis after their recording finishes with
4
+ `review.analysis.maxCostUsd`. The command waits for the result; the TUI and
5
+ Observer show analysis separately from participant execution and task outcomes.
6
+ Without the setting, run behavior stays unchanged.
7
+
8
+ The five supported live recording producers share the same completion boundary.
9
+ Automatic analysis claims one attempt per recording, preserves usage and prior
10
+ findings, and never starts again merely because a view is reopened. Cancellation
11
+ requests target analysis without signaling the finished participant process.
12
+ Recording identity remains pinned through dispatch, reuse and artifact writes.
13
+
14
+ Scripted browser recordings now contribute their action-attached screenshots and
15
+ declared goal to analysis and playback. Earlier saved analyses retain their
16
+ original evidence mapping. Referenced error captures do not become duplicate
17
+ frames. Missing assignments or unmapped captures remain visible limitations.
18
+
19
+ See [automatic analysis](../product/automatic-analysis.md) for configuration,
20
+ route support, spending and cancellation behavior. Analysis uses an admission
21
+ estimate separate from participant caps; it is not a provider billing hard cap.
@@ -0,0 +1,33 @@
1
+ # Humanish 0.91.0
2
+
3
+ Supported live studies now request analysis when the recording finishes, using
4
+ `gpt-6-astra` with high reasoning and a separate $3 admission estimate limit.
5
+ The CLI, preflight and TUI disclose that budget. Set `review.analysis: false` to
6
+ disable the extra request, or supply an analysis mapping with `maxCostUsd` to
7
+ customize it. The estimate is additional to participant and desktop costs and
8
+ is not a provider billing cap. Missing default credentials record a skip while
9
+ preserving a successful recording; explicit analysis failures remain failures.
10
+ Startup failures with no retained participant activity skip default analysis.
11
+
12
+ Analysis now reviews material participant concerns before ranking findings.
13
+ Useful reported uncertainty and recovered mistakes can remain findings even
14
+ when product fault is unproven. **Concerns considered** shows evidence-linked
15
+ decisions, including why material concerns were left out of the ranked list.
16
+ It is model assessment, separate from original participant feedback and human
17
+ reviewer annotations.
18
+
19
+ Evidence selection shares count and text budgets across participants and samples
20
+ their whole sessions, prioritizing beginnings, endings and context around
21
+ recorded failures. Original frame and event identities remain intact. Missing
22
+ assignments remain explicit; historical assignments are never invented.
23
+ Existing analysis versions remain readable and unchanged.
24
+
25
+ Selection is bounded and can still omit important moments. Concern review does
26
+ not certify exhaustive discovery or accurate causal diagnosis. Opening Observer,
27
+ exporting a recording and reading history never start a model request. Dry-runs
28
+ and unsupported default routes remain request-free, and examples advertised as
29
+ zero-spend explicitly disable automatic analysis.
30
+
31
+ See the [analysis contract](../contracts/study-analysis.md),
32
+ [automatic analysis guide](../product/automatic-analysis.md), and
33
+ [bounded evaluation receipt](https://github.com/danielgwilson/humanish/blob/main/docs/goals/computer-use-actor/receipts/analysis-quality-and-defaults-2026-09-15.md).
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "humanish",
3
- "version": "0.89.1",
3
+ "version": "0.91.0",
4
4
  "description": "Open-source-safe CLI for persona simulation, observer review, and public-safe feedback drafts.",
5
5
  "author": "Daniel G Wilson <daniel@danielgwilson.com>",
6
6
  "keywords": [