humanish 0.89.0 → 0.90.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +15 -0
- package/dist/automatic-analysis-completion.d.ts +25 -0
- package/dist/automatic-analysis-completion.js +54 -0
- package/dist/automatic-analysis-completion.js.map +1 -0
- package/dist/automatic-analysis-config.d.ts +17 -0
- package/dist/automatic-analysis-config.js +31 -0
- package/dist/automatic-analysis-config.js.map +1 -0
- package/dist/automatic-study-analysis.d.ts +12 -0
- package/dist/automatic-study-analysis.js +168 -0
- package/dist/automatic-study-analysis.js.map +1 -0
- package/dist/concurrent-shared-world-lab.d.ts +4 -2
- package/dist/concurrent-shared-world-lab.js +10 -3
- package/dist/concurrent-shared-world-lab.js.map +1 -1
- package/dist/cua-actor-lab.d.ts +4 -2
- package/dist/cua-actor-lab.js +10 -5
- package/dist/cua-actor-lab.js.map +1 -1
- package/dist/e2b-terminal-lab.d.ts +4 -2
- package/dist/e2b-terminal-lab.js +10 -3
- package/dist/e2b-terminal-lab.js.map +1 -1
- package/dist/export-bundle.js +2 -0
- package/dist/export-bundle.js.map +1 -1
- package/dist/index.d.ts +4 -0
- package/dist/index.js +1 -0
- package/dist/index.js.map +1 -1
- package/dist/lab-config.d.ts +5 -0
- package/dist/lab-config.js +21 -2
- package/dist/lab-config.js.map +1 -1
- package/dist/lab-engine.d.ts +2 -0
- package/dist/lab-engine.js +12 -3
- package/dist/lab-engine.js.map +1 -1
- package/dist/lab-summary.d.ts +4 -0
- package/dist/lab-summary.js +3 -0
- package/dist/lab-summary.js.map +1 -1
- package/dist/observer-app.html +9 -9
- package/dist/observer.d.ts +3 -0
- package/dist/observer.js +23 -6
- package/dist/observer.js.map +1 -1
- package/dist/oss-lab.d.ts +1 -1
- package/dist/oss-lab.js.map +1 -1
- package/dist/oss-meta-lab.d.ts +1 -1
- package/dist/oss-meta-lab.js.map +1 -1
- package/dist/program.d.ts +5 -0
- package/dist/program.js +42 -7
- package/dist/program.js.map +1 -1
- package/dist/run-detail.d.ts +2 -0
- package/dist/run-detail.js +3 -0
- package/dist/run-detail.js.map +1 -1
- package/dist/run.d.ts +1 -1
- package/dist/run.js.map +1 -1
- package/dist/scripted-browser-lab.d.ts +4 -2
- package/dist/scripted-browser-lab.js +10 -5
- package/dist/scripted-browser-lab.js.map +1 -1
- package/dist/shared-world-lab.d.ts +4 -2
- package/dist/shared-world-lab.js +10 -3
- package/dist/shared-world-lab.js.map +1 -1
- package/dist/study-analysis-engine.d.ts +6 -1
- package/dist/study-analysis-engine.js +17 -3
- package/dist/study-analysis-engine.js.map +1 -1
- package/dist/study-analysis-evidence.js +45 -17
- package/dist/study-analysis-evidence.js.map +1 -1
- package/dist/study-analysis-job.d.ts +87 -0
- package/dist/study-analysis-job.js +197 -0
- package/dist/study-analysis-job.js.map +1 -0
- package/dist/study-analysis-service.d.ts +10 -1
- package/dist/study-analysis-service.js +31 -8
- package/dist/study-analysis-service.js.map +1 -1
- package/dist/study-analysis-sharing.js +13 -5
- package/dist/study-analysis-sharing.js.map +1 -1
- package/dist/study-analysis-store.d.ts +4 -0
- package/dist/study-analysis-store.js +40 -1
- package/dist/study-analysis-store.js.map +1 -1
- package/dist/study-analysis-validation.d.ts +2 -1
- package/dist/study-analysis-validation.js +12 -5
- package/dist/study-analysis-validation.js.map +1 -1
- package/dist/study-analysis.d.ts +6 -0
- package/dist/study-analysis.js.map +1 -1
- package/dist/tui-actions.d.ts +1 -1
- package/dist/tui-actions.js +15 -1
- package/dist/tui-actions.js.map +1 -1
- package/dist/tui-app.js +116 -116
- package/dist/tui-contract.d.ts +3 -2
- package/dist/tui-contract.js.map +1 -1
- package/docs/contracts/schemas.md +1 -1
- package/docs/contracts/study-analysis.md +18 -0
- package/docs/goals/current.md +4 -4
- package/docs/product/automatic-analysis.md +63 -0
- package/docs/ramp/README.md +8 -1
- package/docs/release/0.89.1-analysis-finished-notice.md +24 -0
- package/docs/release/0.90.0-automatic-analysis.md +21 -0
- package/package.json +1 -1
package/dist/tui-contract.d.ts
CHANGED
|
@@ -51,8 +51,9 @@ export interface TuiCapabilities {
|
|
|
51
51
|
openObserver(cwd: string, observerPath: string): Promise<TuiActionResult>;
|
|
52
52
|
/** Stop the sandboxes an interrupted run left behind, keeping its evidence. */
|
|
53
53
|
reclaimRun(cwd: string, runId: string): Promise<ReclaimResult>;
|
|
54
|
-
/** End a run that is still going.
|
|
55
|
-
|
|
54
|
+
/** End a run that is still going. "analysis" is marker-only regardless of the current status;
|
|
55
|
+
* it MUST NOT probe or signal a process. The default "run" intent stops the participant process. */
|
|
56
|
+
stopRun(cwd: string, runId: string, intent?: "run" | "analysis"): Promise<TuiActionResult>;
|
|
56
57
|
/**
|
|
57
58
|
* Set this directory up as a humanish project. The surface's only WRITING action outside of
|
|
58
59
|
* starting runs — offered because "cd somewhere else and run init" is a dead end shown to
|
package/dist/tui-contract.js.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"tui-contract.js","sourceRoot":"","sources":["../src/tui-contract.ts"],"names":[],"mappings":"AAAA,2DAA2D;AAC3D,EAAE;AACF,6FAA6F;AAC7F,mGAAmG;AACnG,+EAA+E;AAC/E,EAAE;AACF,gGAAgG;AAChG,+FAA+F;AAC/F,gGAAgG;AAChG,mGAAmG;AACnG,6DAA6D;AAC7D,EAAE;AACF,mGAAmG;AACnG,mBAAmB;
|
|
1
|
+
{"version":3,"file":"tui-contract.js","sourceRoot":"","sources":["../src/tui-contract.ts"],"names":[],"mappings":"AAAA,2DAA2D;AAC3D,EAAE;AACF,6FAA6F;AAC7F,mGAAmG;AACnG,+EAA+E;AAC/E,EAAE;AACF,gGAAgG;AAChG,+FAA+F;AAC/F,gGAAgG;AAChG,mGAAmG;AACnG,6DAA6D;AAC7D,EAAE;AACF,mGAAmG;AACnG,mBAAmB;AA+FnB,mFAAmF;AACnF,MAAM,CAAC,MAAM,kBAAkB,GAAG,EAAE,CAAC;AAErC;;;GAGG;AACH,MAAM,UAAU,eAAe,CAAC,gBAAwB,OAAO,CAAC,OAAO;IACrE,MAAM,KAAK,GAAG,MAAM,CAAC,QAAQ,CAAC,aAAa,CAAC,OAAO,CAAC,IAAI,EAAE,EAAE,CAAC,CAAC,KAAK,CAAC,GAAG,CAAC,CAAC,CAAC,CAAC,IAAI,EAAE,EAAE,EAAE,CAAC,CAAC;IACvF,OAAO,MAAM,CAAC,QAAQ,CAAC,KAAK,CAAC,IAAI,KAAK,IAAI,kBAAkB,CAAC;AAC/D,CAAC;AAED;;;GAGG;AACH,MAAM,UAAU,YAAY,CAAC,OAAe;IAC1C,OAAO,IAAI,GAAG,CAAC,cAAc,EAAE,OAAO,CAAC,CAAC;AAC1C,CAAC;AAED;;;;;;;;;;GAUG;AACH,MAAM,UAAU,sBAAsB,CAAC,KAKtC;IACC,IAAI,CAAC,KAAK,CAAC,SAAS,EAAE,CAAC;QACrB,OAAO,+BAA+B,kBAAkB,cAAc,KAAK,CAAC,WAAW,mCAAmC,CAAC;IAC7H,CAAC;IACD,IAAI,CAAC,KAAK,CAAC,aAAa,EAAE,CAAC;QACzB,OAAO,8GAA8G,CAAC;IACxH,CAAC;IACD,OAAO,KAAK,CAAC,WAAW;QACtB,CAAC,CAAC,yEAAyE;QAC3E,CAAC,CAAC,kJAAkJ,CAAC;AACzJ,CAAC"}
|
|
@@ -3,7 +3,7 @@
|
|
|
3
3
|
Date: 2026-06-02 (current-state note updated 2026-07-14)
|
|
4
4
|
|
|
5
5
|
Status: reference map for the major contracts shipped through source version
|
|
6
|
-
`0.
|
|
6
|
+
`0.90.0`; it is not an exhaustive inventory of command/result envelopes. Exported types,
|
|
7
7
|
schema constants, parsers, and validators in `src/` are authoritative. Rows
|
|
8
8
|
marked "reserved" name layering intent only — no code emits or validates them
|
|
9
9
|
yet. Do not emit a reserved schema.
|
|
@@ -65,6 +65,15 @@ omissions and unreadable or invalid capture files make declared coverage
|
|
|
65
65
|
incomplete. Coverage records file availability and selection; it does not
|
|
66
66
|
certify visual legibility, correct interpretation, or exhaustive issue discovery.
|
|
67
67
|
|
|
68
|
+
New packets declare `captureVersion: 2`, bound into their input digest. They
|
|
69
|
+
include captures attached to `screenshot` and scripted `ui_action` events,
|
|
70
|
+
preserving the original action IDs and evidence basis. Error notices that refer
|
|
71
|
+
to an earlier capture retain that context without creating another frame.
|
|
72
|
+
Scripted lanes use their recorded `ui.intent` goal when no participant assignment
|
|
73
|
+
exists. Missing assignments and declared captures without supported trace
|
|
74
|
+
references are explicit omissions. Artifacts without `captureVersion` continue
|
|
75
|
+
to validate against the original selection rules.
|
|
76
|
+
|
|
68
77
|
The standard review covers session summary, apparent intent, observed outcome,
|
|
69
78
|
friction, dead ends, recovery, and participant feedback. Findings are ordered by
|
|
70
79
|
observed task impact, replication among exposed participants, and recovery.
|
|
@@ -80,6 +89,8 @@ is correct or every consequential issue was found.
|
|
|
80
89
|
|
|
81
90
|
Elapsed replay time starts at the first retained capture. It is not a video
|
|
82
91
|
offset. Nonvisual events retain event identity without invented frame offsets.
|
|
92
|
+
Scripted captures without recorded timestamps keep null analysis times; any
|
|
93
|
+
uniform playback pacing is an estimate, not an observed duration.
|
|
83
94
|
|
|
84
95
|
## Durable records
|
|
85
96
|
|
|
@@ -94,8 +105,15 @@ status, usage and validated findings:
|
|
|
94
105
|
analysis/<analysis>/corrections/<correction>/correction.json
|
|
95
106
|
analysis-attempts/<analysis>/receipt.json
|
|
96
107
|
observer/study-analysis.json
|
|
108
|
+
analysis-automatic/job.json # opt-in run lifecycle; never a retry instruction
|
|
97
109
|
```
|
|
98
110
|
|
|
111
|
+
The optional automatic job is separate from the immutable analysis. Its view
|
|
112
|
+
binds terminal state to the exact execution receipt and report. A stale or
|
|
113
|
+
unverifiable job remains unknown; reading or exporting it never dispatches.
|
|
114
|
+
Automatic job metadata is omitted from shared derivatives. See
|
|
115
|
+
[automatic analysis](../product/automatic-analysis.md).
|
|
116
|
+
|
|
99
117
|
Version and correction directories are claimed exclusively; publication is
|
|
100
118
|
atomic. Source evidence is not rewritten. Minimal execution receipts retain
|
|
101
119
|
model, budget, status and known usage even if source changes prevent report
|
package/docs/goals/current.md
CHANGED
|
@@ -1,9 +1,9 @@
|
|
|
1
1
|
# Current Goals
|
|
2
2
|
|
|
3
|
-
Status date: 2026-09-
|
|
3
|
+
Status date: 2026-09-15. Release baseline: `0.90.0`.
|
|
4
4
|
|
|
5
5
|
This page guides work on current merged source. Published behavior is described
|
|
6
|
-
in the [release notes](../release/0.
|
|
6
|
+
in the [release notes](../release/0.90.0-automatic-analysis.md).
|
|
7
7
|
The [September 9 history](https://github.com/danielgwilson/humanish/blob/main/docs/goals/current-history-2026-09-09.md)
|
|
8
8
|
preserves the former status log; its queues do not supersede this page.
|
|
9
9
|
|
|
@@ -88,7 +88,7 @@ requires decision-equivalent retained evidence and a real deletion branch.
|
|
|
88
88
|
No first-party deletion branch has met that gate. Public demonstrations do not
|
|
89
89
|
substitute for it.
|
|
90
90
|
|
|
91
|
-
## Current Program Truth (source `0.
|
|
91
|
+
## Current Program Truth (source `0.90.0`)
|
|
92
92
|
|
|
93
93
|
| Surface | Available in merged source | Remaining boundary |
|
|
94
94
|
| --- | --- | --- |
|
|
@@ -99,7 +99,7 @@ substitute for it.
|
|
|
99
99
|
| Shared state | Sequential and concurrent single-origin shared-world studies with retained evidence | Multi-origin implementation remains gated; concurrent state change does not establish per-action causation |
|
|
100
100
|
| Observer | Live/recorded views, participant assignments, action-specific links, saved moments, zoom, comparison and phone-width review | Sparse captures cannot prove every action's effect; visual comparison alone is not a controlled experiment |
|
|
101
101
|
| Review and feedback | Verification grades, feedback drafts, portable HTML, redacted bundle derivatives and computer-use completion-source labels | Sharing requires the appropriate grade; participant reports and condition matches still need task adjudication |
|
|
102
|
-
| Study findings | Explicit `analyze
|
|
102
|
+
| Study findings | Explicit `analyze` or opt-in post-run analysis, bounded evidence selection, versioned findings, exact source links and append-only corrections within the Observer study shell | Model interpretation needs review; bounded selection and source truncation limit coverage; opening Observer never dispatches analysis |
|
|
103
103
|
| TUI and serving | Detached starts, run stopping, reclamation, Observer attachment, loopback serving and run library | Stopping a process does not itself prove sandbox cleanup; TUI views over CLI `stats`/`export` remain follow-ups |
|
|
104
104
|
| Off-app communication | In-sandbox email/SMS catch and digest-only thread evidence | This does not establish real-provider delivery |
|
|
105
105
|
| Mobile and media | Hosted viewport/emulation, desktop geometry checks, bounded dwell and declared camera feed | Physical-device and touch fidelity remain unproven; unsupported microphone declarations are rejected |
|
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
# Automatic study analysis
|
|
2
|
+
|
|
3
|
+
A lab can request analysis after each live recording finishes. Findings remain
|
|
4
|
+
separate from participant feedback and the recorded study verdict.
|
|
5
|
+
|
|
6
|
+
```yaml
|
|
7
|
+
review:
|
|
8
|
+
analysis:
|
|
9
|
+
maxCostUsd: 3
|
|
10
|
+
# Optional; these values match manual analysis defaults.
|
|
11
|
+
model: gpt-6-astra
|
|
12
|
+
timeoutMs: 300000
|
|
13
|
+
maxOutputTokens: 16384
|
|
14
|
+
# question: Where did participants need to recover?
|
|
15
|
+
```
|
|
16
|
+
|
|
17
|
+
Omit `review.analysis` to leave the existing run behavior unchanged. The budget
|
|
18
|
+
is required when analysis is present. It limits an admission estimate, not the
|
|
19
|
+
provider's final bill, and is separate from participant spending limits. Analysis
|
|
20
|
+
sends selected retained text and captures to OpenAI using `OPENAI_API_KEY`.
|
|
21
|
+
Analysis runs in the Humanish runner using its credentials. This setting adds no
|
|
22
|
+
credential channel to the target application; each participant backend retains
|
|
23
|
+
its existing authentication boundary. Review a manifest's opt-in before running
|
|
24
|
+
it live.
|
|
25
|
+
|
|
26
|
+
The same configuration works through `humanish run <lab>`, `lab run <lab>`,
|
|
27
|
+
`watch <lab>`, and TUI live starts. Direct library calls to the five recording
|
|
28
|
+
producers honor it too. Supported routes are computer-use, scripted-browser,
|
|
29
|
+
terminal-product, sequential shared-world and concurrent shared-world. Synthetic,
|
|
30
|
+
smoke and meta routes reject the setting before execution. Dry runs show analysis
|
|
31
|
+
as skipped, without reading analysis credentials or making a provider request.
|
|
32
|
+
|
|
33
|
+
Participant execution finishes and its recording is finalized before analysis is
|
|
34
|
+
queued. A participant who was blocked or interrupted can still have useful
|
|
35
|
+
retained evidence; analysis requires a verified live recording, not a successful
|
|
36
|
+
participant outcome. An active, missing or invalid recording is not analyzed.
|
|
37
|
+
|
|
38
|
+
The command waits for analysis and reports its separate state. A TUI-launched
|
|
39
|
+
runner continues after the TUI closes; reopening the TUI or Observer reads the
|
|
40
|
+
existing job and does not start another request. Concurrent or repeated automatic
|
|
41
|
+
invocations cannot silently retry a paid attempt. If a process disappears while
|
|
42
|
+
an attempt is in flight, its state can be unknown rather than falsely complete.
|
|
43
|
+
Use manual `humanish analyze --run <exact-run-id> --max-cost 3` for an intentional
|
|
44
|
+
follow-up after inspecting the existing attempt and its accounting.
|
|
45
|
+
|
|
46
|
+
Stopping participant execution does not start a fresh automatic analysis. A
|
|
47
|
+
recorded harness cancellation is skipped; ordinary time limits and participant
|
|
48
|
+
abandonment remain eligible evidence.
|
|
49
|
+
|
|
50
|
+
During analysis, Ctrl-C asks the request to cancel. The TUI's **Cancel analysis**
|
|
51
|
+
action writes a cancellation request for that recording; it does not signal the
|
|
52
|
+
finished participant process. Cancelling cannot undo provider work already
|
|
53
|
+
accepted. Known usage is retained; missing usage remains unknown.
|
|
54
|
+
|
|
55
|
+
The CLI's JSON keeps `runOk` for the original backend result, `automaticAnalysis`
|
|
56
|
+
for post-run analysis, and `ok` for the overall request. Failed, cancelled or
|
|
57
|
+
unknown analysis produces exit code 2 without discarding the recording. Partial
|
|
58
|
+
findings remain visibly partial; a valid partial result can succeed, while a
|
|
59
|
+
partial result with an analysis error still fails the command. Recorded task
|
|
60
|
+
outcomes and the deterministic review verdict are never rewritten by analysis.
|
|
61
|
+
|
|
62
|
+
See the [analysis contract](../contracts/study-analysis.md) for selection limits,
|
|
63
|
+
evidence validation, actual usage, corrections and share-safe export behavior.
|
package/docs/ramp/README.md
CHANGED
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
Status: public-safe contributor and agent ramp.
|
|
4
4
|
|
|
5
|
-
Package/source version in this tree: `0.
|
|
5
|
+
Package/source version in this tree: `0.90.0` (2026-09-15). The Observer is phone-usable as a stated requirement (observer/AGENTS.md); interactive primitives start from Base UI. The Observer renderer is the observer/ workspace artifact only; the legacy string-concat renderer was deleted at cutover (#426), and rollback is a version pin to 0.42.0. The containment boundary introduced in
|
|
6
6
|
`0.15.1` remains in force: managed run and output paths bind to validated
|
|
7
7
|
physical filesystem identities, and stored provider IDs are evidence, not
|
|
8
8
|
cleanup authority. The bundled OSS meta-lab is dry-run only until
|
|
@@ -47,6 +47,13 @@ If a change does not improve one of those loops, it probably belongs elsewhere.
|
|
|
47
47
|
|
|
48
48
|
## Current State
|
|
49
49
|
|
|
50
|
+
The [0.90.0 release note](../release/0.90.0-automatic-analysis.md) describes
|
|
51
|
+
opt-in analysis after live runs, truthful job states, cancellation, and scripted
|
|
52
|
+
captures and assignments in findings and playback.
|
|
53
|
+
|
|
54
|
+
The [0.89.1 release note](../release/0.89.1-analysis-finished-notice.md)
|
|
55
|
+
clarifies that an analysis with limitations has finished.
|
|
56
|
+
|
|
50
57
|
The [0.89.0 release note](../release/0.89.0-study-findings.md) describes explicit
|
|
51
58
|
analysis of completed studies, versioned findings and review corrections, exact
|
|
52
59
|
evidence links, and Participants / Findings within one Observer shell. Analysis
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
# Humanish 0.89.1: finished analysis notice
|
|
2
|
+
|
|
3
|
+
Observer now labels a partial analysis result **“Analysis finished with
|
|
4
|
+
limitations.”** The previous notice described findings as covering evidence
|
|
5
|
+
“included so far,” which could make a finished analysis appear to be running.
|
|
6
|
+
|
|
7
|
+
A partial result can contain validated findings after evidence selection limits
|
|
8
|
+
or an exceeded admission estimate. Review the recorded coverage, limitations and
|
|
9
|
+
usage when interpreting that result. The analysis status and saved evidence keep
|
|
10
|
+
their existing meaning.
|
|
11
|
+
|
|
12
|
+
## Update an existing recording
|
|
13
|
+
|
|
14
|
+
```bash
|
|
15
|
+
npm install humanish@0.89.1
|
|
16
|
+
npx humanish observe --run latest
|
|
17
|
+
```
|
|
18
|
+
|
|
19
|
+
This renders the existing study with the updated Observer. An already exported
|
|
20
|
+
HTML file is a saved snapshot; export it again to get the new wording. Reuse the
|
|
21
|
+
appropriate sharing options for that run. No new analysis request is needed.
|
|
22
|
+
|
|
23
|
+
The [0.89.0 release note](0.89.0-study-findings.md) describes the underlying study
|
|
24
|
+
findings feature and its verification limits.
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
# 0.90.0 — Findings after live studies
|
|
2
|
+
|
|
3
|
+
Labs can opt into independent analysis after their recording finishes with
|
|
4
|
+
`review.analysis.maxCostUsd`. The command waits for the result; the TUI and
|
|
5
|
+
Observer show analysis separately from participant execution and task outcomes.
|
|
6
|
+
Without the setting, run behavior stays unchanged.
|
|
7
|
+
|
|
8
|
+
The five supported live recording producers share the same completion boundary.
|
|
9
|
+
Automatic analysis claims one attempt per recording, preserves usage and prior
|
|
10
|
+
findings, and never starts again merely because a view is reopened. Cancellation
|
|
11
|
+
requests target analysis without signaling the finished participant process.
|
|
12
|
+
Recording identity remains pinned through dispatch, reuse and artifact writes.
|
|
13
|
+
|
|
14
|
+
Scripted browser recordings now contribute their action-attached screenshots and
|
|
15
|
+
declared goal to analysis and playback. Earlier saved analyses retain their
|
|
16
|
+
original evidence mapping. Referenced error captures do not become duplicate
|
|
17
|
+
frames. Missing assignments or unmapped captures remain visible limitations.
|
|
18
|
+
|
|
19
|
+
See [automatic analysis](../product/automatic-analysis.md) for configuration,
|
|
20
|
+
route support, spending and cancellation behavior. Analysis uses an admission
|
|
21
|
+
estimate separate from participant caps; it is not a provider billing hard cap.
|
package/package.json
CHANGED