humanish 0.88.0 → 0.88.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (50) hide show
  1. package/README.md +12 -2
  2. package/dist/actor-contract.d.ts +1 -1
  3. package/dist/actor-goal-source.d.ts +5 -0
  4. package/dist/actor-goal-source.js +32 -0
  5. package/dist/actor-goal-source.js.map +1 -0
  6. package/dist/actor-stop-cause.js +1 -0
  7. package/dist/actor-stop-cause.js.map +1 -1
  8. package/dist/computer-use-actor.d.ts +2 -0
  9. package/dist/computer-use-actor.js +6 -3
  10. package/dist/computer-use-actor.js.map +1 -1
  11. package/dist/computer-use.d.ts +5 -2
  12. package/dist/computer-use.js +38 -3
  13. package/dist/computer-use.js.map +1 -1
  14. package/dist/cua-actor-lab.js +9 -7
  15. package/dist/cua-actor-lab.js.map +1 -1
  16. package/dist/cua-diagnostics.d.ts +1 -1
  17. package/dist/cua-diagnostics.js +1 -1
  18. package/dist/cua-diagnostics.js.map +1 -1
  19. package/dist/e2b-desktop-executor.js +4 -2
  20. package/dist/e2b-desktop-executor.js.map +1 -1
  21. package/dist/feedback.js +4 -12
  22. package/dist/feedback.js.map +1 -1
  23. package/dist/index.d.ts +2 -2
  24. package/dist/index.js +1 -1
  25. package/dist/index.js.map +1 -1
  26. package/dist/lab-config.d.ts +1 -1
  27. package/dist/observer-app.html +1 -1
  28. package/dist/observer-data.js +11 -4
  29. package/dist/observer-data.js.map +1 -1
  30. package/dist/openai-responses-cu.d.ts +3 -0
  31. package/dist/openai-responses-cu.js +4 -1
  32. package/dist/openai-responses-cu.js.map +1 -1
  33. package/dist/run.d.ts +33 -4
  34. package/dist/run.js +175 -5
  35. package/dist/run.js.map +1 -1
  36. package/dist/shared-world-lab.d.ts +2 -0
  37. package/dist/shared-world-lab.js +90 -5
  38. package/dist/shared-world-lab.js.map +1 -1
  39. package/dist/stats.js +1 -1
  40. package/dist/stats.js.map +1 -1
  41. package/docs/architecture/examples/state-driven-local-app/README.md +75 -0
  42. package/docs/architecture/examples/state-driven-local-app/app.mjs +48 -0
  43. package/docs/architecture/examples/state-driven-local-app/runner.mjs +104 -0
  44. package/docs/architecture/state-driven-executor.md +19 -26
  45. package/docs/contracts/schemas.md +64 -4
  46. package/docs/goals/current.md +9 -4
  47. package/docs/ramp/README.md +11 -1
  48. package/docs/release/0.88.1-completion-evidence-and-local-app.md +79 -0
  49. package/docs/release/0.88.2-sequential-study-budgets.md +57 -0
  50. package/package.json +1 -1
@@ -0,0 +1,48 @@
1
+ // @ts-check
2
+ import { once } from "node:events";
3
+ import { createServer } from "node:http";
4
+
5
+ /** A synthetic app with a real loopback HTTP state/action contract. */
6
+ export async function startLocalApp() {
7
+ let greeted = false;
8
+ let messages = 0;
9
+ let stateReads = 0;
10
+ const server = createServer(async (request, response) => {
11
+ if (request.method === "GET" && request.url === "/state") {
12
+ stateReads += 1;
13
+ response.writeHead(200, { "content-type": "application/json" });
14
+ response.end(JSON.stringify({ greeted, messages }));
15
+ } else if (request.method === "POST" && request.url === "/chat") {
16
+ // The demo accepts only a short synthetic message; it stores no raw text.
17
+ let text = "";
18
+ for await (const chunk of request) {
19
+ text += chunk.toString();
20
+ if (text.length > 1024) {
21
+ response.writeHead(413).end();
22
+ return;
23
+ }
24
+ }
25
+ messages += 1;
26
+ if (text.toLowerCase().includes("hello")) greeted = true;
27
+ response.writeHead(204).end();
28
+ } else {
29
+ response.writeHead(404).end();
30
+ }
31
+ });
32
+ server.listen(0, "127.0.0.1");
33
+ await once(server, "listening");
34
+ const address = server.address();
35
+ if (!address || typeof address === "string") throw new Error("No loopback port");
36
+
37
+ return {
38
+ appUrl: `http://127.0.0.1:${address.port}/`,
39
+ // These counters make real HTTP reads and state-changing writes inspectable.
40
+ getReceipt: () => ({ greeted, messages, stateReads, serverClosed: !server.listening }),
41
+ async close() {
42
+ await new Promise((resolve, reject) => {
43
+ server.close((error) => error ? reject(error) : resolve(undefined));
44
+ server.closeAllConnections();
45
+ });
46
+ }
47
+ };
48
+ }
@@ -0,0 +1,104 @@
1
+ // @ts-check
2
+ import { LAB_CONFIG_SCHEMA, parseLabConfig, runLab, stableProgressKey, verifyRun } from "humanish";
3
+ import { startLocalApp } from "./app.mjs";
4
+
5
+ /** @param {string} appUrl @returns {import("humanish").CuaExecutor} */
6
+ function createAppContractExecutor(appUrl) {
7
+ return {
8
+ async observe() {
9
+ const response = await fetch(new URL("state", appUrl), { signal: AbortSignal.timeout(5000) });
10
+ if (!response.ok) throw new Error(`State read failed: HTTP ${response.status}`);
11
+ const state = await response.json();
12
+ if (typeof state?.greeted !== "boolean" || !Number.isInteger(state?.messages)) {
13
+ throw new Error("Unexpected app state");
14
+ }
15
+ const appState = { greeted: state.greeted, messages: state.messages };
16
+ return { stateSignature: stableProgressKey(appState), appState };
17
+ },
18
+ async execute(action, signal) {
19
+ signal?.throwIfAborted();
20
+ if (action.kind !== "type") throw new Error(`Unsupported app action: ${action.kind}`);
21
+ const response = await fetch(new URL("chat", appUrl), {
22
+ method: "POST",
23
+ body: action.text,
24
+ signal: AbortSignal.any([AbortSignal.timeout(5000), ...(signal ? [signal] : [])])
25
+ });
26
+ if (!response.ok) throw new Error(`Chat write failed: HTTP ${response.status}`);
27
+ }
28
+ };
29
+ }
30
+
31
+ // A deterministic rule, not an AI participant: no model SDK, requests or credentials.
32
+ /** @type {import("humanish").CuaProvider} */
33
+ const provider = {
34
+ id: "local-app-deterministic-example",
35
+ version: "1.0.0",
36
+ requiresFrame: false,
37
+ capabilities: {
38
+ headless: true,
39
+ structuredTrace: true,
40
+ lanes: ["computer-use"],
41
+ producesScreenshots: false,
42
+ byoModel: true,
43
+ preGrantableApprovals: false,
44
+ inProcessTools: false,
45
+ license: "open"
46
+ },
47
+ async nextTurn(request, signal) {
48
+ signal.throwIfAborted();
49
+ const greeted = request.observation.appState?.greeted;
50
+ if (typeof greeted !== "boolean") throw new Error("Missing greeted observation");
51
+ return {
52
+ actions: greeted ? [] : [{ kind: "type", text: "hello there" }],
53
+ pendingSafetyChecks: [],
54
+ done: greeted,
55
+ message: greeted ? "The app accepted the greeting." : "Sending a greeting."
56
+ };
57
+ }
58
+ };
59
+
60
+ const app = await startLocalApp();
61
+ try {
62
+ // parseLabConfig accepts a decoded OBJECT, not a YAML string.
63
+ const parsed = parseLabConfig({
64
+ schema: LAB_CONFIG_SCHEMA,
65
+ id: "state-driven-local-app-example",
66
+ title: "Deterministic local-app integration example",
67
+ subject: { source: "local-app", appUrl: app.appUrl },
68
+ // This registry id selects the CUA loop. buildProvider supplies the actual provider.
69
+ actors: [{ type: "openai-computer-use", persona: "pixel-pat", mission: "Greet the app." }],
70
+ scenario: { mode: "live" },
71
+ execution: { timeoutMs: 15_000 }
72
+ });
73
+ if (!parsed.ok) throw new Error(parsed.error.message);
74
+ const outcome = await runLab(parsed.config, {
75
+ cwd: process.cwd(),
76
+ dryRun: false,
77
+ cuaHooks: {
78
+ buildExecutor: async ({ appUrl }) => createAppContractExecutor(appUrl),
79
+ buildProvider: async () => provider
80
+ }
81
+ });
82
+ if (outcome.backend !== "cua") throw new Error(`Unexpected backend: ${outcome.backend}`);
83
+ const { result } = outcome;
84
+ const verified = await verifyRun(process.cwd(), result.runId);
85
+ console.log(JSON.stringify({
86
+ runId: result.runId,
87
+ ok: result.ok,
88
+ completionReason: result.session?.completionReason,
89
+ provider: provider.id,
90
+ sandboxCreated: result.sandbox !== undefined,
91
+ screenshots: result.session?.screenshots,
92
+ verification: verified,
93
+ app: app.getReceipt(),
94
+ // A mechanism statement, not an inference from missing provider usage/rates.
95
+ costBasis: "No model calls or hosted resources; only this process and loopback HTTP."
96
+ }, null, 2));
97
+ if (!result.ok || !verified.ok || result.session?.completionReason !== "goal_satisfied"
98
+ || result.sandbox !== undefined || !app.getReceipt().greeted) {
99
+ throw new Error(result.error?.message ?? "Local-app example did not complete and verify");
100
+ }
101
+ } finally {
102
+ await app.close();
103
+ console.log(JSON.stringify({ cleanup: app.getReceipt() }));
104
+ }
@@ -52,8 +52,9 @@ wait and does not cache a previous action's position.
52
52
  `redaction.screenshots` resolves to `"n/a"`. No fabricated `Buffer.alloc(0)`
53
53
  ever reaches disk. (A vision executor still returns a frame, exactly as before.)
54
54
  - **`stateSignature` is still required.** Derive it from your own state — e.g.
55
- `JSON.stringify({ route, turn, modal })`. It is the canonical fallback progress
56
- key when `appState` is absent. It is never written to the trace as text.
55
+ `stableProgressKey({ route, turn, modal })` (exported from `humanish`). It is the
56
+ canonical fallback progress key when `appState` is absent. It is never written
57
+ to the trace as text.
57
58
  - **`appState` is the preferred progress input.** When present, the loop's
58
59
  friction / no-progress detection keys off a deterministic, sorted-key,
59
60
  depth/length-capped projection of it (`stableProgressKey`) rather than the
@@ -168,33 +169,25 @@ from your own state, pair it with a non-vision `CuaProvider`, and call
168
169
 
169
170
  ### 2. `runLab` + `buildExecutor` / `buildProvider` (keeps the composition)
170
171
 
171
- The supported library path. It keeps personas, the Observer, the evidence
172
- bundle, redaction, and the friction loop, while skipping E2B entirely:
172
+ The supported library path keeps personas, the Observer, the evidence bundle,
173
+ redaction, and the friction loop, while skipping E2B entirely. Start with the
174
+ [complete runnable example](examples/state-driven-local-app/README.md), which
175
+ ships in the npm package:
173
176
 
174
- ```ts
175
- import { runLab, parseLabConfig, type CuaExecutor, type CuaProvider } from "humanish";
176
-
177
- // local-app YAML (shareable; fails closed without hooks):
178
- // schema: humanish.lab.v2
179
- // id: downstream-local-app-state
180
- // subject: { source: local-app, appUrl: http://localhost:5173 }
181
- // actors: [{ type: openai-computer-use, persona: curious-tester, mission: "…" }]
182
- // scenario: { mode: live }
183
- const parsed = parseLabConfig(yaml);
184
- if (!parsed.ok) throw new Error(parsed.error.message);
185
-
186
- const outcome = await runLab(parsed.config, {
187
- cwd: process.cwd(),
188
- dryRun: false,
189
- cuaHooks: {
190
- buildExecutor: async ({ appUrl }) => createAppContractExecutor(bridge, appUrl),
191
- buildProvider: async () => createStateBrain(),
192
- },
193
- });
194
- // outcome.backend === "cua"; outcome.result.sandbox === undefined (NO E2B); the
195
- // trace's provider id is the injected brain's id.
177
+ ```bash
178
+ npm install humanish
179
+ node node_modules/humanish/docs/architecture/examples/state-driven-local-app/runner.mjs
196
180
  ```
197
181
 
182
+ The example includes a real loopback app, HTTP state/action bridge, full config
183
+ object (`schema: LAB_CONFIG_SCHEMA`), deterministic non-vision provider,
184
+ `stableProgressKey`, `runLab`, verification, a printed report and `finally`
185
+ cleanup. It makes no model calls and does not demonstrate persona efficacy.
186
+ `parseLabConfig` accepts a decoded object, not a YAML string; narrow its result
187
+ on `.ok`, then narrow `runLab`'s result on `backend === "cua"` before accessing
188
+ the CUA result. The example defines all helpers rather than requiring a consumer
189
+ to reconstruct them.
190
+
198
191
  When `cuaHooks.buildExecutor` is set, `runCuaActorLab` takes a branch that NEVER
199
192
  loads the E2B module, creates a sandbox, runs `prepareDesktop`, provisions a
200
193
  clone, opens a browser, or starts a stream. `sandboxId`/`streamUrl` stay
@@ -3,7 +3,7 @@
3
3
  Date: 2026-06-02 (current-state note updated 2026-07-14)
4
4
 
5
5
  Status: reference map for the major contracts shipped through source version
6
- `0.88.0`; it is not an exhaustive inventory of command/result envelopes. Exported types,
6
+ `0.88.2`; it is not an exhaustive inventory of command/result envelopes. Exported types,
7
7
  schema constants, parsers, and validators in `src/` are authoritative. Rows
8
8
  marked "reserved" name layering intent only — no code emits or validates them
9
9
  yet. Do not emit a reserved schema.
@@ -210,7 +210,11 @@ A lab is a composition over code primitives, not a hardcoded kind:
210
210
  template actually used is recorded in the run bundle as `desktopTemplate`
211
211
  (public-safe — a template name is not a secret). Inert (warned) on every route
212
212
  that creates no desktop, incl. the in-process `local-app` cua route and the
213
- meta route — never silently ignored (invariant 6);
213
+ meta route — never silently ignored (invariant 6). Custom images need the
214
+ Desktop SDK's `xdotool` input support and `xclip` or `xsel` on `PATH` for
215
+ clipboard recovery when direct typing fails. If both clipboard utilities are
216
+ absent, recovery fails with `clipboard-utility-missing`; a dry-run does not
217
+ inspect the image's installed tools;
214
218
  - `execution.desktop.browser` (e2b-desktop computer-use/fan-out routes, plus
215
219
  sequential and concurrent shared-world actor seats): optional browser family
216
220
  preference: `default`, `chrome`, `chromium`, or `firefox`. Absent/default
@@ -509,6 +513,16 @@ shared-world bundle adds TWO additive, optional fields to `humanish.run-bundle.v
509
513
 
510
514
  SEQUENTIAL shape (`topologyMode: sequential`, #164 PR1):
511
515
  - `sequence: [roleId, …]` — the role ids that actually took a turn, in declared order.
516
+ - `skippedTail` (optional, live sequential only) — `{ afterRoleId, roles,
517
+ cause, maxTotalUsd?, estimatedTotalUsd? }`. Each ordered `roles` entry names
518
+ `{ roleId, simId, streamId }` for an unstarted participant. Together with the
519
+ executed prefix it must account for the full declared denominator. The
520
+ predecessor must have a matching `harness_error`, explicit `session_error`,
521
+ `usage_unreported`, or measured `study_spend_limit`. Only the last cause
522
+ carries budget figures, using the same per-participant estimates as the
523
+ tracker. Blocked seats need matching simulation, stream and blocked-event
524
+ evidence, with no actor, trace, screenshot or invented timeline turn.
525
+ Historical bundles without this field still require every role in the timeline.
512
526
  - `timeline: (checkpoint | turn)[]` — a harness-clocked, strictly alternating
513
527
  timeline that starts `cp-baseline`, alternates checkpoint → turn → checkpoint,
514
528
  and ends on a checkpoint:
@@ -1063,8 +1077,9 @@ request's worst-case cost. In-flight requests, retries, concurrent lanes, and
1063
1077
  unreported usage can exceed or escape these estimates. These thresholds are
1064
1078
  not hard provider billing caps and exclude desktop and target-app charges.
1065
1079
 
1066
- A lane with material progress that crosses `maxUsd` ends `budget_reached` /
1067
- `incomplete`; the existing zero-action guard ends `gave_up` / `abandoned`.
1080
+ Crossing `maxUsd` ends `budget_reached` / `incomplete`, including when no
1081
+ material action has executed. The recorded reason distinguishes prior progress
1082
+ from no material progress; a harness budget stop is not participant abandonment.
1068
1083
  Crossing the shared study threshold ends `budget_reached` / `incomplete`, with
1069
1084
  sibling lanes stopping when their next post-response check sees it. Reaching a
1070
1085
  threshold is not proof of task completion.
@@ -1077,6 +1092,41 @@ threshold on a model `src/pricing.ts` cannot price is refused at preflight
1077
1092
  (`HUMANISH_CUA_LAB_UNPRICED_CAP`) before sandbox allocation. This rate-availability
1078
1093
  check is separate from the post-response spend check.
1079
1094
 
1095
+ Sequential shared-world studies (`subject.topology: shared-world` with
1096
+ `execution.concurrency: 1`, using clone or local-tree subjects) enforce these
1097
+ same per-participant and shared model thresholds. Final reported usage,
1098
+ including a closing request, is reconciled before admitting the next participant.
1099
+ After the aggregate threshold is crossed, later participants are `blocked` with
1100
+ a recorded skip reason; they make no model requests and add no executed turn to
1101
+ the checkpoint timeline. An unpriced model with a declared threshold fails
1102
+ before allocation with `HUMANISH_SHARED_WORLD_LAB_INVALID`.
1103
+
1104
+ On this sequential route, an otherwise completed capped interaction that returns
1105
+ missing or partial usage stops with `harness_error`, `stopCause: usage_unreported`, and the label
1106
+ “provider usage unavailable.” It is not recorded as a crossed threshold. A
1107
+ stalled or failed request with unknown spend is not retried by the CUA loop.
1108
+ The default OpenAI adapter also disables HTTP and policy-negotiation retries for
1109
+ these strict capped sessions. The loop cancels its owned request signal when a
1110
+ request ends or its timeout wins; injected providers must honor cancellation
1111
+ and remain responsible for their own internal dispatch. Known usage remains in the
1112
+ trace alongside an explicit unknown; subsequent participants do not start when
1113
+ the shared budget cannot be established. Reported zero input and output counts
1114
+ remain valid zero usage. This stricter unknown-usage policy is specific to
1115
+ sequential capped studies; other routes retain their existing behavior. An
1116
+ explicitly incomplete provider response retains its original interruption cause
1117
+ first, with any missing usage still recorded as unknown.
1118
+
1119
+ Sequential traces persist dated model estimates. Their run cost summary marks
1120
+ desktop compute as unmeasured, so the displayed model subtotal is a lower bound.
1121
+ The sequential route does not provide a running Observer usage stream; its final
1122
+ CLI and Observer projections read these persisted estimates.
1123
+
1124
+ A capped custom session must return the declared model identity on its trace.
1125
+ A mismatch fails orchestration and blocks later participants while preserving
1126
+ the participant's original outcome and the estimate for its returned model.
1127
+ This check does not establish which model an arbitrary custom runner actually
1128
+ called or retrospectively enforce a runner that ignored its cap options.
1129
+
1080
1130
  These computer-use rules do not replace the terminal route's separate
1081
1131
  `scenario.caps` cost-ledger and product-spend rules described above.
1082
1132
 
@@ -1228,6 +1278,16 @@ the stream shape, the meaningful-use rubric, and hard-failure rules.
1228
1278
  Review summarizes whether evidence supports the claim. It does not replace
1229
1279
  verification or maintainer acceptance.
1230
1280
 
1281
+ `verdict` is the run gate result. `participants.reachedGoal` retains the count of
1282
+ recorded successful sessions; it is not an independent adjudication of their
1283
+ claims. Computer-use review and Observer labels distinguish a participant's
1284
+ reported completion from a recorded `stopWhen` or dwell condition match. A
1285
+ condition match establishes that condition, not every aspect of the mission.
1286
+ Missing or malformed completion evidence is labeled unavailable. Other actor
1287
+ routes retain their own completion semantics. Re-reading a historical review
1288
+ refreshes these labels without rewriting the original bundle or actor trace.
1289
+ Share-safety verification does not establish task success.
1290
+
1231
1291
  Core-owned fields:
1232
1292
 
1233
1293
  - `schema`
@@ -1,9 +1,9 @@
1
1
  # Current Goals
2
2
 
3
- Status date: 2026-09-11. Published baseline: `0.88.0`.
3
+ Status date: 2026-09-13. Published baseline: `0.88.2`.
4
4
 
5
5
  This page guides work on current merged source. Published behavior is described
6
- in the [release notes](../release/0.88.0-study-diagnostics.md).
6
+ in the [release notes](../release/0.88.2-sequential-study-budgets.md).
7
7
  The [September 9 history](https://github.com/danielgwilson/humanish/blob/main/docs/goals/current-history-2026-09-09.md)
8
8
  preserves the former status log; its queues do not supersede this page.
9
9
 
@@ -88,7 +88,7 @@ requires decision-equivalent retained evidence and a real deletion branch.
88
88
  No first-party deletion branch has met that gate. Public demonstrations do not
89
89
  substitute for it.
90
90
 
91
- ## Current Program Truth (source `0.88.0`)
91
+ ## Current Program Truth (source `0.88.2`)
92
92
 
93
93
  | Surface | Available in merged source | Remaining boundary |
94
94
  | --- | --- | --- |
@@ -98,7 +98,7 @@ substitute for it.
98
98
  | Task protocol | Hidden criteria and per-task outcomes on supported per-lane CUA paths, including local-agent and desktop-cli | Shared-world, terminal-product, scripted and synthetic routes reject `tasks` before execution |
99
99
  | Shared state | Sequential and concurrent single-origin shared-world studies with retained evidence | Multi-origin implementation remains gated; concurrent state change does not establish per-action causation |
100
100
  | Observer | Live/recorded views, participant assignments, action-specific links, saved moments, zoom, comparison and phone-width review | Sparse captures cannot prove every action's effect; visual comparison alone is not a controlled experiment |
101
- | Review and feedback | Verification grades, feedback drafts, portable HTML and redacted bundle derivatives | Sharing requires the appropriate grade; generated findings still need adjudication |
101
+ | Review and feedback | Verification grades, feedback drafts, portable HTML, redacted bundle derivatives and computer-use completion-source labels | Sharing requires the appropriate grade; participant reports and condition matches still need task adjudication |
102
102
  | TUI and serving | Detached starts, run stopping, reclamation, Observer attachment, loopback serving and run library | Stopping a process does not itself prove sandbox cleanup; TUI views over CLI `stats`/`export` remain follow-ups |
103
103
  | Off-app communication | In-sandbox email/SMS catch and digest-only thread evidence | This does not establish real-provider delivery |
104
104
  | Mobile and media | Hosted viewport/emulation, desktop geometry checks, bounded dwell and declared camera feed | Physical-device and touch fidelity remain unproven; unsupported microphone declarations are rejected |
@@ -108,6 +108,11 @@ Use the [task support matrix](../architecture/task-protocol-support.md),
108
108
  and [CLI reference](https://humanish.dev/docs/cli) when choosing a concrete path.
109
109
  Source behavior and required tests outrank stale status prose.
110
110
 
111
+ The library-assisted `local-app` route now includes a
112
+ [runnable npm example](../architecture/examples/state-driven-local-app/README.md).
113
+ Its deterministic provider demonstrates the integration with a real loopback
114
+ app; it does not establish persona effectiveness or independent adoption.
115
+
111
116
  ## Gates And Deferred Work
112
117
 
113
118
  - Live OSS meta-lab execution remains disabled until repository-derived
@@ -2,7 +2,7 @@
2
2
 
3
3
  Status: public-safe contributor and agent ramp.
4
4
 
5
- Package/source version in this tree: `0.88.0` (2026-09-11). The Observer is phone-usable as a stated requirement (observer/AGENTS.md); interactive primitives start from Base UI. The Observer renderer is the observer/ workspace artifact only; the legacy string-concat renderer was deleted at cutover (#426), and rollback is a version pin to 0.42.0. The containment boundary introduced in
5
+ Package/source version in this tree: `0.88.2` (2026-09-13). The Observer is phone-usable as a stated requirement (observer/AGENTS.md); interactive primitives start from Base UI. The Observer renderer is the observer/ workspace artifact only; the legacy string-concat renderer was deleted at cutover (#426), and rollback is a version pin to 0.42.0. The containment boundary introduced in
6
6
  `0.15.1` remains in force: managed run and output paths bind to validated
7
7
  physical filesystem identities, and stored provider IDs are evidence, not
8
8
  cleanup authority. The bundled OSS meta-lab is dry-run only until
@@ -47,6 +47,16 @@ If a change does not improve one of those loops, it probably belongs elsewhere.
47
47
 
48
48
  ## Current State
49
49
 
50
+ The [0.88.2 release note](../release/0.88.2-sequential-study-budgets.md)
51
+ describes model-spend thresholds on sequential shared-world studies, blocked
52
+ later participants, and explicit unknown-usage accounting.
53
+
54
+ The [0.88.1 release note](../release/0.88.1-completion-evidence-and-local-app.md)
55
+ describes computer-use labels that distinguish participant reports from
56
+ recorded condition matches. It also covers the complete npm local-app example
57
+ and public `stableProgressKey` export, plus the clipboard fallback correction
58
+ for inherited output pipes.
59
+
50
60
  The [0.88.0 release note](../release/0.88.0-study-diagnostics.md) describes
51
61
  computer-use CLI diagnostics, explicit local admission limits and retained
52
62
  uncertainty when an earlier provider request did not report usage.
@@ -0,0 +1,79 @@
1
+ # Humanish 0.88.1: distinguish reported completion from condition matches
2
+
3
+ When a computer-use participant says it finished, review and Observer now
4
+ label that completion as **participant-reported**. A recorded `stopWhen` match
5
+ or completed dwell window identifies a **recorded completion condition** instead.
6
+ Missing, malformed or conflicting completion evidence is labeled unavailable.
7
+ Zero-completion counts use **0/N recorded completions**, and aggregate stats use
8
+ **recorded goal completions**.
9
+
10
+ A matched condition establishes only that condition, not every aspect of the
11
+ mission. The run gate and share-safety verification remain separate from task
12
+ adjudication. This release adds no automatic semantic evaluator.
13
+
14
+ Existing actor statuses, `participants.reachedGoal`, verdict values and
15
+ denominators stay unchanged. Current review commands, Observer rendering and
16
+ newly generated feedback drafts apply these labels to older runs while
17
+ preserving original bundles and actor traces. Existing HTML exports keep their
18
+ original renderer. Other actor routes retain their own completion semantics.
19
+
20
+ ## Run a local app from npm
21
+
22
+ To connect your local app's state and actions to Humanish, start with the
23
+ [complete npm example](../architecture/examples/state-driven-local-app/README.md):
24
+
25
+ ```bash
26
+ npm install humanish@0.88.1
27
+ node node_modules/humanish/docs/architecture/examples/state-driven-local-app/runner.mjs
28
+ ```
29
+
30
+ It starts a synthetic loopback app, reads its state, sends a greeting through
31
+ its HTTP action endpoint and verifies the recorded run. A `finally` block closes
32
+ the app. The provider follows a deterministic rule, so the example needs no
33
+ model credentials or E2B desktop. Use Node.js 20.3 or later.
34
+
35
+ The runner supplies both `CuaExecutor` and `CuaProvider`, passes a decoded object
36
+ to `parseLabConfig`, and checks the config and backend discriminants. The
37
+ existing `stableProgressKey` utility is now exported from `humanish`, so callers
38
+ can use the same bounded state projection as the loop. Replace the two ports
39
+ with your app bridge and provider; the example guide explains the responsibilities
40
+ that remain with them.
41
+
42
+ ## Clipboard fallback returns after the write
43
+
44
+ When direct desktop typing fails, Humanish can write the text to the X clipboard
45
+ and paste it. `xclip` and `xsel` fork a process to keep the selection available;
46
+ inherited output pipes could leave the command runner waiting until timeout.
47
+ Both clipboard-write commands now detach their output while preserving the
48
+ existing exit checks, temporary-file cleanup and paste dispatch.
49
+
50
+ Custom desktop images still need a working `xclip` or `xsel`. Missing utilities
51
+ continue to report `clipboard-utility-missing`. This patch adds no typing retry,
52
+ and recovery after a primary write inserted an unknown partial prefix remains
53
+ unproven. [Issue #340](https://github.com/danielgwilson/humanish/issues/340) remains
54
+ open for that broader typing problem.
55
+
56
+ ## Verification and limits
57
+
58
+ [Completion label checks](https://github.com/danielgwilson/humanish/pull/765)
59
+ cover participant reports, recorded condition matches, zero completions and
60
+ unavailable legacy detail. They preserve recorded counts and statuses while
61
+ refreshing historical review and feedback projections. Existing adapter
62
+ narrative remains available.
63
+
64
+ [The packaged example](https://github.com/danielgwilson/humanish/pull/762) completed
65
+ two independent local runs on Node 20.20.2 and 24.12.0. Each made two HTTP state
66
+ reads and one state-changing request, reached `goal_satisfied`, verified as
67
+ `share_ready` and closed its server. These checks used locally packed candidate
68
+ source labeled 0.88.0, distinct from the registry release with that version.
69
+
70
+ [The clipboard correction](https://github.com/danielgwilson/humanish/pull/761)
71
+ passed two real desktop conformance cases, one with explicitly installed `xclip`
72
+ and one with `xsel`. Each preserved the exact Unicode/newline/quoted text with
73
+ one paste and removed its transfer file. Both cases refused the primary write
74
+ before it could type anything. They do not establish default-image clipboard
75
+ availability or recovery from partial typing.
76
+
77
+ Both checks were model-free. They establish the tested integration and executor
78
+ behavior; persona effectiveness and independent maintainer adoption remain
79
+ separate questions.
@@ -0,0 +1,57 @@
1
+ # Humanish 0.88.2: sequential studies honor model-spend thresholds
2
+
3
+ Sequential shared-world studies now enforce the `execution.caps.maxUsd` and
4
+ `maxTotalUsd` thresholds they previously accepted without applying. This covers
5
+ computer-use participants sharing a clone or local-tree subject with
6
+ `subject.topology: shared-world` and `execution.concurrency: 1`.
7
+
8
+ Each participant's reported usage feeds the per-participant threshold and the
9
+ shared study estimate. Final usage, including a closing report, is reconciled
10
+ before the next participant starts. A participant interrupted by a threshold
11
+ has `budget_reached` / `incomplete`; its reason distinguishes prior activity
12
+ from no material progress. Later participants blocked by the shared threshold
13
+ make no model requests and add no executed turn to the checkpoint timeline.
14
+ A closing report that crosses the threshold preserves the already recorded
15
+ completion condition while blocking subsequent participants.
16
+ The bundle retains all declared participants and an explicit blocked suffix.
17
+ Verification checks the suffix against its preceding interruption, participant
18
+ records and unchanged executed timeline; absent participants cannot masquerade
19
+ as budget-blocked seats.
20
+
21
+ These checks happen after a model response and before its actions or another
22
+ participant turn. The current request can overshoot a threshold. They estimate
23
+ model spend; they do not reserve future requests or cap provider invoices,
24
+ desktop compute, or target-app charges. A zero threshold can still allow the
25
+ first model request. Use an explicit dry-run for a path without provider calls.
26
+
27
+ An unpriced model with a declared threshold is refused before allocation. For
28
+ an otherwise completed response, missing or partial usage ends capped sequential
29
+ execution with `harness_error` and `usage_unreported` (“provider usage unavailable”).
30
+ The CUA loop sends no further participant request, retry or closing request.
31
+ An explicitly incomplete provider response retains its original interruption
32
+ cause first; its missing usage remains unknown. For these strict capped sessions,
33
+ the default OpenAI adapter makes one dispatch per requested turn, including
34
+ HTTP errors and policy negotiation. The loop cancels the request signal when
35
+ its timeout wins, so an outstanding transport cannot retry after the loop ends.
36
+ Injected providers must honor that signal and control their own dispatches;
37
+ cancellation cannot undo an already billed request. Known partial costs stay
38
+ in the trace, and an unknown shared budget blocks later participants.
39
+
40
+ Sequential traces now retain dated model estimates. The run total explicitly
41
+ marks desktop compute as unmeasured, so the model subtotal is a lower bound.
42
+ Observer labels a partial estimate as known cost with the total unknown.
43
+ A custom session returning a different model identity fails orchestration and
44
+ blocks later participants while preserving its original outcome and the
45
+ estimate for its returned model. This does not retrospectively enforce an
46
+ arbitrary custom runner that ignored its cap options.
47
+
48
+ Existing bundles are not rewritten. Uncapped sessions and other execution
49
+ routes retain their existing behavior. The stricter unknown-usage rule applies
50
+ to sequential capped studies; the sequential route still has no running
51
+ Observer usage stream.
52
+
53
+ Verification covers the actual participant loop with captured provider usage,
54
+ individual and shared thresholds, unstarted seats, zero and missing usage,
55
+ closing requests, model mismatches, checkpoint evidence and cleanup. These
56
+ contract checks establish the tested behavior; they do not establish persona
57
+ effectiveness, independent adoption, or exact provider billing.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "humanish",
3
- "version": "0.88.0",
3
+ "version": "0.88.2",
4
4
  "description": "Open-source-safe CLI for persona simulation, observer review, and public-safe feedback drafts.",
5
5
  "author": "Daniel G Wilson <daniel@danielgwilson.com>",
6
6
  "keywords": [