humanish 0.88.0 → 0.88.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +12 -2
- package/dist/actor-contract.d.ts +1 -1
- package/dist/actor-goal-source.d.ts +5 -0
- package/dist/actor-goal-source.js +32 -0
- package/dist/actor-goal-source.js.map +1 -0
- package/dist/actor-stop-cause.js +1 -0
- package/dist/actor-stop-cause.js.map +1 -1
- package/dist/computer-use-actor.d.ts +2 -0
- package/dist/computer-use-actor.js +6 -3
- package/dist/computer-use-actor.js.map +1 -1
- package/dist/computer-use.d.ts +5 -2
- package/dist/computer-use.js +38 -3
- package/dist/computer-use.js.map +1 -1
- package/dist/cua-actor-lab.js +9 -7
- package/dist/cua-actor-lab.js.map +1 -1
- package/dist/cua-diagnostics.d.ts +1 -1
- package/dist/cua-diagnostics.js +1 -1
- package/dist/cua-diagnostics.js.map +1 -1
- package/dist/e2b-desktop-executor.js +4 -2
- package/dist/e2b-desktop-executor.js.map +1 -1
- package/dist/feedback.js +4 -12
- package/dist/feedback.js.map +1 -1
- package/dist/index.d.ts +2 -2
- package/dist/index.js +1 -1
- package/dist/index.js.map +1 -1
- package/dist/lab-config.d.ts +1 -1
- package/dist/observer-app.html +1 -1
- package/dist/observer-data.js +11 -4
- package/dist/observer-data.js.map +1 -1
- package/dist/openai-responses-cu.d.ts +3 -0
- package/dist/openai-responses-cu.js +4 -1
- package/dist/openai-responses-cu.js.map +1 -1
- package/dist/run.d.ts +33 -4
- package/dist/run.js +175 -5
- package/dist/run.js.map +1 -1
- package/dist/shared-world-lab.d.ts +2 -0
- package/dist/shared-world-lab.js +90 -5
- package/dist/shared-world-lab.js.map +1 -1
- package/dist/stats.js +1 -1
- package/dist/stats.js.map +1 -1
- package/docs/architecture/examples/state-driven-local-app/README.md +75 -0
- package/docs/architecture/examples/state-driven-local-app/app.mjs +48 -0
- package/docs/architecture/examples/state-driven-local-app/runner.mjs +104 -0
- package/docs/architecture/state-driven-executor.md +19 -26
- package/docs/contracts/schemas.md +64 -4
- package/docs/goals/current.md +9 -4
- package/docs/ramp/README.md +11 -1
- package/docs/release/0.88.1-completion-evidence-and-local-app.md +79 -0
- package/docs/release/0.88.2-sequential-study-budgets.md +57 -0
- package/package.json +1 -1
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
// @ts-check
|
|
2
|
+
import { once } from "node:events";
|
|
3
|
+
import { createServer } from "node:http";
|
|
4
|
+
|
|
5
|
+
/** A synthetic app with a real loopback HTTP state/action contract. */
|
|
6
|
+
export async function startLocalApp() {
|
|
7
|
+
let greeted = false;
|
|
8
|
+
let messages = 0;
|
|
9
|
+
let stateReads = 0;
|
|
10
|
+
const server = createServer(async (request, response) => {
|
|
11
|
+
if (request.method === "GET" && request.url === "/state") {
|
|
12
|
+
stateReads += 1;
|
|
13
|
+
response.writeHead(200, { "content-type": "application/json" });
|
|
14
|
+
response.end(JSON.stringify({ greeted, messages }));
|
|
15
|
+
} else if (request.method === "POST" && request.url === "/chat") {
|
|
16
|
+
// The demo accepts only a short synthetic message; it stores no raw text.
|
|
17
|
+
let text = "";
|
|
18
|
+
for await (const chunk of request) {
|
|
19
|
+
text += chunk.toString();
|
|
20
|
+
if (text.length > 1024) {
|
|
21
|
+
response.writeHead(413).end();
|
|
22
|
+
return;
|
|
23
|
+
}
|
|
24
|
+
}
|
|
25
|
+
messages += 1;
|
|
26
|
+
if (text.toLowerCase().includes("hello")) greeted = true;
|
|
27
|
+
response.writeHead(204).end();
|
|
28
|
+
} else {
|
|
29
|
+
response.writeHead(404).end();
|
|
30
|
+
}
|
|
31
|
+
});
|
|
32
|
+
server.listen(0, "127.0.0.1");
|
|
33
|
+
await once(server, "listening");
|
|
34
|
+
const address = server.address();
|
|
35
|
+
if (!address || typeof address === "string") throw new Error("No loopback port");
|
|
36
|
+
|
|
37
|
+
return {
|
|
38
|
+
appUrl: `http://127.0.0.1:${address.port}/`,
|
|
39
|
+
// These counters make real HTTP reads and state-changing writes inspectable.
|
|
40
|
+
getReceipt: () => ({ greeted, messages, stateReads, serverClosed: !server.listening }),
|
|
41
|
+
async close() {
|
|
42
|
+
await new Promise((resolve, reject) => {
|
|
43
|
+
server.close((error) => error ? reject(error) : resolve(undefined));
|
|
44
|
+
server.closeAllConnections();
|
|
45
|
+
});
|
|
46
|
+
}
|
|
47
|
+
};
|
|
48
|
+
}
|
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
// @ts-check
|
|
2
|
+
import { LAB_CONFIG_SCHEMA, parseLabConfig, runLab, stableProgressKey, verifyRun } from "humanish";
|
|
3
|
+
import { startLocalApp } from "./app.mjs";
|
|
4
|
+
|
|
5
|
+
/** @param {string} appUrl @returns {import("humanish").CuaExecutor} */
|
|
6
|
+
function createAppContractExecutor(appUrl) {
|
|
7
|
+
return {
|
|
8
|
+
async observe() {
|
|
9
|
+
const response = await fetch(new URL("state", appUrl), { signal: AbortSignal.timeout(5000) });
|
|
10
|
+
if (!response.ok) throw new Error(`State read failed: HTTP ${response.status}`);
|
|
11
|
+
const state = await response.json();
|
|
12
|
+
if (typeof state?.greeted !== "boolean" || !Number.isInteger(state?.messages)) {
|
|
13
|
+
throw new Error("Unexpected app state");
|
|
14
|
+
}
|
|
15
|
+
const appState = { greeted: state.greeted, messages: state.messages };
|
|
16
|
+
return { stateSignature: stableProgressKey(appState), appState };
|
|
17
|
+
},
|
|
18
|
+
async execute(action, signal) {
|
|
19
|
+
signal?.throwIfAborted();
|
|
20
|
+
if (action.kind !== "type") throw new Error(`Unsupported app action: ${action.kind}`);
|
|
21
|
+
const response = await fetch(new URL("chat", appUrl), {
|
|
22
|
+
method: "POST",
|
|
23
|
+
body: action.text,
|
|
24
|
+
signal: AbortSignal.any([AbortSignal.timeout(5000), ...(signal ? [signal] : [])])
|
|
25
|
+
});
|
|
26
|
+
if (!response.ok) throw new Error(`Chat write failed: HTTP ${response.status}`);
|
|
27
|
+
}
|
|
28
|
+
};
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
// A deterministic rule, not an AI participant: no model SDK, requests or credentials.
|
|
32
|
+
/** @type {import("humanish").CuaProvider} */
|
|
33
|
+
const provider = {
|
|
34
|
+
id: "local-app-deterministic-example",
|
|
35
|
+
version: "1.0.0",
|
|
36
|
+
requiresFrame: false,
|
|
37
|
+
capabilities: {
|
|
38
|
+
headless: true,
|
|
39
|
+
structuredTrace: true,
|
|
40
|
+
lanes: ["computer-use"],
|
|
41
|
+
producesScreenshots: false,
|
|
42
|
+
byoModel: true,
|
|
43
|
+
preGrantableApprovals: false,
|
|
44
|
+
inProcessTools: false,
|
|
45
|
+
license: "open"
|
|
46
|
+
},
|
|
47
|
+
async nextTurn(request, signal) {
|
|
48
|
+
signal.throwIfAborted();
|
|
49
|
+
const greeted = request.observation.appState?.greeted;
|
|
50
|
+
if (typeof greeted !== "boolean") throw new Error("Missing greeted observation");
|
|
51
|
+
return {
|
|
52
|
+
actions: greeted ? [] : [{ kind: "type", text: "hello there" }],
|
|
53
|
+
pendingSafetyChecks: [],
|
|
54
|
+
done: greeted,
|
|
55
|
+
message: greeted ? "The app accepted the greeting." : "Sending a greeting."
|
|
56
|
+
};
|
|
57
|
+
}
|
|
58
|
+
};
|
|
59
|
+
|
|
60
|
+
const app = await startLocalApp();
|
|
61
|
+
try {
|
|
62
|
+
// parseLabConfig accepts a decoded OBJECT, not a YAML string.
|
|
63
|
+
const parsed = parseLabConfig({
|
|
64
|
+
schema: LAB_CONFIG_SCHEMA,
|
|
65
|
+
id: "state-driven-local-app-example",
|
|
66
|
+
title: "Deterministic local-app integration example",
|
|
67
|
+
subject: { source: "local-app", appUrl: app.appUrl },
|
|
68
|
+
// This registry id selects the CUA loop. buildProvider supplies the actual provider.
|
|
69
|
+
actors: [{ type: "openai-computer-use", persona: "pixel-pat", mission: "Greet the app." }],
|
|
70
|
+
scenario: { mode: "live" },
|
|
71
|
+
execution: { timeoutMs: 15_000 }
|
|
72
|
+
});
|
|
73
|
+
if (!parsed.ok) throw new Error(parsed.error.message);
|
|
74
|
+
const outcome = await runLab(parsed.config, {
|
|
75
|
+
cwd: process.cwd(),
|
|
76
|
+
dryRun: false,
|
|
77
|
+
cuaHooks: {
|
|
78
|
+
buildExecutor: async ({ appUrl }) => createAppContractExecutor(appUrl),
|
|
79
|
+
buildProvider: async () => provider
|
|
80
|
+
}
|
|
81
|
+
});
|
|
82
|
+
if (outcome.backend !== "cua") throw new Error(`Unexpected backend: ${outcome.backend}`);
|
|
83
|
+
const { result } = outcome;
|
|
84
|
+
const verified = await verifyRun(process.cwd(), result.runId);
|
|
85
|
+
console.log(JSON.stringify({
|
|
86
|
+
runId: result.runId,
|
|
87
|
+
ok: result.ok,
|
|
88
|
+
completionReason: result.session?.completionReason,
|
|
89
|
+
provider: provider.id,
|
|
90
|
+
sandboxCreated: result.sandbox !== undefined,
|
|
91
|
+
screenshots: result.session?.screenshots,
|
|
92
|
+
verification: verified,
|
|
93
|
+
app: app.getReceipt(),
|
|
94
|
+
// A mechanism statement, not an inference from missing provider usage/rates.
|
|
95
|
+
costBasis: "No model calls or hosted resources; only this process and loopback HTTP."
|
|
96
|
+
}, null, 2));
|
|
97
|
+
if (!result.ok || !verified.ok || result.session?.completionReason !== "goal_satisfied"
|
|
98
|
+
|| result.sandbox !== undefined || !app.getReceipt().greeted) {
|
|
99
|
+
throw new Error(result.error?.message ?? "Local-app example did not complete and verify");
|
|
100
|
+
}
|
|
101
|
+
} finally {
|
|
102
|
+
await app.close();
|
|
103
|
+
console.log(JSON.stringify({ cleanup: app.getReceipt() }));
|
|
104
|
+
}
|
|
@@ -52,8 +52,9 @@ wait and does not cache a previous action's position.
|
|
|
52
52
|
`redaction.screenshots` resolves to `"n/a"`. No fabricated `Buffer.alloc(0)`
|
|
53
53
|
ever reaches disk. (A vision executor still returns a frame, exactly as before.)
|
|
54
54
|
- **`stateSignature` is still required.** Derive it from your own state — e.g.
|
|
55
|
-
`
|
|
56
|
-
key when `appState` is absent. It is never written
|
|
55
|
+
`stableProgressKey({ route, turn, modal })` (exported from `humanish`). It is the
|
|
56
|
+
canonical fallback progress key when `appState` is absent. It is never written
|
|
57
|
+
to the trace as text.
|
|
57
58
|
- **`appState` is the preferred progress input.** When present, the loop's
|
|
58
59
|
friction / no-progress detection keys off a deterministic, sorted-key,
|
|
59
60
|
depth/length-capped projection of it (`stableProgressKey`) rather than the
|
|
@@ -168,33 +169,25 @@ from your own state, pair it with a non-vision `CuaProvider`, and call
|
|
|
168
169
|
|
|
169
170
|
### 2. `runLab` + `buildExecutor` / `buildProvider` (keeps the composition)
|
|
170
171
|
|
|
171
|
-
The supported library path
|
|
172
|
-
|
|
172
|
+
The supported library path keeps personas, the Observer, the evidence bundle,
|
|
173
|
+
redaction, and the friction loop, while skipping E2B entirely. Start with the
|
|
174
|
+
[complete runnable example](examples/state-driven-local-app/README.md), which
|
|
175
|
+
ships in the npm package:
|
|
173
176
|
|
|
174
|
-
```
|
|
175
|
-
|
|
176
|
-
|
|
177
|
-
// local-app YAML (shareable; fails closed without hooks):
|
|
178
|
-
// schema: humanish.lab.v2
|
|
179
|
-
// id: downstream-local-app-state
|
|
180
|
-
// subject: { source: local-app, appUrl: http://localhost:5173 }
|
|
181
|
-
// actors: [{ type: openai-computer-use, persona: curious-tester, mission: "…" }]
|
|
182
|
-
// scenario: { mode: live }
|
|
183
|
-
const parsed = parseLabConfig(yaml);
|
|
184
|
-
if (!parsed.ok) throw new Error(parsed.error.message);
|
|
185
|
-
|
|
186
|
-
const outcome = await runLab(parsed.config, {
|
|
187
|
-
cwd: process.cwd(),
|
|
188
|
-
dryRun: false,
|
|
189
|
-
cuaHooks: {
|
|
190
|
-
buildExecutor: async ({ appUrl }) => createAppContractExecutor(bridge, appUrl),
|
|
191
|
-
buildProvider: async () => createStateBrain(),
|
|
192
|
-
},
|
|
193
|
-
});
|
|
194
|
-
// outcome.backend === "cua"; outcome.result.sandbox === undefined (NO E2B); the
|
|
195
|
-
// trace's provider id is the injected brain's id.
|
|
177
|
+
```bash
|
|
178
|
+
npm install humanish
|
|
179
|
+
node node_modules/humanish/docs/architecture/examples/state-driven-local-app/runner.mjs
|
|
196
180
|
```
|
|
197
181
|
|
|
182
|
+
The example includes a real loopback app, HTTP state/action bridge, full config
|
|
183
|
+
object (`schema: LAB_CONFIG_SCHEMA`), deterministic non-vision provider,
|
|
184
|
+
`stableProgressKey`, `runLab`, verification, a printed report and `finally`
|
|
185
|
+
cleanup. It makes no model calls and does not demonstrate persona efficacy.
|
|
186
|
+
`parseLabConfig` accepts a decoded object, not a YAML string; narrow its result
|
|
187
|
+
on `.ok`, then narrow `runLab`'s result on `backend === "cua"` before accessing
|
|
188
|
+
the CUA result. The example defines all helpers rather than requiring a consumer
|
|
189
|
+
to reconstruct them.
|
|
190
|
+
|
|
198
191
|
When `cuaHooks.buildExecutor` is set, `runCuaActorLab` takes a branch that NEVER
|
|
199
192
|
loads the E2B module, creates a sandbox, runs `prepareDesktop`, provisions a
|
|
200
193
|
clone, opens a browser, or starts a stream. `sandboxId`/`streamUrl` stay
|
|
@@ -3,7 +3,7 @@
|
|
|
3
3
|
Date: 2026-06-02 (current-state note updated 2026-07-14)
|
|
4
4
|
|
|
5
5
|
Status: reference map for the major contracts shipped through source version
|
|
6
|
-
`0.88.
|
|
6
|
+
`0.88.2`; it is not an exhaustive inventory of command/result envelopes. Exported types,
|
|
7
7
|
schema constants, parsers, and validators in `src/` are authoritative. Rows
|
|
8
8
|
marked "reserved" name layering intent only — no code emits or validates them
|
|
9
9
|
yet. Do not emit a reserved schema.
|
|
@@ -210,7 +210,11 @@ A lab is a composition over code primitives, not a hardcoded kind:
|
|
|
210
210
|
template actually used is recorded in the run bundle as `desktopTemplate`
|
|
211
211
|
(public-safe — a template name is not a secret). Inert (warned) on every route
|
|
212
212
|
that creates no desktop, incl. the in-process `local-app` cua route and the
|
|
213
|
-
meta route — never silently ignored (invariant 6)
|
|
213
|
+
meta route — never silently ignored (invariant 6). Custom images need the
|
|
214
|
+
Desktop SDK's `xdotool` input support and `xclip` or `xsel` on `PATH` for
|
|
215
|
+
clipboard recovery when direct typing fails. If both clipboard utilities are
|
|
216
|
+
absent, recovery fails with `clipboard-utility-missing`; a dry-run does not
|
|
217
|
+
inspect the image's installed tools;
|
|
214
218
|
- `execution.desktop.browser` (e2b-desktop computer-use/fan-out routes, plus
|
|
215
219
|
sequential and concurrent shared-world actor seats): optional browser family
|
|
216
220
|
preference: `default`, `chrome`, `chromium`, or `firefox`. Absent/default
|
|
@@ -509,6 +513,16 @@ shared-world bundle adds TWO additive, optional fields to `humanish.run-bundle.v
|
|
|
509
513
|
|
|
510
514
|
SEQUENTIAL shape (`topologyMode: sequential`, #164 PR1):
|
|
511
515
|
- `sequence: [roleId, …]` — the role ids that actually took a turn, in declared order.
|
|
516
|
+
- `skippedTail` (optional, live sequential only) — `{ afterRoleId, roles,
|
|
517
|
+
cause, maxTotalUsd?, estimatedTotalUsd? }`. Each ordered `roles` entry names
|
|
518
|
+
`{ roleId, simId, streamId }` for an unstarted participant. Together with the
|
|
519
|
+
executed prefix it must account for the full declared denominator. The
|
|
520
|
+
predecessor must have a matching `harness_error`, explicit `session_error`,
|
|
521
|
+
`usage_unreported`, or measured `study_spend_limit`. Only the last cause
|
|
522
|
+
carries budget figures, using the same per-participant estimates as the
|
|
523
|
+
tracker. Blocked seats need matching simulation, stream and blocked-event
|
|
524
|
+
evidence, with no actor, trace, screenshot or invented timeline turn.
|
|
525
|
+
Historical bundles without this field still require every role in the timeline.
|
|
512
526
|
- `timeline: (checkpoint | turn)[]` — a harness-clocked, strictly alternating
|
|
513
527
|
timeline that starts `cp-baseline`, alternates checkpoint → turn → checkpoint,
|
|
514
528
|
and ends on a checkpoint:
|
|
@@ -1063,8 +1077,9 @@ request's worst-case cost. In-flight requests, retries, concurrent lanes, and
|
|
|
1063
1077
|
unreported usage can exceed or escape these estimates. These thresholds are
|
|
1064
1078
|
not hard provider billing caps and exclude desktop and target-app charges.
|
|
1065
1079
|
|
|
1066
|
-
|
|
1067
|
-
|
|
1080
|
+
Crossing `maxUsd` ends `budget_reached` / `incomplete`, including when no
|
|
1081
|
+
material action has executed. The recorded reason distinguishes prior progress
|
|
1082
|
+
from no material progress; a harness budget stop is not participant abandonment.
|
|
1068
1083
|
Crossing the shared study threshold ends `budget_reached` / `incomplete`, with
|
|
1069
1084
|
sibling lanes stopping when their next post-response check sees it. Reaching a
|
|
1070
1085
|
threshold is not proof of task completion.
|
|
@@ -1077,6 +1092,41 @@ threshold on a model `src/pricing.ts` cannot price is refused at preflight
|
|
|
1077
1092
|
(`HUMANISH_CUA_LAB_UNPRICED_CAP`) before sandbox allocation. This rate-availability
|
|
1078
1093
|
check is separate from the post-response spend check.
|
|
1079
1094
|
|
|
1095
|
+
Sequential shared-world studies (`subject.topology: shared-world` with
|
|
1096
|
+
`execution.concurrency: 1`, using clone or local-tree subjects) enforce these
|
|
1097
|
+
same per-participant and shared model thresholds. Final reported usage,
|
|
1098
|
+
including a closing request, is reconciled before admitting the next participant.
|
|
1099
|
+
After the aggregate threshold is crossed, later participants are `blocked` with
|
|
1100
|
+
a recorded skip reason; they make no model requests and add no executed turn to
|
|
1101
|
+
the checkpoint timeline. An unpriced model with a declared threshold fails
|
|
1102
|
+
before allocation with `HUMANISH_SHARED_WORLD_LAB_INVALID`.
|
|
1103
|
+
|
|
1104
|
+
On this sequential route, an otherwise completed capped interaction that returns
|
|
1105
|
+
missing or partial usage stops with `harness_error`, `stopCause: usage_unreported`, and the label
|
|
1106
|
+
“provider usage unavailable.” It is not recorded as a crossed threshold. A
|
|
1107
|
+
stalled or failed request with unknown spend is not retried by the CUA loop.
|
|
1108
|
+
The default OpenAI adapter also disables HTTP and policy-negotiation retries for
|
|
1109
|
+
these strict capped sessions. The loop cancels its owned request signal when a
|
|
1110
|
+
request ends or its timeout wins; injected providers must honor cancellation
|
|
1111
|
+
and remain responsible for their own internal dispatch. Known usage remains in the
|
|
1112
|
+
trace alongside an explicit unknown; subsequent participants do not start when
|
|
1113
|
+
the shared budget cannot be established. Reported zero input and output counts
|
|
1114
|
+
remain valid zero usage. This stricter unknown-usage policy is specific to
|
|
1115
|
+
sequential capped studies; other routes retain their existing behavior. An
|
|
1116
|
+
explicitly incomplete provider response retains its original interruption cause
|
|
1117
|
+
first, with any missing usage still recorded as unknown.
|
|
1118
|
+
|
|
1119
|
+
Sequential traces persist dated model estimates. Their run cost summary marks
|
|
1120
|
+
desktop compute as unmeasured, so the displayed model subtotal is a lower bound.
|
|
1121
|
+
The sequential route does not provide a running Observer usage stream; its final
|
|
1122
|
+
CLI and Observer projections read these persisted estimates.
|
|
1123
|
+
|
|
1124
|
+
A capped custom session must return the declared model identity on its trace.
|
|
1125
|
+
A mismatch fails orchestration and blocks later participants while preserving
|
|
1126
|
+
the participant's original outcome and the estimate for its returned model.
|
|
1127
|
+
This check does not establish which model an arbitrary custom runner actually
|
|
1128
|
+
called or retrospectively enforce a runner that ignored its cap options.
|
|
1129
|
+
|
|
1080
1130
|
These computer-use rules do not replace the terminal route's separate
|
|
1081
1131
|
`scenario.caps` cost-ledger and product-spend rules described above.
|
|
1082
1132
|
|
|
@@ -1228,6 +1278,16 @@ the stream shape, the meaningful-use rubric, and hard-failure rules.
|
|
|
1228
1278
|
Review summarizes whether evidence supports the claim. It does not replace
|
|
1229
1279
|
verification or maintainer acceptance.
|
|
1230
1280
|
|
|
1281
|
+
`verdict` is the run gate result. `participants.reachedGoal` retains the count of
|
|
1282
|
+
recorded successful sessions; it is not an independent adjudication of their
|
|
1283
|
+
claims. Computer-use review and Observer labels distinguish a participant's
|
|
1284
|
+
reported completion from a recorded `stopWhen` or dwell condition match. A
|
|
1285
|
+
condition match establishes that condition, not every aspect of the mission.
|
|
1286
|
+
Missing or malformed completion evidence is labeled unavailable. Other actor
|
|
1287
|
+
routes retain their own completion semantics. Re-reading a historical review
|
|
1288
|
+
refreshes these labels without rewriting the original bundle or actor trace.
|
|
1289
|
+
Share-safety verification does not establish task success.
|
|
1290
|
+
|
|
1231
1291
|
Core-owned fields:
|
|
1232
1292
|
|
|
1233
1293
|
- `schema`
|
package/docs/goals/current.md
CHANGED
|
@@ -1,9 +1,9 @@
|
|
|
1
1
|
# Current Goals
|
|
2
2
|
|
|
3
|
-
Status date: 2026-09-
|
|
3
|
+
Status date: 2026-09-13. Published baseline: `0.88.2`.
|
|
4
4
|
|
|
5
5
|
This page guides work on current merged source. Published behavior is described
|
|
6
|
-
in the [release notes](../release/0.88.
|
|
6
|
+
in the [release notes](../release/0.88.2-sequential-study-budgets.md).
|
|
7
7
|
The [September 9 history](https://github.com/danielgwilson/humanish/blob/main/docs/goals/current-history-2026-09-09.md)
|
|
8
8
|
preserves the former status log; its queues do not supersede this page.
|
|
9
9
|
|
|
@@ -88,7 +88,7 @@ requires decision-equivalent retained evidence and a real deletion branch.
|
|
|
88
88
|
No first-party deletion branch has met that gate. Public demonstrations do not
|
|
89
89
|
substitute for it.
|
|
90
90
|
|
|
91
|
-
## Current Program Truth (source `0.88.
|
|
91
|
+
## Current Program Truth (source `0.88.2`)
|
|
92
92
|
|
|
93
93
|
| Surface | Available in merged source | Remaining boundary |
|
|
94
94
|
| --- | --- | --- |
|
|
@@ -98,7 +98,7 @@ substitute for it.
|
|
|
98
98
|
| Task protocol | Hidden criteria and per-task outcomes on supported per-lane CUA paths, including local-agent and desktop-cli | Shared-world, terminal-product, scripted and synthetic routes reject `tasks` before execution |
|
|
99
99
|
| Shared state | Sequential and concurrent single-origin shared-world studies with retained evidence | Multi-origin implementation remains gated; concurrent state change does not establish per-action causation |
|
|
100
100
|
| Observer | Live/recorded views, participant assignments, action-specific links, saved moments, zoom, comparison and phone-width review | Sparse captures cannot prove every action's effect; visual comparison alone is not a controlled experiment |
|
|
101
|
-
| Review and feedback | Verification grades, feedback drafts, portable HTML
|
|
101
|
+
| Review and feedback | Verification grades, feedback drafts, portable HTML, redacted bundle derivatives and computer-use completion-source labels | Sharing requires the appropriate grade; participant reports and condition matches still need task adjudication |
|
|
102
102
|
| TUI and serving | Detached starts, run stopping, reclamation, Observer attachment, loopback serving and run library | Stopping a process does not itself prove sandbox cleanup; TUI views over CLI `stats`/`export` remain follow-ups |
|
|
103
103
|
| Off-app communication | In-sandbox email/SMS catch and digest-only thread evidence | This does not establish real-provider delivery |
|
|
104
104
|
| Mobile and media | Hosted viewport/emulation, desktop geometry checks, bounded dwell and declared camera feed | Physical-device and touch fidelity remain unproven; unsupported microphone declarations are rejected |
|
|
@@ -108,6 +108,11 @@ Use the [task support matrix](../architecture/task-protocol-support.md),
|
|
|
108
108
|
and [CLI reference](https://humanish.dev/docs/cli) when choosing a concrete path.
|
|
109
109
|
Source behavior and required tests outrank stale status prose.
|
|
110
110
|
|
|
111
|
+
The library-assisted `local-app` route now includes a
|
|
112
|
+
[runnable npm example](../architecture/examples/state-driven-local-app/README.md).
|
|
113
|
+
Its deterministic provider demonstrates the integration with a real loopback
|
|
114
|
+
app; it does not establish persona effectiveness or independent adoption.
|
|
115
|
+
|
|
111
116
|
## Gates And Deferred Work
|
|
112
117
|
|
|
113
118
|
- Live OSS meta-lab execution remains disabled until repository-derived
|
package/docs/ramp/README.md
CHANGED
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
Status: public-safe contributor and agent ramp.
|
|
4
4
|
|
|
5
|
-
Package/source version in this tree: `0.88.
|
|
5
|
+
Package/source version in this tree: `0.88.2` (2026-09-13). The Observer is phone-usable as a stated requirement (observer/AGENTS.md); interactive primitives start from Base UI. The Observer renderer is the observer/ workspace artifact only; the legacy string-concat renderer was deleted at cutover (#426), and rollback is a version pin to 0.42.0. The containment boundary introduced in
|
|
6
6
|
`0.15.1` remains in force: managed run and output paths bind to validated
|
|
7
7
|
physical filesystem identities, and stored provider IDs are evidence, not
|
|
8
8
|
cleanup authority. The bundled OSS meta-lab is dry-run only until
|
|
@@ -47,6 +47,16 @@ If a change does not improve one of those loops, it probably belongs elsewhere.
|
|
|
47
47
|
|
|
48
48
|
## Current State
|
|
49
49
|
|
|
50
|
+
The [0.88.2 release note](../release/0.88.2-sequential-study-budgets.md)
|
|
51
|
+
describes model-spend thresholds on sequential shared-world studies, blocked
|
|
52
|
+
later participants, and explicit unknown-usage accounting.
|
|
53
|
+
|
|
54
|
+
The [0.88.1 release note](../release/0.88.1-completion-evidence-and-local-app.md)
|
|
55
|
+
describes computer-use labels that distinguish participant reports from
|
|
56
|
+
recorded condition matches. It also covers the complete npm local-app example
|
|
57
|
+
and public `stableProgressKey` export, plus the clipboard fallback correction
|
|
58
|
+
for inherited output pipes.
|
|
59
|
+
|
|
50
60
|
The [0.88.0 release note](../release/0.88.0-study-diagnostics.md) describes
|
|
51
61
|
computer-use CLI diagnostics, explicit local admission limits and retained
|
|
52
62
|
uncertainty when an earlier provider request did not report usage.
|
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
# Humanish 0.88.1: distinguish reported completion from condition matches
|
|
2
|
+
|
|
3
|
+
When a computer-use participant says it finished, review and Observer now
|
|
4
|
+
label that completion as **participant-reported**. A recorded `stopWhen` match
|
|
5
|
+
or completed dwell window identifies a **recorded completion condition** instead.
|
|
6
|
+
Missing, malformed or conflicting completion evidence is labeled unavailable.
|
|
7
|
+
Zero-completion counts use **0/N recorded completions**, and aggregate stats use
|
|
8
|
+
**recorded goal completions**.
|
|
9
|
+
|
|
10
|
+
A matched condition establishes only that condition, not every aspect of the
|
|
11
|
+
mission. The run gate and share-safety verification remain separate from task
|
|
12
|
+
adjudication. This release adds no automatic semantic evaluator.
|
|
13
|
+
|
|
14
|
+
Existing actor statuses, `participants.reachedGoal`, verdict values and
|
|
15
|
+
denominators stay unchanged. Current review commands, Observer rendering and
|
|
16
|
+
newly generated feedback drafts apply these labels to older runs while
|
|
17
|
+
preserving original bundles and actor traces. Existing HTML exports keep their
|
|
18
|
+
original renderer. Other actor routes retain their own completion semantics.
|
|
19
|
+
|
|
20
|
+
## Run a local app from npm
|
|
21
|
+
|
|
22
|
+
To connect your local app's state and actions to Humanish, start with the
|
|
23
|
+
[complete npm example](../architecture/examples/state-driven-local-app/README.md):
|
|
24
|
+
|
|
25
|
+
```bash
|
|
26
|
+
npm install humanish@0.88.1
|
|
27
|
+
node node_modules/humanish/docs/architecture/examples/state-driven-local-app/runner.mjs
|
|
28
|
+
```
|
|
29
|
+
|
|
30
|
+
It starts a synthetic loopback app, reads its state, sends a greeting through
|
|
31
|
+
its HTTP action endpoint and verifies the recorded run. A `finally` block closes
|
|
32
|
+
the app. The provider follows a deterministic rule, so the example needs no
|
|
33
|
+
model credentials or E2B desktop. Use Node.js 20.3 or later.
|
|
34
|
+
|
|
35
|
+
The runner supplies both `CuaExecutor` and `CuaProvider`, passes a decoded object
|
|
36
|
+
to `parseLabConfig`, and checks the config and backend discriminants. The
|
|
37
|
+
existing `stableProgressKey` utility is now exported from `humanish`, so callers
|
|
38
|
+
can use the same bounded state projection as the loop. Replace the two ports
|
|
39
|
+
with your app bridge and provider; the example guide explains the responsibilities
|
|
40
|
+
that remain with them.
|
|
41
|
+
|
|
42
|
+
## Clipboard fallback returns after the write
|
|
43
|
+
|
|
44
|
+
When direct desktop typing fails, Humanish can write the text to the X clipboard
|
|
45
|
+
and paste it. `xclip` and `xsel` fork a process to keep the selection available;
|
|
46
|
+
inherited output pipes could leave the command runner waiting until timeout.
|
|
47
|
+
Both clipboard-write commands now detach their output while preserving the
|
|
48
|
+
existing exit checks, temporary-file cleanup and paste dispatch.
|
|
49
|
+
|
|
50
|
+
Custom desktop images still need a working `xclip` or `xsel`. Missing utilities
|
|
51
|
+
continue to report `clipboard-utility-missing`. This patch adds no typing retry,
|
|
52
|
+
and recovery after a primary write inserted an unknown partial prefix remains
|
|
53
|
+
unproven. [Issue #340](https://github.com/danielgwilson/humanish/issues/340) remains
|
|
54
|
+
open for that broader typing problem.
|
|
55
|
+
|
|
56
|
+
## Verification and limits
|
|
57
|
+
|
|
58
|
+
[Completion label checks](https://github.com/danielgwilson/humanish/pull/765)
|
|
59
|
+
cover participant reports, recorded condition matches, zero completions and
|
|
60
|
+
unavailable legacy detail. They preserve recorded counts and statuses while
|
|
61
|
+
refreshing historical review and feedback projections. Existing adapter
|
|
62
|
+
narrative remains available.
|
|
63
|
+
|
|
64
|
+
[The packaged example](https://github.com/danielgwilson/humanish/pull/762) completed
|
|
65
|
+
two independent local runs on Node 20.20.2 and 24.12.0. Each made two HTTP state
|
|
66
|
+
reads and one state-changing request, reached `goal_satisfied`, verified as
|
|
67
|
+
`share_ready` and closed its server. These checks used locally packed candidate
|
|
68
|
+
source labeled 0.88.0, distinct from the registry release with that version.
|
|
69
|
+
|
|
70
|
+
[The clipboard correction](https://github.com/danielgwilson/humanish/pull/761)
|
|
71
|
+
passed two real desktop conformance cases, one with explicitly installed `xclip`
|
|
72
|
+
and one with `xsel`. Each preserved the exact Unicode/newline/quoted text with
|
|
73
|
+
one paste and removed its transfer file. Both cases refused the primary write
|
|
74
|
+
before it could type anything. They do not establish default-image clipboard
|
|
75
|
+
availability or recovery from partial typing.
|
|
76
|
+
|
|
77
|
+
Both checks were model-free. They establish the tested integration and executor
|
|
78
|
+
behavior; persona effectiveness and independent maintainer adoption remain
|
|
79
|
+
separate questions.
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
# Humanish 0.88.2: sequential studies honor model-spend thresholds
|
|
2
|
+
|
|
3
|
+
Sequential shared-world studies now enforce the `execution.caps.maxUsd` and
|
|
4
|
+
`maxTotalUsd` thresholds they previously accepted without applying. This covers
|
|
5
|
+
computer-use participants sharing a clone or local-tree subject with
|
|
6
|
+
`subject.topology: shared-world` and `execution.concurrency: 1`.
|
|
7
|
+
|
|
8
|
+
Each participant's reported usage feeds the per-participant threshold and the
|
|
9
|
+
shared study estimate. Final usage, including a closing report, is reconciled
|
|
10
|
+
before the next participant starts. A participant interrupted by a threshold
|
|
11
|
+
has `budget_reached` / `incomplete`; its reason distinguishes prior activity
|
|
12
|
+
from no material progress. Later participants blocked by the shared threshold
|
|
13
|
+
make no model requests and add no executed turn to the checkpoint timeline.
|
|
14
|
+
A closing report that crosses the threshold preserves the already recorded
|
|
15
|
+
completion condition while blocking subsequent participants.
|
|
16
|
+
The bundle retains all declared participants and an explicit blocked suffix.
|
|
17
|
+
Verification checks the suffix against its preceding interruption, participant
|
|
18
|
+
records and unchanged executed timeline; absent participants cannot masquerade
|
|
19
|
+
as budget-blocked seats.
|
|
20
|
+
|
|
21
|
+
These checks happen after a model response and before its actions or another
|
|
22
|
+
participant turn. The current request can overshoot a threshold. They estimate
|
|
23
|
+
model spend; they do not reserve future requests or cap provider invoices,
|
|
24
|
+
desktop compute, or target-app charges. A zero threshold can still allow the
|
|
25
|
+
first model request. Use an explicit dry-run for a path without provider calls.
|
|
26
|
+
|
|
27
|
+
An unpriced model with a declared threshold is refused before allocation. For
|
|
28
|
+
an otherwise completed response, missing or partial usage ends capped sequential
|
|
29
|
+
execution with `harness_error` and `usage_unreported` (“provider usage unavailable”).
|
|
30
|
+
The CUA loop sends no further participant request, retry or closing request.
|
|
31
|
+
An explicitly incomplete provider response retains its original interruption
|
|
32
|
+
cause first; its missing usage remains unknown. For these strict capped sessions,
|
|
33
|
+
the default OpenAI adapter makes one dispatch per requested turn, including
|
|
34
|
+
HTTP errors and policy negotiation. The loop cancels the request signal when
|
|
35
|
+
its timeout wins, so an outstanding transport cannot retry after the loop ends.
|
|
36
|
+
Injected providers must honor that signal and control their own dispatches;
|
|
37
|
+
cancellation cannot undo an already billed request. Known partial costs stay
|
|
38
|
+
in the trace, and an unknown shared budget blocks later participants.
|
|
39
|
+
|
|
40
|
+
Sequential traces now retain dated model estimates. The run total explicitly
|
|
41
|
+
marks desktop compute as unmeasured, so the model subtotal is a lower bound.
|
|
42
|
+
Observer labels a partial estimate as known cost with the total unknown.
|
|
43
|
+
A custom session returning a different model identity fails orchestration and
|
|
44
|
+
blocks later participants while preserving its original outcome and the
|
|
45
|
+
estimate for its returned model. This does not retrospectively enforce an
|
|
46
|
+
arbitrary custom runner that ignored its cap options.
|
|
47
|
+
|
|
48
|
+
Existing bundles are not rewritten. Uncapped sessions and other execution
|
|
49
|
+
routes retain their existing behavior. The stricter unknown-usage rule applies
|
|
50
|
+
to sequential capped studies; the sequential route still has no running
|
|
51
|
+
Observer usage stream.
|
|
52
|
+
|
|
53
|
+
Verification covers the actual participant loop with captured provider usage,
|
|
54
|
+
individual and shared thresholds, unstarted seats, zero and missing usage,
|
|
55
|
+
closing requests, model mismatches, checkpoint evidence and cleanup. These
|
|
56
|
+
contract checks establish the tested behavior; they do not establish persona
|
|
57
|
+
effectiveness, independent adoption, or exact provider billing.
|
package/package.json
CHANGED