humanish 0.39.0 → 0.41.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +5 -2
- package/dist/actor-contract.d.ts +24 -1
- package/dist/actor-contract.js +4 -0
- package/dist/actor-contract.js.map +1 -1
- package/dist/claude-agent-sdk.js +4 -0
- package/dist/claude-agent-sdk.js.map +1 -1
- package/dist/comms-sandbox-catch.d.ts +8 -1
- package/dist/comms-sandbox-catch.js +128 -1
- package/dist/comms-sandbox-catch.js.map +1 -1
- package/dist/computer-use-actor.d.ts +1 -1
- package/dist/computer-use.d.ts +13 -1
- package/dist/computer-use.js +100 -14
- package/dist/computer-use.js.map +1 -1
- package/dist/concurrent-shared-world-lab.js +1 -1
- package/dist/concurrent-shared-world-lab.js.map +1 -1
- package/dist/cua-actor-lab.js +114 -9
- package/dist/cua-actor-lab.js.map +1 -1
- package/dist/e2b-desktop-executor.d.ts +21 -6
- package/dist/e2b-desktop-executor.js +65 -21
- package/dist/e2b-desktop-executor.js.map +1 -1
- package/dist/init-templates.js +4 -1
- package/dist/init-templates.js.map +1 -1
- package/dist/lab-config.d.ts +48 -1
- package/dist/lab-config.js +130 -2
- package/dist/lab-config.js.map +1 -1
- package/dist/observer-assets.js +8 -1
- package/dist/observer-assets.js.map +1 -1
- package/dist/observer-data.d.ts +12 -0
- package/dist/observer-data.js +13 -1
- package/dist/observer-data.js.map +1 -1
- package/dist/openai-responses-cu.js +10 -1
- package/dist/openai-responses-cu.js.map +1 -1
- package/dist/orientation.d.ts +29 -0
- package/dist/orientation.js +95 -0
- package/dist/orientation.js.map +1 -0
- package/dist/pi-agent-core.js +4 -0
- package/dist/pi-agent-core.js.map +1 -1
- package/dist/pricing.d.ts +8 -0
- package/dist/pricing.js +22 -6
- package/dist/pricing.js.map +1 -1
- package/dist/program.js +13 -0
- package/dist/program.js.map +1 -1
- package/dist/run.d.ts +55 -2
- package/dist/run.js +74 -1
- package/dist/run.js.map +1 -1
- package/dist/scripted-browser-lab.js +1 -1
- package/dist/scripted-browser-lab.js.map +1 -1
- package/dist/shared-world-lab.js +1 -1
- package/dist/shared-world-lab.js.map +1 -1
- package/dist/subject-runtime.d.ts +23 -0
- package/dist/subject-runtime.js +70 -0
- package/dist/subject-runtime.js.map +1 -0
- package/dist/tasks.d.ts +77 -0
- package/dist/tasks.js +101 -0
- package/dist/tasks.js.map +1 -0
- package/docs/contracts/schemas.md +1 -1
- package/docs/goals/current.md +2 -2
- package/docs/principles/three-roles.md +68 -0
- package/docs/ramp/README.md +1 -1
- package/package.json +1 -1
package/dist/run.d.ts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { type CodexAppServerTrace } from "./codex-app-server.js";
|
|
2
|
-
import { type ActorTrace } from "./actor-contract.js";
|
|
2
|
+
import { type ActorStatus, type ActorTrace } from "./actor-contract.js";
|
|
3
3
|
import { type CapturedGitState } from "./core/git-state.js";
|
|
4
4
|
import type { E2BDesktopModule } from "./e2b-desktop-launch.js";
|
|
5
5
|
import { type PreparedRunArtifactPaths } from "./run-paths.js";
|
|
@@ -22,7 +22,7 @@ export interface RunOptions {
|
|
|
22
22
|
timeoutMs?: number;
|
|
23
23
|
}
|
|
24
24
|
export type RunStreamKind = "ui" | "browser" | "terminal" | "tui" | "codex-ui" | "artifact" | "summary";
|
|
25
|
-
export type RunSimulationStatus = "queued" | "preparing" | "running" | "passed" | "complete" | "blocked" | "timed_out" | "failed" | "contract_proof_only";
|
|
25
|
+
export type RunSimulationStatus = "queued" | "preparing" | "running" | "passed" | "abandoned" | "incomplete" | "complete" | "blocked" | "timed_out" | "failed" | "contract_proof_only";
|
|
26
26
|
export interface RunStreamCompletion {
|
|
27
27
|
actorLogPath?: string;
|
|
28
28
|
actorLogTail?: string;
|
|
@@ -807,12 +807,65 @@ export interface RunRerunLineage {
|
|
|
807
807
|
completionReason?: string;
|
|
808
808
|
}>;
|
|
809
809
|
}
|
|
810
|
+
/**
|
|
811
|
+
* What happened to the PARTICIPANTS in a study, with the denominator attached.
|
|
812
|
+
*
|
|
813
|
+
* A stakeholder watching through the glass forms conclusions from vivid moments — that is the
|
|
814
|
+
* classic failure of the viewing room, and it is why researchers synthesize rather than letting the
|
|
815
|
+
* room decide. So anything shown to a stakeholder carries its count, or it becomes a machine for
|
|
816
|
+
* manufacturing certainty from n=1 (docs/principles/three-roles.md).
|
|
817
|
+
*
|
|
818
|
+
* These are OUTCOMES, not scores. `abandoned` is the most valuable thing a usability study
|
|
819
|
+
* produces, and `harnessFailed` is the only member that says the instrument, rather than the
|
|
820
|
+
* product, is what went wrong.
|
|
821
|
+
*/
|
|
822
|
+
export interface ParticipantOutcomes {
|
|
823
|
+
/** Participants whose sessions reached a terminal state — the denominator for every count below. */
|
|
824
|
+
total: number;
|
|
825
|
+
/** Reached the goal. */
|
|
826
|
+
reachedGoal: number;
|
|
827
|
+
/** Stopped trying. A finding about the product. */
|
|
828
|
+
abandoned: number;
|
|
829
|
+
/** Ran out of session or budget before reaching the goal. */
|
|
830
|
+
ranOut: number;
|
|
831
|
+
/** Needed an approval the run could not give. */
|
|
832
|
+
blocked: number;
|
|
833
|
+
/** The harness failed them: a dead sandbox, a provider error, a broken artifact. */
|
|
834
|
+
harnessFailed: number;
|
|
835
|
+
/**
|
|
836
|
+
* Participants who reported friction or a defect on the way, whatever their outcome.
|
|
837
|
+
*
|
|
838
|
+
* This is NOT a failure count and it overlaps the others on purpose — someone can reach the goal
|
|
839
|
+
* and still tell you the road there was broken. A live two-persona run made the case: both
|
|
840
|
+
* participants signed in, so "2/2 reached the goal" was true, and the keyboard-first one also
|
|
841
|
+
* reported that the signature step could not be completed without a mouse. Reporting only the
|
|
842
|
+
* outcome would have buried the single most useful thing that run produced.
|
|
843
|
+
*/
|
|
844
|
+
reportedFriction: number;
|
|
845
|
+
}
|
|
810
846
|
export interface ReviewSummary {
|
|
811
847
|
schema: typeof REVIEW_SCHEMA;
|
|
812
848
|
verdict: "contract_proof_only" | "pass" | "fail" | "blocked" | "timed_out";
|
|
813
849
|
summary: string;
|
|
814
850
|
gaps: string[];
|
|
851
|
+
/**
|
|
852
|
+
* The study result, separate from the verdict above.
|
|
853
|
+
*
|
|
854
|
+
* `verdict` answers a gate-shaped question and has to collapse a run to one word. This answers
|
|
855
|
+
* the research question — what happened to the people in the study — and does not collapse: a run
|
|
856
|
+
* where two of three participants finished is not usefully "fail", and a run where the harness
|
|
857
|
+
* broke is a different thing from one where a persona gave up. Absent on a dry-run contract
|
|
858
|
+
* bundle, which has no participants.
|
|
859
|
+
*/
|
|
860
|
+
participants?: ParticipantOutcomes;
|
|
815
861
|
}
|
|
862
|
+
/** Tally participant outcomes from actor statuses. Statuses this does not recognise are counted in
|
|
863
|
+
* `total` but nowhere else, so the parts can never exceed the whole. */
|
|
864
|
+
export declare function tallyParticipantOutcomes(statuses: readonly ActorStatus[],
|
|
865
|
+
/** Per-participant: did this one report friction or a defect? Same order as `statuses`. */
|
|
866
|
+
reportedFriction?: readonly boolean[]): ParticipantOutcomes;
|
|
867
|
+
/** One line a stakeholder can read, with the denominator attached to every number. */
|
|
868
|
+
export declare function formatParticipantOutcomes(outcomes: ParticipantOutcomes): string;
|
|
816
869
|
export declare function buildRunSource(args: {
|
|
817
870
|
cwd: string;
|
|
818
871
|
capturedAt?: Date | string;
|
package/dist/run.js
CHANGED
|
@@ -49,6 +49,53 @@ const SAFE_GIT_NOTES = new Set([
|
|
|
49
49
|
"public-safe synthetic fixture",
|
|
50
50
|
"public-safe synthetic OSS meta-lab fixture"
|
|
51
51
|
]);
|
|
52
|
+
/** Tally participant outcomes from actor statuses. Statuses this does not recognise are counted in
|
|
53
|
+
* `total` but nowhere else, so the parts can never exceed the whole. */
|
|
54
|
+
export function tallyParticipantOutcomes(statuses,
|
|
55
|
+
/** Per-participant: did this one report friction or a defect? Same order as `statuses`. */
|
|
56
|
+
reportedFriction = []) {
|
|
57
|
+
const tally = {
|
|
58
|
+
total: statuses.length,
|
|
59
|
+
reachedGoal: 0,
|
|
60
|
+
abandoned: 0,
|
|
61
|
+
ranOut: 0,
|
|
62
|
+
blocked: 0,
|
|
63
|
+
harnessFailed: 0,
|
|
64
|
+
reportedFriction: reportedFriction.filter(Boolean).length
|
|
65
|
+
};
|
|
66
|
+
for (const status of statuses) {
|
|
67
|
+
if (status === "passed")
|
|
68
|
+
tally.reachedGoal += 1;
|
|
69
|
+
else if (status === "abandoned")
|
|
70
|
+
tally.abandoned += 1;
|
|
71
|
+
else if (status === "incomplete" || status === "timed_out")
|
|
72
|
+
tally.ranOut += 1;
|
|
73
|
+
else if (status === "blocked")
|
|
74
|
+
tally.blocked += 1;
|
|
75
|
+
else if (status === "failed")
|
|
76
|
+
tally.harnessFailed += 1;
|
|
77
|
+
}
|
|
78
|
+
return tally;
|
|
79
|
+
}
|
|
80
|
+
/** One line a stakeholder can read, with the denominator attached to every number. */
|
|
81
|
+
export function formatParticipantOutcomes(outcomes) {
|
|
82
|
+
if (outcomes.total === 0)
|
|
83
|
+
return "no participants reached a terminal state";
|
|
84
|
+
const parts = [`${outcomes.reachedGoal}/${outcomes.total} reached the goal`];
|
|
85
|
+
if (outcomes.abandoned > 0)
|
|
86
|
+
parts.push(`${outcomes.abandoned} gave up`);
|
|
87
|
+
if (outcomes.ranOut > 0)
|
|
88
|
+
parts.push(`${outcomes.ranOut} ran out of session`);
|
|
89
|
+
if (outcomes.blocked > 0)
|
|
90
|
+
parts.push(`${outcomes.blocked} blocked on an approval`);
|
|
91
|
+
if (outcomes.harnessFailed > 0)
|
|
92
|
+
parts.push(`${outcomes.harnessFailed} lost to a harness failure`);
|
|
93
|
+
// Last, and separate, because it cuts across the outcomes rather than partitioning them: someone
|
|
94
|
+
// can reach the goal and still have found the road there broken.
|
|
95
|
+
if (outcomes.reportedFriction > 0)
|
|
96
|
+
parts.push(`${outcomes.reportedFriction} reported friction`);
|
|
97
|
+
return parts.join(", ");
|
|
98
|
+
}
|
|
52
99
|
export async function buildRunSource(args) {
|
|
53
100
|
const gitOptions = args.capturedAt === undefined ? {} : { capturedAt: args.capturedAt };
|
|
54
101
|
return {
|
|
@@ -2869,7 +2916,29 @@ export async function doctor(cwdInput) {
|
|
|
2869
2916
|
name: "runtime ignore",
|
|
2870
2917
|
ok: await safeCheck(async () => (await readImplicitProjectFile(projectRoot, ".gitignore"))?.includes(".humanish/") ?? false),
|
|
2871
2918
|
message: ".gitignore safely contains .humanish/"
|
|
2872
|
-
}
|
|
2919
|
+
},
|
|
2920
|
+
// The optional peer dep every live browser and terminal lane needs (#346). `npx -y humanish`
|
|
2921
|
+
// does not pull optional peers, so an adopter's FIRST live run used to fail on it — safely and
|
|
2922
|
+
// at $0, but as a burned first impression on the flagship path. Answering it here means the
|
|
2923
|
+
// readiness command actually answers readiness.
|
|
2924
|
+
await (async () => {
|
|
2925
|
+
const present = await safeCheck(async () => {
|
|
2926
|
+
try {
|
|
2927
|
+
await import("@e2b/desktop");
|
|
2928
|
+
return true;
|
|
2929
|
+
}
|
|
2930
|
+
catch {
|
|
2931
|
+
return false;
|
|
2932
|
+
}
|
|
2933
|
+
});
|
|
2934
|
+
return {
|
|
2935
|
+
name: "e2b desktop sdk",
|
|
2936
|
+
ok: present,
|
|
2937
|
+
message: present
|
|
2938
|
+
? "optional peer @e2b/desktop is installed; live desktop lanes can launch"
|
|
2939
|
+
: "optional peer @e2b/desktop is NOT installed — dry runs work, but any live desktop lane will fail closed. Install it with `npm i -D @e2b/desktop`."
|
|
2940
|
+
};
|
|
2941
|
+
})()
|
|
2873
2942
|
];
|
|
2874
2943
|
return {
|
|
2875
2944
|
schema: DOCTOR_SCHEMA,
|
|
@@ -5175,6 +5244,10 @@ function isRunSimulationStatus(value) {
|
|
|
5175
5244
|
|| value === "preparing"
|
|
5176
5245
|
|| value === "running"
|
|
5177
5246
|
|| value === "passed"
|
|
5247
|
+
// Participant outcomes (docs/principles/three-roles.md). This runtime allowlist is the actual
|
|
5248
|
+
// gate — the TS union alone does not validate a bundle read back from disk.
|
|
5249
|
+
|| value === "abandoned"
|
|
5250
|
+
|| value === "incomplete"
|
|
5178
5251
|
|| value === "complete"
|
|
5179
5252
|
|| value === "blocked"
|
|
5180
5253
|
|| value === "timed_out"
|