humanish 0.88.1 → 0.89.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +8 -0
- package/dist/actor-contract.d.ts +1 -1
- package/dist/actor-stop-cause.js +1 -0
- package/dist/actor-stop-cause.js.map +1 -1
- package/dist/computer-use-actor.d.ts +2 -0
- package/dist/computer-use-actor.js +6 -3
- package/dist/computer-use-actor.js.map +1 -1
- package/dist/computer-use.d.ts +5 -2
- package/dist/computer-use.js +38 -3
- package/dist/computer-use.js.map +1 -1
- package/dist/cua-actor-lab.js +3 -1
- package/dist/cua-actor-lab.js.map +1 -1
- package/dist/cua-diagnostics.d.ts +1 -1
- package/dist/cua-diagnostics.js +1 -1
- package/dist/cua-diagnostics.js.map +1 -1
- package/dist/export-bundle.js +6 -0
- package/dist/export-bundle.js.map +1 -1
- package/dist/export.js +67 -10
- package/dist/export.js.map +1 -1
- package/dist/feedback.d.ts +10 -0
- package/dist/feedback.js +58 -1
- package/dist/feedback.js.map +1 -1
- package/dist/index.d.ts +5 -1
- package/dist/index.js +2 -0
- package/dist/index.js.map +1 -1
- package/dist/lab-config.d.ts +1 -1
- package/dist/observer-app.html +10 -9
- package/dist/observer.d.ts +2 -0
- package/dist/observer.js +52 -5
- package/dist/observer.js.map +1 -1
- package/dist/openai-responses-cu.d.ts +3 -0
- package/dist/openai-responses-cu.js +4 -1
- package/dist/openai-responses-cu.js.map +1 -1
- package/dist/pricing.js +2 -2
- package/dist/pricing.js.map +1 -1
- package/dist/program.js +146 -3
- package/dist/program.js.map +1 -1
- package/dist/run.d.ts +19 -1
- package/dist/run.js +151 -21
- package/dist/run.js.map +1 -1
- package/dist/shared-world-lab.d.ts +2 -0
- package/dist/shared-world-lab.js +90 -5
- package/dist/shared-world-lab.js.map +1 -1
- package/dist/study-analysis-engine.d.ts +34 -0
- package/dist/study-analysis-engine.js +217 -0
- package/dist/study-analysis-engine.js.map +1 -0
- package/dist/study-analysis-evidence.d.ts +27 -0
- package/dist/study-analysis-evidence.js +420 -0
- package/dist/study-analysis-evidence.js.map +1 -0
- package/dist/study-analysis-provider.d.ts +38 -0
- package/dist/study-analysis-provider.js +140 -0
- package/dist/study-analysis-provider.js.map +1 -0
- package/dist/study-analysis-service.d.ts +45 -0
- package/dist/study-analysis-service.js +207 -0
- package/dist/study-analysis-service.js.map +1 -0
- package/dist/study-analysis-sharing.d.ts +11 -0
- package/dist/study-analysis-sharing.js +27 -0
- package/dist/study-analysis-sharing.js.map +1 -0
- package/dist/study-analysis-store.d.ts +30 -0
- package/dist/study-analysis-store.js +343 -0
- package/dist/study-analysis-store.js.map +1 -0
- package/dist/study-analysis-validation.d.ts +448 -0
- package/dist/study-analysis-validation.js +331 -0
- package/dist/study-analysis-validation.js.map +1 -0
- package/dist/study-analysis.d.ts +167 -0
- package/dist/study-analysis.js +4 -0
- package/dist/study-analysis.js.map +1 -0
- package/docs/architecture/observer.md +10 -1
- package/docs/contracts/feedback.md +9 -3
- package/docs/contracts/schemas.md +52 -3
- package/docs/contracts/study-analysis.md +150 -0
- package/docs/goals/current.md +4 -3
- package/docs/ramp/README.md +13 -2
- package/docs/release/0.88.2-sequential-study-budgets.md +57 -0
- package/docs/release/0.89.0-study-findings.md +79 -0
- package/package.json +3 -2
package/dist/program.js
CHANGED
|
@@ -36,6 +36,9 @@ import { readRunDetail } from "./run-detail.js";
|
|
|
36
36
|
import { launchRun, readLaunchLogTail } from "./tui-launch.js";
|
|
37
37
|
import { TUI_MIN_NODE_MAJOR, nodeSupportsTui, tuiBundleUrl } from "./tui-contract.js";
|
|
38
38
|
import { forTerminal } from "./terminal-encoding.js";
|
|
39
|
+
import { analyzeStudy, correctStudyAnalysis, showStudyAnalysis } from "./study-analysis-service.js";
|
|
40
|
+
import { listStudyAnalyses, listStudyAnalysisExecutions } from "./study-analysis-store.js";
|
|
41
|
+
import { resolveRunPath } from "./run.js";
|
|
39
42
|
import { detectAgentSession } from "./agent-session.js";
|
|
40
43
|
import { runCommsCatchHost } from "./comms-catch-host.js";
|
|
41
44
|
import { DEFAULT_SANDBOX_CATCH_PORT } from "./comms-sandbox-catch.js";
|
|
@@ -395,6 +398,7 @@ export function createProgram(io = {}) {
|
|
|
395
398
|
registerVerifyCommand(program, cliIo);
|
|
396
399
|
registerCleanupCommand(program, cliIo);
|
|
397
400
|
registerReviewCommand(program, cliIo);
|
|
401
|
+
registerAnalyzeCommand(program, cliIo);
|
|
398
402
|
registerRunsCommand(program, cliIo);
|
|
399
403
|
registerStatsCommand(program, cliIo);
|
|
400
404
|
registerExportCommand(program, cliIo);
|
|
@@ -947,6 +951,128 @@ function registerReviewCommand(parent, io) {
|
|
|
947
951
|
io.setExitCode("ok" in result && result.ok === false ? 2 : 0);
|
|
948
952
|
});
|
|
949
953
|
}
|
|
954
|
+
/** Commander may collect a shared flag on the parent; only explicit values override leaf defaults. */
|
|
955
|
+
function analysisSelection(options, command) {
|
|
956
|
+
const parent = command.parent;
|
|
957
|
+
const selected = { ...options };
|
|
958
|
+
for (const key of ["cwd", "run"]) {
|
|
959
|
+
if (parent?.getOptionValueSource(key) === "cli" && command.getOptionValueSource(key) !== "cli") {
|
|
960
|
+
selected[key] = parent.getOptionValue(key);
|
|
961
|
+
}
|
|
962
|
+
}
|
|
963
|
+
return selected;
|
|
964
|
+
}
|
|
965
|
+
function registerAnalyzeCommand(parent, io) {
|
|
966
|
+
const analyze = parent.command("analyze")
|
|
967
|
+
.enablePositionalOptions()
|
|
968
|
+
.description("Analyze retained participant evidence into versioned findings. Explicit opt-in: sends selected text and captures to OpenAI. Opening Observer never starts analysis.")
|
|
969
|
+
.summary("Generate evidence-linked study findings.")
|
|
970
|
+
.option("--run <id>", "Completed run id or latest pointer.", "latest")
|
|
971
|
+
.option("--cwd <path>", "Target project directory.", ".")
|
|
972
|
+
.option("--max-cost <usd>", "Required, including with --dry-run: admission estimate ceiling in USD; not an exact billing cap.")
|
|
973
|
+
.option("--model <id>", "Supported vision analysis model; analysis uses high reasoning effort.", "gpt-6-astra")
|
|
974
|
+
.option("--question <text>", "Additional reviewer question; does not change participant instructions.")
|
|
975
|
+
.option("--timeout-ms <ms>", "Request timeout, at most 600000 ms.", "300000")
|
|
976
|
+
.option("--max-output-tokens <n>", "Bound response tokens, including reasoning, from 256 to 32768.", "16384")
|
|
977
|
+
.option("--dry-run", "Capture and validate local input and estimate admission; no request or analysis artifact.")
|
|
978
|
+
.option("--rerun", "Create a new immutable version even when the same input and configuration were analyzed.")
|
|
979
|
+
.option("--json", JSON_OPTION_DESCRIPTION)
|
|
980
|
+
.action(async (options, command) => {
|
|
981
|
+
const controller = new AbortController();
|
|
982
|
+
const cancel = () => controller.abort();
|
|
983
|
+
process.once("SIGINT", cancel);
|
|
984
|
+
try {
|
|
985
|
+
const result = await analyzeStudy(options.cwd, options.run, {
|
|
986
|
+
config: { model: options.model, maxCostUsd: Number(options.maxCost), question: options.question ?? null,
|
|
987
|
+
timeoutMs: Number(options.timeoutMs), maxOutputTokens: Number(options.maxOutputTokens) },
|
|
988
|
+
...(options.dryRun === undefined ? {} : { dryRun: options.dryRun }),
|
|
989
|
+
...(options.rerun === undefined ? {} : { rerun: options.rerun })
|
|
990
|
+
}, { signal: controller.signal, onProgress: (progress) => io.writeErr(`Analysis ${progress.phase}: ${progress.evidenceCount} evidence items, ${progress.captureCount} captures.\n`) });
|
|
991
|
+
writeResult(command, io, result, (value) => {
|
|
992
|
+
if (value.ok && value.dryRun)
|
|
993
|
+
return `Admission estimate: $${value.admission?.estimatedCostUsd ?? "unknown"}. No request sent.\n`;
|
|
994
|
+
const lines = [];
|
|
995
|
+
if (!value.ok)
|
|
996
|
+
lines.push(value.error?.message ?? "Analysis unavailable.", value.error?.code ?? "");
|
|
997
|
+
if (value.artifactPath)
|
|
998
|
+
lines.push(`${value.reused ? "Reused" : "Saved"} ${value.status} analysis: ${value.artifactPath}`);
|
|
999
|
+
if (value.executionReceiptPath)
|
|
1000
|
+
lines.push(`Execution receipt: ${value.executionReceiptPath}`);
|
|
1001
|
+
if (value.usage) {
|
|
1002
|
+
lines.push(`Recorded attempt usage: ${value.usage.inputTokens ?? "unknown"} input tokens, ${value.usage.outputTokens ?? "unknown"} output tokens.`);
|
|
1003
|
+
lines.push(`Estimated attempt cost: ${value.usage.estimatedCostUsd === null ? "unknown" : `$${value.usage.estimatedCostUsd}`}.`);
|
|
1004
|
+
}
|
|
1005
|
+
if (value.reused)
|
|
1006
|
+
lines.push("No new request sent.");
|
|
1007
|
+
lines.push(...value.warnings);
|
|
1008
|
+
return forTerminal(lines.filter(Boolean).join("\n") + "\n");
|
|
1009
|
+
});
|
|
1010
|
+
io.setExitCode(result.ok ? 0 : 2);
|
|
1011
|
+
}
|
|
1012
|
+
finally {
|
|
1013
|
+
process.removeListener("SIGINT", cancel);
|
|
1014
|
+
}
|
|
1015
|
+
});
|
|
1016
|
+
analyze.command("list").description("List immutable analysis versions, including failed attempts.")
|
|
1017
|
+
.option("--run <id>", "Run id or latest pointer.", "latest")
|
|
1018
|
+
.option("--cwd <path>", "Target project directory.", ".")
|
|
1019
|
+
.option("--json", JSON_OPTION_DESCRIPTION)
|
|
1020
|
+
.action(async (options, command) => {
|
|
1021
|
+
options = analysisSelection(options, command);
|
|
1022
|
+
const prepared = await resolveRunPath(resolve(options.cwd), options.run).catch(() => null);
|
|
1023
|
+
const versions = prepared ? await listStudyAnalyses(prepared) : [];
|
|
1024
|
+
const executions = prepared ? await listStudyAnalysisExecutions(prepared) : { receipts: [], warnings: [] };
|
|
1025
|
+
const result = { schema: "humanish.analysis-history.v1", ok: prepared !== null, run: options.run,
|
|
1026
|
+
executions: executions.receipts, warnings: executions.warnings,
|
|
1027
|
+
versions: versions.map(({ id, state, analysis, warnings }) => ({ id, state, status: analysis?.status ?? null,
|
|
1028
|
+
createdAt: analysis?.createdAt ?? null, findings: analysis?.result?.findings.length ?? null,
|
|
1029
|
+
usage: analysis?.usage ?? null, warnings })) };
|
|
1030
|
+
writeResult(command, io, result, (value) => forTerminal(JSON.stringify(value, null, 2) + "\n"));
|
|
1031
|
+
io.setExitCode(result.ok ? 0 : 2);
|
|
1032
|
+
});
|
|
1033
|
+
analyze.command("show").description("Read validated analysis and correction history. Defaults to the latest usable version.")
|
|
1034
|
+
.option("--run <id>", "Run id or latest pointer.", "latest")
|
|
1035
|
+
.option("--id <id>", "Exact analysis version.")
|
|
1036
|
+
.option("--cwd <path>", "Target project directory.", ".")
|
|
1037
|
+
.option("--json", JSON_OPTION_DESCRIPTION)
|
|
1038
|
+
.action(async (options, command) => {
|
|
1039
|
+
options = analysisSelection(options, command);
|
|
1040
|
+
const result = await showStudyAnalysis(options.cwd, options.run, options.id);
|
|
1041
|
+
writeResult(command, io, result, (value) => forTerminal(JSON.stringify(value, null, 2) + "\n"));
|
|
1042
|
+
io.setExitCode(result.state === "invalid" ? 2 : 0);
|
|
1043
|
+
});
|
|
1044
|
+
analyze.command("correct").description("Append a human review note bound to one exact finding version; original claims remain intact.")
|
|
1045
|
+
.option("--run <id>", "Run id or latest pointer.", "latest")
|
|
1046
|
+
.option("--cwd <path>", "Target project directory.", ".")
|
|
1047
|
+
.requiredOption("--analysis <id>", "Analysis version to review.")
|
|
1048
|
+
.requiredOption("--finding <id>", "Finding to review.")
|
|
1049
|
+
.addOption(new Option("--status <status>", "Review disposition.").choices(["confirmed", "dismissed", "amended"]).makeOptionMandatory())
|
|
1050
|
+
.requiredOption("--reason <text>", "Why this disposition is supported.")
|
|
1051
|
+
.option("--claim <text>", "Replacement claim, required only for amended findings.")
|
|
1052
|
+
.option("--json", JSON_OPTION_DESCRIPTION)
|
|
1053
|
+
.action(async (options, command) => {
|
|
1054
|
+
options = analysisSelection(options, command);
|
|
1055
|
+
try {
|
|
1056
|
+
const correction = await correctStudyAnalysis(options.cwd, options.run, { analysisId: options.analysis,
|
|
1057
|
+
findingId: options.finding, status: options.status, reason: options.reason,
|
|
1058
|
+
...(options.claim === undefined ? {} : { replacementClaim: options.claim }) });
|
|
1059
|
+
writeResult(command, io, { schema: "humanish.analysis-correction-result.v1", ok: true, correction }, (value) => `Saved correction ${value.correction.id}. Original analysis preserved.\n`);
|
|
1060
|
+
io.setExitCode(0);
|
|
1061
|
+
}
|
|
1062
|
+
catch (error) {
|
|
1063
|
+
const code = error instanceof Error && ["ANALYSIS_BUSY", "ANALYSIS_CORRECTION_HISTORY_UNAVAILABLE"].includes(error.message)
|
|
1064
|
+
? error.message : "ANALYSIS_CORRECTION_INVALID";
|
|
1065
|
+
const message = code === "ANALYSIS_BUSY"
|
|
1066
|
+
? "Another analysis or correction holds this run's lock. Retry after it finishes."
|
|
1067
|
+
: code === "ANALYSIS_CORRECTION_HISTORY_UNAVAILABLE"
|
|
1068
|
+
? "Correction history is unavailable or full. No correction was added; existing records were preserved."
|
|
1069
|
+
: "Correction requires a current valid finding, a reason, and a replacement claim only for amended status. Sensitive text is rejected.";
|
|
1070
|
+
writeResult(command, io, { schema: "humanish.analysis-correction-result.v1", ok: false,
|
|
1071
|
+
error: { code, message } }, (value) => value.error.message + "\n");
|
|
1072
|
+
io.setExitCode(2);
|
|
1073
|
+
}
|
|
1074
|
+
});
|
|
1075
|
+
}
|
|
950
1076
|
function registerExportCommand(parent, io) {
|
|
951
1077
|
parent
|
|
952
1078
|
.command("export")
|
|
@@ -1737,7 +1863,8 @@ function registerFeedbackCommands(parent, io) {
|
|
|
1737
1863
|
.summary("Create public-safe feedback drafts, no GitHub API.");
|
|
1738
1864
|
feedback
|
|
1739
1865
|
.command("list")
|
|
1740
|
-
.description("List feedback
|
|
1866
|
+
.description("List recorded feedback candidates and any saved draft.")
|
|
1867
|
+
.addHelpText("after", "\nWith no candidates, feedback draft and feedback issue can generate a run-summary follow-up. Public drafting still requires share_ready verification.\n")
|
|
1741
1868
|
.option("--run <id>", "Run id or latest pointer.", "latest")
|
|
1742
1869
|
.option("--cwd <path>", "Target project directory.", ".")
|
|
1743
1870
|
.option("--json", JSON_OPTION_DESCRIPTION)
|
|
@@ -1751,6 +1878,8 @@ function registerFeedbackCommands(parent, io) {
|
|
|
1751
1878
|
.description("Generate a public-safe feedback draft from verified evidence.")
|
|
1752
1879
|
.option("--run <id>", "Run id or latest pointer.", "latest")
|
|
1753
1880
|
.option("--cwd <path>", "Target project directory.", ".")
|
|
1881
|
+
.option("--analysis <id>", "Independent analysis version; use with --finding.")
|
|
1882
|
+
.option("--finding <id>", "Finding within --analysis, separate from participant candidates.")
|
|
1754
1883
|
.option("--candidate <id>", "Which finding to draft (ids from `feedback list`); default: the first.")
|
|
1755
1884
|
.option("--json", JSON_OPTION_DESCRIPTION)
|
|
1756
1885
|
.action(async (options, command) => {
|
|
@@ -1763,6 +1892,8 @@ function registerFeedbackCommands(parent, io) {
|
|
|
1763
1892
|
.description("Verify the feedback draft for public issue eligibility.")
|
|
1764
1893
|
.option("--run <id>", "Run id or latest pointer.", "latest")
|
|
1765
1894
|
.option("--cwd <path>", "Target project directory.", ".")
|
|
1895
|
+
.option("--analysis <id>", "Independent analysis version; use with --finding.")
|
|
1896
|
+
.option("--finding <id>", "Finding within --analysis, separate from participant candidates.")
|
|
1766
1897
|
.option("--candidate <id>", "Which finding to verify (ids from `feedback list`); default: the first.")
|
|
1767
1898
|
.option("--json", JSON_OPTION_DESCRIPTION)
|
|
1768
1899
|
.action(async (options, command) => {
|
|
@@ -1777,6 +1908,8 @@ function registerFeedbackCommands(parent, io) {
|
|
|
1777
1908
|
.option("--cwd <path>", "Target project directory.", ".")
|
|
1778
1909
|
.requiredOption("--repo <owner/repo>", "Repository slug used in rendered filing instructions.")
|
|
1779
1910
|
.option("--format <format>", "Output format.", "markdown")
|
|
1911
|
+
.option("--analysis <id>", "Independent analysis version; use with --finding.")
|
|
1912
|
+
.option("--finding <id>", "Finding within --analysis, separate from participant candidates.")
|
|
1780
1913
|
.option("--candidate <id>", "Which finding to file (ids from `feedback list`); default: the first.")
|
|
1781
1914
|
.option("--json", JSON_OPTION_DESCRIPTION)
|
|
1782
1915
|
.action(async (options, command) => {
|
|
@@ -1803,6 +1936,8 @@ function registerFeedbackCommands(parent, io) {
|
|
|
1803
1936
|
.option("--run <id>", "Run id or latest pointer.", "latest")
|
|
1804
1937
|
.option("--cwd <path>", "Target project directory.", ".")
|
|
1805
1938
|
.requiredOption("--repo <owner/repo>", "Repository slug used in the generated URL.")
|
|
1939
|
+
.option("--analysis <id>", "Independent analysis version; use with --finding.")
|
|
1940
|
+
.option("--finding <id>", "Finding within --analysis, separate from participant candidates.")
|
|
1806
1941
|
.option("--candidate <id>", "Which finding to link (ids from `feedback list`); default: the first.")
|
|
1807
1942
|
.option("--json", JSON_OPTION_DESCRIPTION)
|
|
1808
1943
|
.action(async (options, command) => {
|
|
@@ -3455,19 +3590,27 @@ function exitCodeForSignal(signal) {
|
|
|
3455
3590
|
}
|
|
3456
3591
|
/** `--candidate` as the module option, absent when not given (exactOptionalPropertyTypes). */
|
|
3457
3592
|
function candidateOption(options) {
|
|
3458
|
-
return options.candidate === undefined ? {} : { candidate: options.candidate }
|
|
3593
|
+
return { ...(options.candidate === undefined ? {} : { candidate: options.candidate }),
|
|
3594
|
+
...(options.analysis === undefined ? {} : { analysis: options.analysis }),
|
|
3595
|
+
...(options.finding === undefined ? {} : { finding: options.finding }) };
|
|
3459
3596
|
}
|
|
3460
3597
|
function formatFeedbackHuman(result) {
|
|
3461
3598
|
if (!result.ok) {
|
|
3462
3599
|
return `${result.error?.code}: ${result.error?.message}\n`;
|
|
3463
3600
|
}
|
|
3464
3601
|
const candidates = result.candidates ?? [];
|
|
3602
|
+
const noCandidates = result.candidates !== undefined && candidates.length === 0;
|
|
3465
3603
|
return [
|
|
3466
|
-
"humanish feedback ready",
|
|
3604
|
+
noCandidates && result.draft === undefined ? "humanish feedback: no recorded candidates" : "humanish feedback ready",
|
|
3467
3605
|
`run: ${result.run}`,
|
|
3468
3606
|
...(result.draftPath ? [`draft: ${result.draftPath}`] : []),
|
|
3469
3607
|
...(result.issuePath ? [`issue: ${result.issuePath}`] : []),
|
|
3470
3608
|
...(result.draft?.source_candidate_id ? [`candidate: ${result.draft.source_candidate_id}`] : []),
|
|
3609
|
+
...(noCandidates ? [
|
|
3610
|
+
"candidates: none recorded",
|
|
3611
|
+
"With no candidates, feedback draft and feedback issue can generate a run-summary follow-up after share_ready verification.",
|
|
3612
|
+
...(result.draft ? [`summary: ${result.draft.summary}`] : [])
|
|
3613
|
+
] : []),
|
|
3471
3614
|
// Every finding the run produced, so the second and third are one flag away (#609).
|
|
3472
3615
|
...(candidates.length > 1 || (candidates.length === 1 && result.draft === undefined)
|
|
3473
3616
|
? [
|