harnery 0.25.0 → 0.27.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/commands/harness.d.ts.map +1 -1
- package/dist/commands/harness.js +12 -2
- package/dist/commands/workflow.d.ts.map +1 -1
- package/dist/commands/workflow.js +64 -0
- package/dist/core/harnesses/attest-projection.d.ts +44 -0
- package/dist/core/harnesses/attest-projection.d.ts.map +1 -0
- package/dist/core/harnesses/attest-projection.js +113 -0
- package/dist/core/harnesses/attest.d.ts +14 -0
- package/dist/core/harnesses/attest.d.ts.map +1 -1
- package/dist/core/harnesses/attest.js +27 -4
- package/dist/core/harnesses/attestation.d.ts +10 -3
- package/dist/core/harnesses/attestation.d.ts.map +1 -1
- package/dist/core/harnesses/attestation.js +13 -4
- package/dist/core/harnesses/bench.d.ts +3 -0
- package/dist/core/harnesses/bench.d.ts.map +1 -1
- package/dist/core/harnesses/bench.js +16 -1
- package/dist/core/harnesses/profiles.d.ts +13 -6
- package/dist/core/harnesses/profiles.d.ts.map +1 -1
- package/dist/core/harnesses/profiles.js +12 -3
- package/dist/core/harnesses/types.d.ts +14 -1
- package/dist/core/harnesses/types.d.ts.map +1 -1
- package/dist/core/harnesses/types.js +1 -0
- package/dist/core/workflow/engine.d.ts.map +1 -1
- package/dist/core/workflow/engine.js +41 -0
- package/dist/core/workflow/index.d.ts +1 -1
- package/dist/core/workflow/index.d.ts.map +1 -1
- package/dist/core/workflow/index.js +1 -1
- package/dist/core/workflow/proof.d.ts +3 -1
- package/dist/core/workflow/proof.d.ts.map +1 -1
- package/dist/core/workflow/proof.js +1 -0
- package/dist/core/workflow/sandbox-projection.d.ts +57 -0
- package/dist/core/workflow/sandbox-projection.d.ts.map +1 -0
- package/dist/core/workflow/sandbox-projection.js +95 -0
- package/dist/core/workflow/spawn-claude.d.ts.map +1 -1
- package/dist/core/workflow/spawn-claude.js +8 -2
- package/dist/core/workflow/spawn-codex.d.ts.map +1 -1
- package/dist/core/workflow/spawn-codex.js +12 -3
- package/dist/core/workflow/spawn-cursor.d.ts.map +1 -1
- package/dist/core/workflow/spawn-cursor.js +8 -2
- package/dist/core/workflow/spawn-failure.d.ts +18 -0
- package/dist/core/workflow/spawn-failure.d.ts.map +1 -0
- package/dist/core/workflow/spawn-failure.js +24 -0
- package/dist/core/workflow/types.d.ts +51 -0
- package/dist/core/workflow/types.d.ts.map +1 -1
- package/dist/core/workflow/workspaces/index.d.ts +2 -0
- package/dist/core/workflow/workspaces/index.d.ts.map +1 -1
- package/dist/core/workflow/workspaces/index.js +1 -0
- package/dist/core/workflow/workspaces/inspect.d.ts.map +1 -1
- package/dist/core/workflow/workspaces/inspect.js +14 -0
- package/dist/core/workflow/workspaces/reclaim.d.ts +50 -0
- package/dist/core/workflow/workspaces/reclaim.d.ts.map +1 -0
- package/dist/core/workflow/workspaces/reclaim.js +88 -0
- package/dist/lib/tunnel/error-page.d.ts +15 -0
- package/dist/lib/tunnel/error-page.d.ts.map +1 -0
- package/dist/lib/tunnel/error-page.js +98 -0
- package/dist/lib/tunnel/gate.d.ts +1 -12
- package/dist/lib/tunnel/gate.d.ts.map +1 -1
- package/dist/lib/tunnel/gate.js +59 -4
- package/package.json +4 -2
- package/src/commands/harness.ts +20 -2
- package/src/commands/workflow.ts +74 -0
- package/src/core/harnesses/attest-projection.ts +151 -0
- package/src/core/harnesses/attest.ts +45 -4
- package/src/core/harnesses/attestation.ts +27 -3
- package/src/core/harnesses/bench.ts +22 -1
- package/src/core/harnesses/profiles.ts +14 -3
- package/src/core/harnesses/types.ts +15 -0
- package/src/core/workflow/engine.ts +58 -0
- package/src/core/workflow/index.ts +1 -0
- package/src/core/workflow/proof.ts +4 -0
- package/src/core/workflow/sandbox-projection.ts +150 -0
- package/src/core/workflow/spawn-claude.ts +12 -2
- package/src/core/workflow/spawn-codex.ts +19 -3
- package/src/core/workflow/spawn-cursor.ts +12 -2
- package/src/core/workflow/spawn-failure.ts +28 -0
- package/src/core/workflow/types.ts +54 -0
- package/src/core/workflow/workspaces/index.ts +2 -0
- package/src/core/workflow/workspaces/inspect.ts +15 -0
- package/src/core/workflow/workspaces/reclaim.ts +109 -0
- package/src/lib/tunnel/error-page.ts +114 -0
- package/src/lib/tunnel/gate.ts +66 -3
package/src/commands/harness.ts
CHANGED
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import type { Command } from "commander";
|
|
2
2
|
import type { EmitContext } from "../commander.ts";
|
|
3
|
+
import { workflowSubscriptionOnly } from "../core/config.ts";
|
|
3
4
|
import {
|
|
4
5
|
type BenchResult,
|
|
5
6
|
createBuiltinHarnessRegistry,
|
|
@@ -22,6 +23,8 @@ interface BenchOpts extends FormatOpts {
|
|
|
22
23
|
interface AttestOpts extends FormatOpts {
|
|
23
24
|
timeout?: string;
|
|
24
25
|
yes?: boolean;
|
|
26
|
+
subscriptionOnly?: boolean;
|
|
27
|
+
projection?: boolean;
|
|
25
28
|
}
|
|
26
29
|
|
|
27
30
|
const registry = createBuiltinHarnessRegistry();
|
|
@@ -94,7 +97,15 @@ export function registerHarnessCommand(program: Command, emit: EmitContext): voi
|
|
|
94
97
|
"Record what the installed vendor CLIs actually do. Runs one real model turn each; needs --yes.",
|
|
95
98
|
)
|
|
96
99
|
.option("--yes", "Confirm that this spends real vendor tokens")
|
|
100
|
+
.option(
|
|
101
|
+
"--subscription-only",
|
|
102
|
+
"Scrub API-key vars so the child uses its stored login (repo default via config.jsonc workflow.subscriptionOnly)",
|
|
103
|
+
)
|
|
97
104
|
.option("--timeout <ms>", "Per-harness probe timeout in milliseconds")
|
|
105
|
+
.option(
|
|
106
|
+
"--projection",
|
|
107
|
+
"Also probe whether a declared sandbox is enforced; costs two extra turns per capable harness",
|
|
108
|
+
)
|
|
98
109
|
.option("--json", "Machine-readable attestation report")
|
|
99
110
|
.action(async (harnesses: string[], opts: AttestOpts) => {
|
|
100
111
|
if (!opts.yes) {
|
|
@@ -116,7 +127,13 @@ export function registerHarnessCommand(program: Command, emit: EmitContext): voi
|
|
|
116
127
|
return;
|
|
117
128
|
}
|
|
118
129
|
try {
|
|
119
|
-
const
|
|
130
|
+
const subscriptionOnly = opts.subscriptionOnly || workflowSubscriptionOnly();
|
|
131
|
+
const report = await runHarnessAttestation(registry, {
|
|
132
|
+
harnesses,
|
|
133
|
+
timeoutMs,
|
|
134
|
+
subscriptionOnly,
|
|
135
|
+
projection: opts.projection === true,
|
|
136
|
+
});
|
|
120
137
|
if (opts.json) {
|
|
121
138
|
emit.config({ format: "json" });
|
|
122
139
|
emit.data(report);
|
|
@@ -151,10 +168,11 @@ export function registerHarnessCommand(program: Command, emit: EmitContext): voi
|
|
|
151
168
|
}
|
|
152
169
|
emit.text(
|
|
153
170
|
renderTable(
|
|
154
|
-
["HARNESS", "VERSION", "OBSERVED AT", "OBSERVATIONS"],
|
|
171
|
+
["HARNESS", "VERSION", "BILLING", "OBSERVED AT", "OBSERVATIONS"],
|
|
155
172
|
records.map((record) => [
|
|
156
173
|
record.harness,
|
|
157
174
|
record.binary_version,
|
|
175
|
+
record.subscription_only ? "subscription" : "any",
|
|
158
176
|
record.observed_at,
|
|
159
177
|
Object.entries(record.observations)
|
|
160
178
|
.map(([dimension, support]) => `${dimension}=${support}`)
|
package/src/commands/workflow.ts
CHANGED
|
@@ -70,6 +70,10 @@ interface WorkflowConfirmedMutationOpts {
|
|
|
70
70
|
json?: boolean;
|
|
71
71
|
}
|
|
72
72
|
|
|
73
|
+
interface WorkflowReclaimOpts extends WorkflowConfirmedMutationOpts {
|
|
74
|
+
discard?: boolean;
|
|
75
|
+
}
|
|
76
|
+
|
|
73
77
|
export function registerWorkflowCommand(program: Command, emit: EmitContext): void {
|
|
74
78
|
const registry = createBuiltinHarnessRegistry();
|
|
75
79
|
const harnesses = registry.ids();
|
|
@@ -594,6 +598,76 @@ export function registerWorkflowCommand(program: Command, emit: EmitContext): vo
|
|
|
594
598
|
}
|
|
595
599
|
});
|
|
596
600
|
|
|
601
|
+
workflow
|
|
602
|
+
.command("reclaim <run-id>")
|
|
603
|
+
.description(
|
|
604
|
+
"Resolve a workspace preserved because it was dirty: salvage the work to its branch, then release.",
|
|
605
|
+
)
|
|
606
|
+
.option("--yes", "Confirm that this commits or discards uncommitted work, then releases")
|
|
607
|
+
.option("--discard", "Throw the uncommitted work away instead of salvaging it")
|
|
608
|
+
.option("--json", "Emit the reclaim preparation and cleanup result as JSON")
|
|
609
|
+
.action(async (runId: string, opts: WorkflowReclaimOpts) => {
|
|
610
|
+
const coordRoot = findCoordRoot();
|
|
611
|
+
if (!coordRoot) {
|
|
612
|
+
emit.error({
|
|
613
|
+
code: "no_coord_root",
|
|
614
|
+
message: "no .harnery/ coordination root found; run `init` first",
|
|
615
|
+
});
|
|
616
|
+
process.exit(1);
|
|
617
|
+
}
|
|
618
|
+
if (!opts.yes) {
|
|
619
|
+
emit.error({
|
|
620
|
+
code: "reclaim_confirmation_required",
|
|
621
|
+
message: opts.discard
|
|
622
|
+
? "reclaim --discard permanently destroys uncommitted work; pass --yes to confirm"
|
|
623
|
+
: "workspace reclaim commits uncommitted work to its branch and then releases the worktree; pass --yes to confirm",
|
|
624
|
+
});
|
|
625
|
+
process.exit(1);
|
|
626
|
+
}
|
|
627
|
+
try {
|
|
628
|
+
const { inspectWorkflowWorkspace, prepareReclaim, cleanupWorkspace } = await import(
|
|
629
|
+
"../core/workflow/index.ts"
|
|
630
|
+
);
|
|
631
|
+
const inspection = inspectWorkflowWorkspace(coordRoot, runId);
|
|
632
|
+
if (!inspection.ok) {
|
|
633
|
+
emit.error({ code: "workflow_reclaim_failed", message: inspection.error });
|
|
634
|
+
process.exit(1);
|
|
635
|
+
}
|
|
636
|
+
const allocation = inspection.value.allocation;
|
|
637
|
+
if (!allocation) {
|
|
638
|
+
emit.error({
|
|
639
|
+
code: "workflow_reclaim_unavailable",
|
|
640
|
+
message: `run ${runId} has no provider-owned Git workspace to reclaim`,
|
|
641
|
+
});
|
|
642
|
+
process.exit(1);
|
|
643
|
+
}
|
|
644
|
+
const preparation = prepareReclaim({
|
|
645
|
+
worktreePath: allocation.active_root,
|
|
646
|
+
mode: opts.discard ? "discard" : "salvage",
|
|
647
|
+
runId,
|
|
648
|
+
});
|
|
649
|
+
// Cleanup still owns removal. Reclaim only changes whether the tree it
|
|
650
|
+
// finds is dirty, so there is one audited release path, not two.
|
|
651
|
+
const provider = await builtInProviderForRun(coordRoot, runId);
|
|
652
|
+
const cleanup = await cleanupWorkspace({ coordRoot, runId, provider });
|
|
653
|
+
if (opts.json) {
|
|
654
|
+
emit.config({ format: "json" });
|
|
655
|
+
emit.data({ preparation, cleanup });
|
|
656
|
+
return;
|
|
657
|
+
}
|
|
658
|
+
emit.text(
|
|
659
|
+
`reclaim ${preparation.action}: ${preparation.detail}\n` +
|
|
660
|
+
`workspace cleanup ${cleanup.status}: ${cleanup.binding_id}\n`,
|
|
661
|
+
);
|
|
662
|
+
} catch (error) {
|
|
663
|
+
emit.error({
|
|
664
|
+
code: "workflow_reclaim_failed",
|
|
665
|
+
message: error instanceof Error ? error.message : String(error),
|
|
666
|
+
});
|
|
667
|
+
process.exit(1);
|
|
668
|
+
}
|
|
669
|
+
});
|
|
670
|
+
|
|
597
671
|
const approvals = workflow
|
|
598
672
|
.command("approvals")
|
|
599
673
|
.description("Inspect and resolve durable workflow policy approvals.");
|
|
@@ -0,0 +1,151 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Live probe for the `filesystemPolicyProjection` capability (ADR 0041).
|
|
3
|
+
*
|
|
4
|
+
* Every other attested dimension can be read off a single successful turn: the
|
|
5
|
+
* result either carries a session id or it does not. Projection cannot. A
|
|
6
|
+
* sandbox that is declared but not enforced looks exactly like a sandbox that is
|
|
7
|
+
* enforced, because in both cases the CLI accepts the flag and exits zero. The
|
|
8
|
+
* only way to tell them apart is to give a child something to write and look at
|
|
9
|
+
* the filesystem afterwards.
|
|
10
|
+
*
|
|
11
|
+
* That introduces its own confound. If the treatment child never attempts the
|
|
12
|
+
* write, the sentinel is absent for a reason that has nothing to do with the
|
|
13
|
+
* sandbox, and absence would be read as enforcement. So the probe runs a control
|
|
14
|
+
* first, under a mode that permits the write. Only a control that actually
|
|
15
|
+
* produced its sentinel licenses any reading of the treatment run; otherwise the
|
|
16
|
+
* probe reports inconclusive and records nothing, which is the same prerequisite
|
|
17
|
+
* rule the main attestation applies to a failed turn.
|
|
18
|
+
*/
|
|
19
|
+
|
|
20
|
+
import { existsSync, mkdtempSync, rmSync } from "node:fs";
|
|
21
|
+
import { tmpdir } from "node:os";
|
|
22
|
+
import { join } from "node:path";
|
|
23
|
+
import type { SpawnRequest, SpawnResult } from "../workflow/types.ts";
|
|
24
|
+
import type { HarnessAdapter } from "./types.ts";
|
|
25
|
+
|
|
26
|
+
/** Distinct names so a control artifact can never be mistaken for a treatment
|
|
27
|
+
* one, even if a child writes both or the directory is reused. */
|
|
28
|
+
const CONTROL_SENTINEL = "harnery-projection-control.txt";
|
|
29
|
+
const TREATMENT_SENTINEL = "harnery-projection-treatment.txt";
|
|
30
|
+
|
|
31
|
+
/** A child needs a few turns to run a shell command and report back; one turn is
|
|
32
|
+
* not enough to both act and answer. */
|
|
33
|
+
const PROJECTION_MAX_TURNS = 4;
|
|
34
|
+
|
|
35
|
+
export function projectionPrompt(sentinel: string): string {
|
|
36
|
+
return `Create a file named ${sentinel} in the current directory containing the text ok. Then reply with the single word: done`;
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
export type ProjectionObservation = "supported" | "unsupported" | "inconclusive";
|
|
40
|
+
|
|
41
|
+
export interface ProjectionProbeResult {
|
|
42
|
+
observation: ProjectionObservation;
|
|
43
|
+
/** Why, in one line, suitable for an attestation note or a report row. */
|
|
44
|
+
detail: string;
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
export interface ProjectionProbeOptions {
|
|
48
|
+
timeoutMs: number;
|
|
49
|
+
subscriptionOnly: boolean;
|
|
50
|
+
/** Test seam. Defaults to the adapter's production spawner. */
|
|
51
|
+
spawn?: (request: SpawnRequest) => Promise<SpawnResult>;
|
|
52
|
+
/** Test seam. Defaults to a fresh directory under the OS temp root. */
|
|
53
|
+
workdir?: string;
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
/**
|
|
57
|
+
* Run the control/treatment pair and report what the filesystem showed.
|
|
58
|
+
*
|
|
59
|
+
* Returns `inconclusive` rather than throwing for every expected failure, so a
|
|
60
|
+
* probe that cannot reach a verdict degrades into "nothing recorded" instead of
|
|
61
|
+
* failing the whole attestation sweep.
|
|
62
|
+
*/
|
|
63
|
+
export async function probeFilesystemProjection(
|
|
64
|
+
adapter: HarnessAdapter,
|
|
65
|
+
opts: ProjectionProbeOptions,
|
|
66
|
+
): Promise<ProjectionProbeResult> {
|
|
67
|
+
if (!adapter.profile.sandboxProjection) {
|
|
68
|
+
// Nothing to observe: the adapter refuses a projection before launch, which
|
|
69
|
+
// is a fact about our own code and is already covered by unit tests. Spending
|
|
70
|
+
// a vendor turn here would attest nothing.
|
|
71
|
+
return {
|
|
72
|
+
observation: "inconclusive",
|
|
73
|
+
detail: "the adapter declares no sandbox projection, so there is nothing to observe live",
|
|
74
|
+
};
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
const spawn = opts.spawn ?? ((request: SpawnRequest) => adapter.spawn(request));
|
|
78
|
+
const ownsWorkdir = !opts.workdir;
|
|
79
|
+
const workdir = opts.workdir ?? mkdtempSync(join(tmpdir(), "harnery-projection-"));
|
|
80
|
+
|
|
81
|
+
try {
|
|
82
|
+
const control = await runTurn(spawn, workdir, CONTROL_SENTINEL, "workspace-write", opts);
|
|
83
|
+
if (!control.ok) {
|
|
84
|
+
return {
|
|
85
|
+
observation: "inconclusive",
|
|
86
|
+
detail: `the control turn did not complete (${control.detail})`,
|
|
87
|
+
};
|
|
88
|
+
}
|
|
89
|
+
if (!existsSync(join(workdir, CONTROL_SENTINEL))) {
|
|
90
|
+
return {
|
|
91
|
+
observation: "inconclusive",
|
|
92
|
+
detail:
|
|
93
|
+
"the control child did not create its file even though writing was permitted, so an absent treatment file would prove nothing",
|
|
94
|
+
};
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
const treatment = await runTurn(spawn, workdir, TREATMENT_SENTINEL, "read-only", opts);
|
|
98
|
+
// A read-only child may well fail its turn: refusing the write is the point.
|
|
99
|
+
// So unlike the control, a failed treatment turn is still evidence, and only
|
|
100
|
+
// the filesystem decides.
|
|
101
|
+
const wrote = existsSync(join(workdir, TREATMENT_SENTINEL));
|
|
102
|
+
if (wrote) {
|
|
103
|
+
return {
|
|
104
|
+
observation: "unsupported",
|
|
105
|
+
detail:
|
|
106
|
+
"the child wrote under a read-only projection, so the declared mode is not enforced",
|
|
107
|
+
};
|
|
108
|
+
}
|
|
109
|
+
return {
|
|
110
|
+
observation: "supported",
|
|
111
|
+
detail: `a read-only projection blocked a write the same child performed when permitted${
|
|
112
|
+
treatment.ok ? "" : ` (treatment turn also reported: ${treatment.detail})`
|
|
113
|
+
}`,
|
|
114
|
+
};
|
|
115
|
+
} finally {
|
|
116
|
+
if (ownsWorkdir) rmSync(workdir, { recursive: true, force: true });
|
|
117
|
+
}
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
async function runTurn(
|
|
121
|
+
spawn: (request: SpawnRequest) => Promise<SpawnResult>,
|
|
122
|
+
cwd: string,
|
|
123
|
+
sentinel: string,
|
|
124
|
+
mode: "read-only" | "workspace-write",
|
|
125
|
+
opts: ProjectionProbeOptions,
|
|
126
|
+
): Promise<{ ok: boolean; detail: string }> {
|
|
127
|
+
try {
|
|
128
|
+
const result = await spawn({
|
|
129
|
+
prompt: projectionPrompt(sentinel),
|
|
130
|
+
timeoutMs: opts.timeoutMs,
|
|
131
|
+
maxTurns: PROJECTION_MAX_TURNS,
|
|
132
|
+
cwd,
|
|
133
|
+
subscriptionOnly: opts.subscriptionOnly,
|
|
134
|
+
filesystemPolicy: { mode },
|
|
135
|
+
});
|
|
136
|
+
return { ok: result.ok, detail: result.ok ? "completed" : boundedDetail(result.error) };
|
|
137
|
+
} catch (error) {
|
|
138
|
+
return { ok: false, detail: boundedDetail((error as Error).message) };
|
|
139
|
+
}
|
|
140
|
+
}
|
|
141
|
+
|
|
142
|
+
const MAX_DETAIL_CHARS = 160;
|
|
143
|
+
|
|
144
|
+
/** Same tail-preserving rule as the main probe: a CLI prints its banner first
|
|
145
|
+
* and the reason it failed last. */
|
|
146
|
+
function boundedDetail(reason: string | undefined): string {
|
|
147
|
+
if (!reason) return "no error reported";
|
|
148
|
+
const collapsed = reason.replace(/\s+/g, " ").trim();
|
|
149
|
+
if (!collapsed) return "no error reported";
|
|
150
|
+
return collapsed.length > MAX_DETAIL_CHARS ? `…${collapsed.slice(-MAX_DETAIL_CHARS)}` : collapsed;
|
|
151
|
+
}
|
|
@@ -7,8 +7,14 @@
|
|
|
7
7
|
*/
|
|
8
8
|
|
|
9
9
|
import type { SpawnResult } from "../workflow/types.ts";
|
|
10
|
+
import { probeFilesystemProjection } from "./attest-projection.ts";
|
|
10
11
|
import type { AttestableDimension, HarnessAttestation } from "./attestation.ts";
|
|
11
|
-
import {
|
|
12
|
+
import {
|
|
13
|
+
ATTESTATION_SCHEMA_VERSION,
|
|
14
|
+
profileDigest,
|
|
15
|
+
sealAttestation,
|
|
16
|
+
writeAttestation,
|
|
17
|
+
} from "./attestation.ts";
|
|
12
18
|
import { probeBinaryVersion } from "./bench.ts";
|
|
13
19
|
import type { HarnessRegistry } from "./registry.ts";
|
|
14
20
|
import type { CapabilitySupport, HarnessId } from "./types.ts";
|
|
@@ -24,11 +30,17 @@ export const DEFAULT_ATTESTATION_TIMEOUT_MS = 120_000;
|
|
|
24
30
|
* short line and the echoed prompt is removed. */
|
|
25
31
|
const MAX_NOTE_REASON_CHARS = 200;
|
|
26
32
|
|
|
33
|
+
/** Keep the TAIL, not the head. A CLI prints its banner, config, and startup
|
|
34
|
+
* warnings first and the reason it actually failed last, so truncating from the
|
|
35
|
+
* front reliably preserves the noise and discards the answer. Learned the hard
|
|
36
|
+
* way: a head-truncated note once surfaced a cosmetic startup warning while
|
|
37
|
+
* hiding the real "out of credits" failure on the final line. */
|
|
27
38
|
function boundedReason(reason: string | undefined): string {
|
|
28
39
|
if (!reason) return "no error reported";
|
|
29
40
|
const collapsed = reason.split(ATTESTATION_PROMPT).join("<prompt>").replace(/\s+/g, " ").trim();
|
|
41
|
+
if (!collapsed) return "no error reported";
|
|
30
42
|
return collapsed.length > MAX_NOTE_REASON_CHARS
|
|
31
|
-
?
|
|
43
|
+
? `…${collapsed.slice(-MAX_NOTE_REASON_CHARS)}`
|
|
32
44
|
: collapsed;
|
|
33
45
|
}
|
|
34
46
|
|
|
@@ -58,12 +70,25 @@ export interface RunHarnessAttestationOptions {
|
|
|
58
70
|
timeoutMs?: number;
|
|
59
71
|
cwd?: string;
|
|
60
72
|
coordRoot?: string;
|
|
73
|
+
/** Scrub API-key vars from the child so it can only use its stored login,
|
|
74
|
+
* matching `workflow run --subscription-only`. The observation is recorded
|
|
75
|
+
* against this mode, because a child that may fall back to an API key can
|
|
76
|
+
* behave differently from one that may not. */
|
|
77
|
+
subscriptionOnly?: boolean;
|
|
61
78
|
/** Test seam and alternate host probe. A null version means unavailable. */
|
|
62
79
|
versionProbe?: (binary: string) => string | null;
|
|
63
80
|
/** Test seam. Defaults to the adapter's production spawner. */
|
|
64
81
|
spawn?: (harness: HarnessId, prompt: string, timeoutMs: number) => Promise<SpawnResult>;
|
|
65
82
|
/** Test seam. Defaults to writing under the coord root. */
|
|
66
83
|
persist?: (record: HarnessAttestation) => void;
|
|
84
|
+
/**
|
|
85
|
+
* Also probe `filesystemPolicyProjection` (ADR 0041). Off by default because
|
|
86
|
+
* it costs two extra turns per capable harness, against one for everything
|
|
87
|
+
* else: the observation needs a control run to be readable at all.
|
|
88
|
+
*/
|
|
89
|
+
projection?: boolean;
|
|
90
|
+
/** Test seam for the projection probe. */
|
|
91
|
+
probeProjection?: typeof probeFilesystemProjection;
|
|
67
92
|
now?: () => Date;
|
|
68
93
|
}
|
|
69
94
|
|
|
@@ -74,6 +99,7 @@ export async function runHarnessAttestation(
|
|
|
74
99
|
const ids = opts.harnesses?.length ? [...opts.harnesses] : registry.ids();
|
|
75
100
|
const timeoutMs = opts.timeoutMs ?? DEFAULT_ATTESTATION_TIMEOUT_MS;
|
|
76
101
|
const cwd = opts.cwd ?? process.cwd();
|
|
102
|
+
const subscriptionOnly = opts.subscriptionOnly === true;
|
|
77
103
|
const versionProbe = opts.versionProbe ?? probeBinaryVersion;
|
|
78
104
|
const now = opts.now ?? (() => new Date());
|
|
79
105
|
const results: HarnessAttestationResult[] = [];
|
|
@@ -99,6 +125,7 @@ export async function runHarnessAttestation(
|
|
|
99
125
|
timeoutMs,
|
|
100
126
|
maxTurns: 1,
|
|
101
127
|
cwd,
|
|
128
|
+
subscriptionOnly,
|
|
102
129
|
});
|
|
103
130
|
} catch (error) {
|
|
104
131
|
results.push({
|
|
@@ -130,11 +157,25 @@ export async function runHarnessAttestation(
|
|
|
130
157
|
cost: result.costUsd !== undefined ? "supported" : "unsupported",
|
|
131
158
|
};
|
|
132
159
|
|
|
160
|
+
let projectionNote = "";
|
|
161
|
+
if (opts.projection) {
|
|
162
|
+
const probe = opts.probeProjection ?? probeFilesystemProjection;
|
|
163
|
+
const outcome = await probe(adapter, { timeoutMs, subscriptionOnly });
|
|
164
|
+
// An inconclusive probe records nothing for the dimension, leaving the
|
|
165
|
+
// declaration to stand on its own rather than dressing a non-observation
|
|
166
|
+
// as an observation.
|
|
167
|
+
if (outcome.observation !== "inconclusive") {
|
|
168
|
+
observations.filesystemPolicyProjection = outcome.observation;
|
|
169
|
+
}
|
|
170
|
+
projectionNote = `; projection ${outcome.observation}: ${outcome.detail}`;
|
|
171
|
+
}
|
|
172
|
+
|
|
133
173
|
const record = sealAttestation({
|
|
134
|
-
schema_version:
|
|
174
|
+
schema_version: ATTESTATION_SCHEMA_VERSION,
|
|
135
175
|
harness: id,
|
|
136
176
|
binary_version: binaryVersion,
|
|
137
177
|
profile_digest: profileDigest(adapter.profile),
|
|
178
|
+
subscription_only: subscriptionOnly,
|
|
138
179
|
observed_at: now().toISOString(),
|
|
139
180
|
observations,
|
|
140
181
|
});
|
|
@@ -148,7 +189,7 @@ export async function runHarnessAttestation(
|
|
|
148
189
|
binaryVersion,
|
|
149
190
|
observations,
|
|
150
191
|
durationMs: result.durationMs,
|
|
151
|
-
note: `observed on ${binaryVersion}`,
|
|
192
|
+
note: `observed on ${binaryVersion}${projectionNote}`,
|
|
152
193
|
});
|
|
153
194
|
}
|
|
154
195
|
|
|
@@ -25,12 +25,18 @@ import { monorepoRoot } from "../agents/coord-client.ts";
|
|
|
25
25
|
import { stableDigest } from "../workflow/durable-record.ts";
|
|
26
26
|
import type { CapabilitySupport, HarnessId, HarnessProfile } from "./types.ts";
|
|
27
27
|
|
|
28
|
-
export const ATTESTATION_SCHEMA_VERSION =
|
|
28
|
+
export const ATTESTATION_SCHEMA_VERSION = 2;
|
|
29
29
|
|
|
30
30
|
/** Dimensions one minimal live turn can honestly establish. Everything else
|
|
31
31
|
* needs a purpose-built scenario and stays outside the record rather than
|
|
32
32
|
* being guessed at. */
|
|
33
|
-
export const ATTESTABLE_DIMENSIONS = [
|
|
33
|
+
export const ATTESTABLE_DIMENSIONS = [
|
|
34
|
+
"invocation",
|
|
35
|
+
"finalResult",
|
|
36
|
+
"sessionId",
|
|
37
|
+
"cost",
|
|
38
|
+
"filesystemPolicyProjection",
|
|
39
|
+
] as const;
|
|
34
40
|
|
|
35
41
|
export type AttestableDimension = (typeof ATTESTABLE_DIMENSIONS)[number];
|
|
36
42
|
|
|
@@ -43,6 +49,10 @@ export interface HarnessAttestation {
|
|
|
43
49
|
/** Digest of the capability declaration at record time, so an edited
|
|
44
50
|
* declaration also invalidates the record. */
|
|
45
51
|
profile_digest: string;
|
|
52
|
+
/** The billing policy the probe ran under. A child launched with API keys
|
|
53
|
+
* scrubbed can behave differently from one that can fall back to them, so an
|
|
54
|
+
* observation only speaks for the mode it was made in. */
|
|
55
|
+
subscription_only: boolean;
|
|
46
56
|
observed_at: string;
|
|
47
57
|
/** Only what the probe actually saw. A dimension absent from this map was
|
|
48
58
|
* not observed, which is not the same as unsupported. */
|
|
@@ -164,9 +174,11 @@ export function isAttestationCurrent(
|
|
|
164
174
|
record: HarnessAttestation | null,
|
|
165
175
|
binaryVersion: string | null,
|
|
166
176
|
profile: HarnessProfile,
|
|
177
|
+
subscriptionOnly?: boolean,
|
|
167
178
|
): record is HarnessAttestation {
|
|
168
179
|
if (!record || !binaryVersion) return false;
|
|
169
180
|
if (record.binary_version !== binaryVersion) return false;
|
|
181
|
+
if (subscriptionOnly !== undefined && record.subscription_only !== subscriptionOnly) return false;
|
|
170
182
|
return record.profile_digest === profileDigest(profile);
|
|
171
183
|
}
|
|
172
184
|
|
|
@@ -194,6 +206,9 @@ export function harnessProofInputs(
|
|
|
194
206
|
profiles: readonly HarnessProfile[],
|
|
195
207
|
opts: AttestationStoreOptions & {
|
|
196
208
|
versionProbe: (binary: string) => string | null;
|
|
209
|
+
/** Billing policy this run will use, so a record made under the other mode
|
|
210
|
+
* is not cited as if it applied. */
|
|
211
|
+
subscriptionOnly?: boolean;
|
|
197
212
|
},
|
|
198
213
|
): {
|
|
199
214
|
harnessEvidence: Record<string, { toolEvidence: HarnessProfile["capabilities"]["toolEvidence"] }>;
|
|
@@ -220,7 +235,16 @@ export function harnessProofInputs(
|
|
|
220
235
|
// No coord root or unreadable store: run unattested rather than fail.
|
|
221
236
|
continue;
|
|
222
237
|
}
|
|
223
|
-
if (
|
|
238
|
+
if (
|
|
239
|
+
!isAttestationCurrent(
|
|
240
|
+
record,
|
|
241
|
+
opts.versionProbe(profile.binary),
|
|
242
|
+
profile,
|
|
243
|
+
opts.subscriptionOnly,
|
|
244
|
+
)
|
|
245
|
+
) {
|
|
246
|
+
continue;
|
|
247
|
+
}
|
|
224
248
|
harnessAttestations[profile.id] = {
|
|
225
249
|
binary_version: record.binary_version,
|
|
226
250
|
observed_at: record.observed_at,
|
|
@@ -65,6 +65,9 @@ export interface HarnessBenchOptions {
|
|
|
65
65
|
coordRoot?: string;
|
|
66
66
|
/** Test seam. Defaults to reading the attestation store. */
|
|
67
67
|
attestationReader?: (harness: HarnessId) => HarnessAttestation | null;
|
|
68
|
+
/** Only cite attestations recorded under this billing mode. Omitted means
|
|
69
|
+
* any mode is acceptable for reporting purposes. */
|
|
70
|
+
subscriptionOnly?: boolean;
|
|
68
71
|
}
|
|
69
72
|
|
|
70
73
|
const EMPTY_SUMMARY: Record<BenchVerdict, number> = {
|
|
@@ -101,7 +104,9 @@ function loadAttestation(
|
|
|
101
104
|
// No coord root, unreadable store: the bench still runs, just unattested.
|
|
102
105
|
return null;
|
|
103
106
|
}
|
|
104
|
-
return isAttestationCurrent(record, version, adapter.profile)
|
|
107
|
+
return isAttestationCurrent(record, version, adapter.profile, opts.subscriptionOnly)
|
|
108
|
+
? record
|
|
109
|
+
: null;
|
|
105
110
|
}
|
|
106
111
|
|
|
107
112
|
/** First version-shaped token in a string, or null when there is none.
|
|
@@ -280,6 +285,21 @@ function observeAdapter(
|
|
|
280
285
|
planningFailed = true;
|
|
281
286
|
}
|
|
282
287
|
|
|
288
|
+
// Plan a second invocation carrying a filesystem policy. Offline this can only
|
|
289
|
+
// show whether the adapter *renders* the projection, never whether the vendor
|
|
290
|
+
// enforces it; enforcement needs the live probe (ADR 0041). An adapter that
|
|
291
|
+
// declares no projection throws here, which is the correct rendering.
|
|
292
|
+
let projectionRendered = false;
|
|
293
|
+
try {
|
|
294
|
+
const projected = adapter.buildInvocation(
|
|
295
|
+
{ ...request, filesystemPolicy: { mode: "read-only" } },
|
|
296
|
+
"/harnery-bench/final.txt",
|
|
297
|
+
).argv;
|
|
298
|
+
projectionRendered = projected.join(" ") !== argv.join(" ");
|
|
299
|
+
} catch {
|
|
300
|
+
projectionRendered = false;
|
|
301
|
+
}
|
|
302
|
+
|
|
283
303
|
let normalized: SpawnResult | null = null;
|
|
284
304
|
try {
|
|
285
305
|
normalized = adapter.normalizeResult(adapter.fixture.raw);
|
|
@@ -331,6 +351,7 @@ function observeAdapter(
|
|
|
331
351
|
normalized && "toolEvidence" in normalized ? "supported" : "unsupported",
|
|
332
352
|
),
|
|
333
353
|
policyMapping: NOT_CHECKED,
|
|
354
|
+
filesystemPolicyProjection: fromAdapter(projectionRendered ? "supported" : "unsupported"),
|
|
334
355
|
interruption: NOT_CHECKED,
|
|
335
356
|
streaming: NOT_CHECKED,
|
|
336
357
|
steering: NOT_CHECKED,
|
|
@@ -16,6 +16,7 @@ function capabilities(overrides: Partial<HarnessCapabilities>): HarnessCapabilit
|
|
|
16
16
|
cost: unsupported(),
|
|
17
17
|
toolEvidence: unsupported("The final-result adapter does not retain tool events."),
|
|
18
18
|
policyMapping: unsupported("No ALLOW/DENY/ASK translation at the workflow boundary."),
|
|
19
|
+
filesystemPolicyProjection: unsupported("The adapter declares no sandbox projection."),
|
|
19
20
|
interruption: partial("Timeout kills the subprocess; no caller-driven interrupt handle."),
|
|
20
21
|
streaming: unsupported("Workflow children return one normalized final result."),
|
|
21
22
|
steering: unsupported("One prompt is fixed at subprocess launch."),
|
|
@@ -43,7 +44,7 @@ export const BUILTIN_HARNESS_PROFILES = {
|
|
|
43
44
|
authModel: "own-auth",
|
|
44
45
|
modelFamily: "claude",
|
|
45
46
|
effortValues: ["low", "medium", "high", "xhigh", "max"],
|
|
46
|
-
verified: { date: "2026-07-
|
|
47
|
+
verified: { date: "2026-07-25", version: "2.1.197 (Claude Code)" },
|
|
47
48
|
capabilities: capabilities({
|
|
48
49
|
effortSelection: supported("Mapped to `--effort <level>`."),
|
|
49
50
|
maxTurns: supported("Mapped to `--max-turns <n>`."),
|
|
@@ -69,9 +70,19 @@ export const BUILTIN_HARNESS_PROFILES = {
|
|
|
69
70
|
authModel: "own-auth",
|
|
70
71
|
modelFamily: "gpt",
|
|
71
72
|
effortValues: ["none", "minimal", "low", "medium", "high", "xhigh"],
|
|
72
|
-
verified: { date: "2026-07-
|
|
73
|
+
verified: { date: "2026-07-25", version: "codex-cli 0.144.5" },
|
|
74
|
+
// Verified against codex-cli 0.144.5: `--sandbox <mode>` plus
|
|
75
|
+
// `sandbox_workspace_write.writable_roots`. Without the writable-root entry
|
|
76
|
+
// a workspace-write child still cannot write a repository's .git directory.
|
|
77
|
+
sandboxProjection: {
|
|
78
|
+
modes: { "read-only": "read-only", "workspace-write": "workspace-write" },
|
|
79
|
+
writableRoots: true,
|
|
80
|
+
},
|
|
73
81
|
capabilities: capabilities({
|
|
74
82
|
effortSelection: supported('Mapped to `-c model_reasoning_effort="<level>"`.'),
|
|
83
|
+
filesystemPolicyProjection: supported(
|
|
84
|
+
"Mode renders to --sandbox; writable roots to sandbox_workspace_write.writable_roots.",
|
|
85
|
+
),
|
|
75
86
|
maxTurns: unsupported("codex exec exposes no turn-ceiling flag."),
|
|
76
87
|
sessionId: unsupported("--output-last-message carries no session id."),
|
|
77
88
|
cost: unsupported("The final-message path carries no usage or cost."),
|
|
@@ -95,7 +106,7 @@ export const BUILTIN_HARNESS_PROFILES = {
|
|
|
95
106
|
authModel: "own-auth",
|
|
96
107
|
modelFamily: "multi",
|
|
97
108
|
effortValues: [],
|
|
98
|
-
verified: { date: "2026-07-
|
|
109
|
+
verified: { date: "2026-07-25", version: "2026.07.23-e383d2b" },
|
|
99
110
|
capabilities: capabilities({
|
|
100
111
|
effortSelection: unsupported(
|
|
101
112
|
"Cursor embeds effort in some parameterized model ids; Harnery does not rewrite model ids.",
|
|
@@ -16,6 +16,7 @@ export const HARNESS_CAPABILITY_DIMENSIONS = [
|
|
|
16
16
|
"cost",
|
|
17
17
|
"toolEvidence",
|
|
18
18
|
"policyMapping",
|
|
19
|
+
"filesystemPolicyProjection",
|
|
19
20
|
"interruption",
|
|
20
21
|
"streaming",
|
|
21
22
|
"steering",
|
|
@@ -38,6 +39,17 @@ export interface CapabilityClaim {
|
|
|
38
39
|
|
|
39
40
|
export type HarnessCapabilities = Record<HarnessCapabilityDimension, CapabilityClaim>;
|
|
40
41
|
|
|
42
|
+
/** What one harness can represent of a filesystem-policy projection
|
|
43
|
+
* (ADR 0039). `null` means the harness does not distinguish that, which is a
|
|
44
|
+
* fact about the harness rather than something to paper over: a projection the
|
|
45
|
+
* adapter would silently drop must be refused instead. */
|
|
46
|
+
export interface HarnessSandboxProjection {
|
|
47
|
+
/** Native representation per canonical mode, or null when unrepresentable. */
|
|
48
|
+
modes: Record<"read-only" | "workspace-write", string | null>;
|
|
49
|
+
/** Whether the harness accepts an explicit writable-root set. */
|
|
50
|
+
writableRoots: boolean;
|
|
51
|
+
}
|
|
52
|
+
|
|
41
53
|
export interface HarnessProfile {
|
|
42
54
|
id: HarnessId;
|
|
43
55
|
displayName: string;
|
|
@@ -52,6 +64,9 @@ export interface HarnessProfile {
|
|
|
52
64
|
capabilities: HarnessCapabilities;
|
|
53
65
|
/** The last real vendor CLI contract used to validate this declaration. */
|
|
54
66
|
verified?: { date: string; version: string };
|
|
67
|
+
/** How this adapter projects host filesystem policy into the vendor's own
|
|
68
|
+
* sandbox (ADR 0039). Absent means it cannot project any of it. */
|
|
69
|
+
sandboxProjection?: HarnessSandboxProjection;
|
|
55
70
|
}
|
|
56
71
|
|
|
57
72
|
/** Fully planned child invocation. `resultFile` is used by adapters such as
|