harnery 0.25.0 → 0.27.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (81) hide show
  1. package/dist/commands/harness.d.ts.map +1 -1
  2. package/dist/commands/harness.js +12 -2
  3. package/dist/commands/workflow.d.ts.map +1 -1
  4. package/dist/commands/workflow.js +64 -0
  5. package/dist/core/harnesses/attest-projection.d.ts +44 -0
  6. package/dist/core/harnesses/attest-projection.d.ts.map +1 -0
  7. package/dist/core/harnesses/attest-projection.js +113 -0
  8. package/dist/core/harnesses/attest.d.ts +14 -0
  9. package/dist/core/harnesses/attest.d.ts.map +1 -1
  10. package/dist/core/harnesses/attest.js +27 -4
  11. package/dist/core/harnesses/attestation.d.ts +10 -3
  12. package/dist/core/harnesses/attestation.d.ts.map +1 -1
  13. package/dist/core/harnesses/attestation.js +13 -4
  14. package/dist/core/harnesses/bench.d.ts +3 -0
  15. package/dist/core/harnesses/bench.d.ts.map +1 -1
  16. package/dist/core/harnesses/bench.js +16 -1
  17. package/dist/core/harnesses/profiles.d.ts +13 -6
  18. package/dist/core/harnesses/profiles.d.ts.map +1 -1
  19. package/dist/core/harnesses/profiles.js +12 -3
  20. package/dist/core/harnesses/types.d.ts +14 -1
  21. package/dist/core/harnesses/types.d.ts.map +1 -1
  22. package/dist/core/harnesses/types.js +1 -0
  23. package/dist/core/workflow/engine.d.ts.map +1 -1
  24. package/dist/core/workflow/engine.js +41 -0
  25. package/dist/core/workflow/index.d.ts +1 -1
  26. package/dist/core/workflow/index.d.ts.map +1 -1
  27. package/dist/core/workflow/index.js +1 -1
  28. package/dist/core/workflow/proof.d.ts +3 -1
  29. package/dist/core/workflow/proof.d.ts.map +1 -1
  30. package/dist/core/workflow/proof.js +1 -0
  31. package/dist/core/workflow/sandbox-projection.d.ts +57 -0
  32. package/dist/core/workflow/sandbox-projection.d.ts.map +1 -0
  33. package/dist/core/workflow/sandbox-projection.js +95 -0
  34. package/dist/core/workflow/spawn-claude.d.ts.map +1 -1
  35. package/dist/core/workflow/spawn-claude.js +8 -2
  36. package/dist/core/workflow/spawn-codex.d.ts.map +1 -1
  37. package/dist/core/workflow/spawn-codex.js +12 -3
  38. package/dist/core/workflow/spawn-cursor.d.ts.map +1 -1
  39. package/dist/core/workflow/spawn-cursor.js +8 -2
  40. package/dist/core/workflow/spawn-failure.d.ts +18 -0
  41. package/dist/core/workflow/spawn-failure.d.ts.map +1 -0
  42. package/dist/core/workflow/spawn-failure.js +24 -0
  43. package/dist/core/workflow/types.d.ts +51 -0
  44. package/dist/core/workflow/types.d.ts.map +1 -1
  45. package/dist/core/workflow/workspaces/index.d.ts +2 -0
  46. package/dist/core/workflow/workspaces/index.d.ts.map +1 -1
  47. package/dist/core/workflow/workspaces/index.js +1 -0
  48. package/dist/core/workflow/workspaces/inspect.d.ts.map +1 -1
  49. package/dist/core/workflow/workspaces/inspect.js +14 -0
  50. package/dist/core/workflow/workspaces/reclaim.d.ts +50 -0
  51. package/dist/core/workflow/workspaces/reclaim.d.ts.map +1 -0
  52. package/dist/core/workflow/workspaces/reclaim.js +88 -0
  53. package/dist/lib/tunnel/error-page.d.ts +15 -0
  54. package/dist/lib/tunnel/error-page.d.ts.map +1 -0
  55. package/dist/lib/tunnel/error-page.js +98 -0
  56. package/dist/lib/tunnel/gate.d.ts +1 -12
  57. package/dist/lib/tunnel/gate.d.ts.map +1 -1
  58. package/dist/lib/tunnel/gate.js +59 -4
  59. package/package.json +4 -2
  60. package/src/commands/harness.ts +20 -2
  61. package/src/commands/workflow.ts +74 -0
  62. package/src/core/harnesses/attest-projection.ts +151 -0
  63. package/src/core/harnesses/attest.ts +45 -4
  64. package/src/core/harnesses/attestation.ts +27 -3
  65. package/src/core/harnesses/bench.ts +22 -1
  66. package/src/core/harnesses/profiles.ts +14 -3
  67. package/src/core/harnesses/types.ts +15 -0
  68. package/src/core/workflow/engine.ts +58 -0
  69. package/src/core/workflow/index.ts +1 -0
  70. package/src/core/workflow/proof.ts +4 -0
  71. package/src/core/workflow/sandbox-projection.ts +150 -0
  72. package/src/core/workflow/spawn-claude.ts +12 -2
  73. package/src/core/workflow/spawn-codex.ts +19 -3
  74. package/src/core/workflow/spawn-cursor.ts +12 -2
  75. package/src/core/workflow/spawn-failure.ts +28 -0
  76. package/src/core/workflow/types.ts +54 -0
  77. package/src/core/workflow/workspaces/index.ts +2 -0
  78. package/src/core/workflow/workspaces/inspect.ts +15 -0
  79. package/src/core/workflow/workspaces/reclaim.ts +109 -0
  80. package/src/lib/tunnel/error-page.ts +114 -0
  81. package/src/lib/tunnel/gate.ts +66 -3
@@ -1,5 +1,6 @@
1
1
  import type { Command } from "commander";
2
2
  import type { EmitContext } from "../commander.ts";
3
+ import { workflowSubscriptionOnly } from "../core/config.ts";
3
4
  import {
4
5
  type BenchResult,
5
6
  createBuiltinHarnessRegistry,
@@ -22,6 +23,8 @@ interface BenchOpts extends FormatOpts {
22
23
  interface AttestOpts extends FormatOpts {
23
24
  timeout?: string;
24
25
  yes?: boolean;
26
+ subscriptionOnly?: boolean;
27
+ projection?: boolean;
25
28
  }
26
29
 
27
30
  const registry = createBuiltinHarnessRegistry();
@@ -94,7 +97,15 @@ export function registerHarnessCommand(program: Command, emit: EmitContext): voi
94
97
  "Record what the installed vendor CLIs actually do. Runs one real model turn each; needs --yes.",
95
98
  )
96
99
  .option("--yes", "Confirm that this spends real vendor tokens")
100
+ .option(
101
+ "--subscription-only",
102
+ "Scrub API-key vars so the child uses its stored login (repo default via config.jsonc workflow.subscriptionOnly)",
103
+ )
97
104
  .option("--timeout <ms>", "Per-harness probe timeout in milliseconds")
105
+ .option(
106
+ "--projection",
107
+ "Also probe whether a declared sandbox is enforced; costs two extra turns per capable harness",
108
+ )
98
109
  .option("--json", "Machine-readable attestation report")
99
110
  .action(async (harnesses: string[], opts: AttestOpts) => {
100
111
  if (!opts.yes) {
@@ -116,7 +127,13 @@ export function registerHarnessCommand(program: Command, emit: EmitContext): voi
116
127
  return;
117
128
  }
118
129
  try {
119
- const report = await runHarnessAttestation(registry, { harnesses, timeoutMs });
130
+ const subscriptionOnly = opts.subscriptionOnly || workflowSubscriptionOnly();
131
+ const report = await runHarnessAttestation(registry, {
132
+ harnesses,
133
+ timeoutMs,
134
+ subscriptionOnly,
135
+ projection: opts.projection === true,
136
+ });
120
137
  if (opts.json) {
121
138
  emit.config({ format: "json" });
122
139
  emit.data(report);
@@ -151,10 +168,11 @@ export function registerHarnessCommand(program: Command, emit: EmitContext): voi
151
168
  }
152
169
  emit.text(
153
170
  renderTable(
154
- ["HARNESS", "VERSION", "OBSERVED AT", "OBSERVATIONS"],
171
+ ["HARNESS", "VERSION", "BILLING", "OBSERVED AT", "OBSERVATIONS"],
155
172
  records.map((record) => [
156
173
  record.harness,
157
174
  record.binary_version,
175
+ record.subscription_only ? "subscription" : "any",
158
176
  record.observed_at,
159
177
  Object.entries(record.observations)
160
178
  .map(([dimension, support]) => `${dimension}=${support}`)
@@ -70,6 +70,10 @@ interface WorkflowConfirmedMutationOpts {
70
70
  json?: boolean;
71
71
  }
72
72
 
73
+ interface WorkflowReclaimOpts extends WorkflowConfirmedMutationOpts {
74
+ discard?: boolean;
75
+ }
76
+
73
77
  export function registerWorkflowCommand(program: Command, emit: EmitContext): void {
74
78
  const registry = createBuiltinHarnessRegistry();
75
79
  const harnesses = registry.ids();
@@ -594,6 +598,76 @@ export function registerWorkflowCommand(program: Command, emit: EmitContext): vo
594
598
  }
595
599
  });
596
600
 
601
+ workflow
602
+ .command("reclaim <run-id>")
603
+ .description(
604
+ "Resolve a workspace preserved because it was dirty: salvage the work to its branch, then release.",
605
+ )
606
+ .option("--yes", "Confirm that this commits or discards uncommitted work, then releases")
607
+ .option("--discard", "Throw the uncommitted work away instead of salvaging it")
608
+ .option("--json", "Emit the reclaim preparation and cleanup result as JSON")
609
+ .action(async (runId: string, opts: WorkflowReclaimOpts) => {
610
+ const coordRoot = findCoordRoot();
611
+ if (!coordRoot) {
612
+ emit.error({
613
+ code: "no_coord_root",
614
+ message: "no .harnery/ coordination root found; run `init` first",
615
+ });
616
+ process.exit(1);
617
+ }
618
+ if (!opts.yes) {
619
+ emit.error({
620
+ code: "reclaim_confirmation_required",
621
+ message: opts.discard
622
+ ? "reclaim --discard permanently destroys uncommitted work; pass --yes to confirm"
623
+ : "workspace reclaim commits uncommitted work to its branch and then releases the worktree; pass --yes to confirm",
624
+ });
625
+ process.exit(1);
626
+ }
627
+ try {
628
+ const { inspectWorkflowWorkspace, prepareReclaim, cleanupWorkspace } = await import(
629
+ "../core/workflow/index.ts"
630
+ );
631
+ const inspection = inspectWorkflowWorkspace(coordRoot, runId);
632
+ if (!inspection.ok) {
633
+ emit.error({ code: "workflow_reclaim_failed", message: inspection.error });
634
+ process.exit(1);
635
+ }
636
+ const allocation = inspection.value.allocation;
637
+ if (!allocation) {
638
+ emit.error({
639
+ code: "workflow_reclaim_unavailable",
640
+ message: `run ${runId} has no provider-owned Git workspace to reclaim`,
641
+ });
642
+ process.exit(1);
643
+ }
644
+ const preparation = prepareReclaim({
645
+ worktreePath: allocation.active_root,
646
+ mode: opts.discard ? "discard" : "salvage",
647
+ runId,
648
+ });
649
+ // Cleanup still owns removal. Reclaim only changes whether the tree it
650
+ // finds is dirty, so there is one audited release path, not two.
651
+ const provider = await builtInProviderForRun(coordRoot, runId);
652
+ const cleanup = await cleanupWorkspace({ coordRoot, runId, provider });
653
+ if (opts.json) {
654
+ emit.config({ format: "json" });
655
+ emit.data({ preparation, cleanup });
656
+ return;
657
+ }
658
+ emit.text(
659
+ `reclaim ${preparation.action}: ${preparation.detail}\n` +
660
+ `workspace cleanup ${cleanup.status}: ${cleanup.binding_id}\n`,
661
+ );
662
+ } catch (error) {
663
+ emit.error({
664
+ code: "workflow_reclaim_failed",
665
+ message: error instanceof Error ? error.message : String(error),
666
+ });
667
+ process.exit(1);
668
+ }
669
+ });
670
+
597
671
  const approvals = workflow
598
672
  .command("approvals")
599
673
  .description("Inspect and resolve durable workflow policy approvals.");
@@ -0,0 +1,151 @@
1
+ /**
2
+ * Live probe for the `filesystemPolicyProjection` capability (ADR 0041).
3
+ *
4
+ * Every other attested dimension can be read off a single successful turn: the
5
+ * result either carries a session id or it does not. Projection cannot. A
6
+ * sandbox that is declared but not enforced looks exactly like a sandbox that is
7
+ * enforced, because in both cases the CLI accepts the flag and exits zero. The
8
+ * only way to tell them apart is to give a child something to write and look at
9
+ * the filesystem afterwards.
10
+ *
11
+ * That introduces its own confound. If the treatment child never attempts the
12
+ * write, the sentinel is absent for a reason that has nothing to do with the
13
+ * sandbox, and absence would be read as enforcement. So the probe runs a control
14
+ * first, under a mode that permits the write. Only a control that actually
15
+ * produced its sentinel licenses any reading of the treatment run; otherwise the
16
+ * probe reports inconclusive and records nothing, which is the same prerequisite
17
+ * rule the main attestation applies to a failed turn.
18
+ */
19
+
20
+ import { existsSync, mkdtempSync, rmSync } from "node:fs";
21
+ import { tmpdir } from "node:os";
22
+ import { join } from "node:path";
23
+ import type { SpawnRequest, SpawnResult } from "../workflow/types.ts";
24
+ import type { HarnessAdapter } from "./types.ts";
25
+
26
+ /** Distinct names so a control artifact can never be mistaken for a treatment
27
+ * one, even if a child writes both or the directory is reused. */
28
+ const CONTROL_SENTINEL = "harnery-projection-control.txt";
29
+ const TREATMENT_SENTINEL = "harnery-projection-treatment.txt";
30
+
31
+ /** A child needs a few turns to run a shell command and report back; one turn is
32
+ * not enough to both act and answer. */
33
+ const PROJECTION_MAX_TURNS = 4;
34
+
35
+ export function projectionPrompt(sentinel: string): string {
36
+ return `Create a file named ${sentinel} in the current directory containing the text ok. Then reply with the single word: done`;
37
+ }
38
+
39
+ export type ProjectionObservation = "supported" | "unsupported" | "inconclusive";
40
+
41
+ export interface ProjectionProbeResult {
42
+ observation: ProjectionObservation;
43
+ /** Why, in one line, suitable for an attestation note or a report row. */
44
+ detail: string;
45
+ }
46
+
47
+ export interface ProjectionProbeOptions {
48
+ timeoutMs: number;
49
+ subscriptionOnly: boolean;
50
+ /** Test seam. Defaults to the adapter's production spawner. */
51
+ spawn?: (request: SpawnRequest) => Promise<SpawnResult>;
52
+ /** Test seam. Defaults to a fresh directory under the OS temp root. */
53
+ workdir?: string;
54
+ }
55
+
56
+ /**
57
+ * Run the control/treatment pair and report what the filesystem showed.
58
+ *
59
+ * Returns `inconclusive` rather than throwing for every expected failure, so a
60
+ * probe that cannot reach a verdict degrades into "nothing recorded" instead of
61
+ * failing the whole attestation sweep.
62
+ */
63
+ export async function probeFilesystemProjection(
64
+ adapter: HarnessAdapter,
65
+ opts: ProjectionProbeOptions,
66
+ ): Promise<ProjectionProbeResult> {
67
+ if (!adapter.profile.sandboxProjection) {
68
+ // Nothing to observe: the adapter refuses a projection before launch, which
69
+ // is a fact about our own code and is already covered by unit tests. Spending
70
+ // a vendor turn here would attest nothing.
71
+ return {
72
+ observation: "inconclusive",
73
+ detail: "the adapter declares no sandbox projection, so there is nothing to observe live",
74
+ };
75
+ }
76
+
77
+ const spawn = opts.spawn ?? ((request: SpawnRequest) => adapter.spawn(request));
78
+ const ownsWorkdir = !opts.workdir;
79
+ const workdir = opts.workdir ?? mkdtempSync(join(tmpdir(), "harnery-projection-"));
80
+
81
+ try {
82
+ const control = await runTurn(spawn, workdir, CONTROL_SENTINEL, "workspace-write", opts);
83
+ if (!control.ok) {
84
+ return {
85
+ observation: "inconclusive",
86
+ detail: `the control turn did not complete (${control.detail})`,
87
+ };
88
+ }
89
+ if (!existsSync(join(workdir, CONTROL_SENTINEL))) {
90
+ return {
91
+ observation: "inconclusive",
92
+ detail:
93
+ "the control child did not create its file even though writing was permitted, so an absent treatment file would prove nothing",
94
+ };
95
+ }
96
+
97
+ const treatment = await runTurn(spawn, workdir, TREATMENT_SENTINEL, "read-only", opts);
98
+ // A read-only child may well fail its turn: refusing the write is the point.
99
+ // So unlike the control, a failed treatment turn is still evidence, and only
100
+ // the filesystem decides.
101
+ const wrote = existsSync(join(workdir, TREATMENT_SENTINEL));
102
+ if (wrote) {
103
+ return {
104
+ observation: "unsupported",
105
+ detail:
106
+ "the child wrote under a read-only projection, so the declared mode is not enforced",
107
+ };
108
+ }
109
+ return {
110
+ observation: "supported",
111
+ detail: `a read-only projection blocked a write the same child performed when permitted${
112
+ treatment.ok ? "" : ` (treatment turn also reported: ${treatment.detail})`
113
+ }`,
114
+ };
115
+ } finally {
116
+ if (ownsWorkdir) rmSync(workdir, { recursive: true, force: true });
117
+ }
118
+ }
119
+
120
+ async function runTurn(
121
+ spawn: (request: SpawnRequest) => Promise<SpawnResult>,
122
+ cwd: string,
123
+ sentinel: string,
124
+ mode: "read-only" | "workspace-write",
125
+ opts: ProjectionProbeOptions,
126
+ ): Promise<{ ok: boolean; detail: string }> {
127
+ try {
128
+ const result = await spawn({
129
+ prompt: projectionPrompt(sentinel),
130
+ timeoutMs: opts.timeoutMs,
131
+ maxTurns: PROJECTION_MAX_TURNS,
132
+ cwd,
133
+ subscriptionOnly: opts.subscriptionOnly,
134
+ filesystemPolicy: { mode },
135
+ });
136
+ return { ok: result.ok, detail: result.ok ? "completed" : boundedDetail(result.error) };
137
+ } catch (error) {
138
+ return { ok: false, detail: boundedDetail((error as Error).message) };
139
+ }
140
+ }
141
+
142
+ const MAX_DETAIL_CHARS = 160;
143
+
144
+ /** Same tail-preserving rule as the main probe: a CLI prints its banner first
145
+ * and the reason it failed last. */
146
+ function boundedDetail(reason: string | undefined): string {
147
+ if (!reason) return "no error reported";
148
+ const collapsed = reason.replace(/\s+/g, " ").trim();
149
+ if (!collapsed) return "no error reported";
150
+ return collapsed.length > MAX_DETAIL_CHARS ? `…${collapsed.slice(-MAX_DETAIL_CHARS)}` : collapsed;
151
+ }
@@ -7,8 +7,14 @@
7
7
  */
8
8
 
9
9
  import type { SpawnResult } from "../workflow/types.ts";
10
+ import { probeFilesystemProjection } from "./attest-projection.ts";
10
11
  import type { AttestableDimension, HarnessAttestation } from "./attestation.ts";
11
- import { profileDigest, sealAttestation, writeAttestation } from "./attestation.ts";
12
+ import {
13
+ ATTESTATION_SCHEMA_VERSION,
14
+ profileDigest,
15
+ sealAttestation,
16
+ writeAttestation,
17
+ } from "./attestation.ts";
12
18
  import { probeBinaryVersion } from "./bench.ts";
13
19
  import type { HarnessRegistry } from "./registry.ts";
14
20
  import type { CapabilitySupport, HarnessId } from "./types.ts";
@@ -24,11 +30,17 @@ export const DEFAULT_ATTESTATION_TIMEOUT_MS = 120_000;
24
30
  * short line and the echoed prompt is removed. */
25
31
  const MAX_NOTE_REASON_CHARS = 200;
26
32
 
33
+ /** Keep the TAIL, not the head. A CLI prints its banner, config, and startup
34
+ * warnings first and the reason it actually failed last, so truncating from the
35
+ * front reliably preserves the noise and discards the answer. Learned the hard
36
+ * way: a head-truncated note once surfaced a cosmetic startup warning while
37
+ * hiding the real "out of credits" failure on the final line. */
27
38
  function boundedReason(reason: string | undefined): string {
28
39
  if (!reason) return "no error reported";
29
40
  const collapsed = reason.split(ATTESTATION_PROMPT).join("<prompt>").replace(/\s+/g, " ").trim();
41
+ if (!collapsed) return "no error reported";
30
42
  return collapsed.length > MAX_NOTE_REASON_CHARS
31
- ? `${collapsed.slice(0, MAX_NOTE_REASON_CHARS)}…`
43
+ ? `…${collapsed.slice(-MAX_NOTE_REASON_CHARS)}`
32
44
  : collapsed;
33
45
  }
34
46
 
@@ -58,12 +70,25 @@ export interface RunHarnessAttestationOptions {
58
70
  timeoutMs?: number;
59
71
  cwd?: string;
60
72
  coordRoot?: string;
73
+ /** Scrub API-key vars from the child so it can only use its stored login,
74
+ * matching `workflow run --subscription-only`. The observation is recorded
75
+ * against this mode, because a child that may fall back to an API key can
76
+ * behave differently from one that may not. */
77
+ subscriptionOnly?: boolean;
61
78
  /** Test seam and alternate host probe. A null version means unavailable. */
62
79
  versionProbe?: (binary: string) => string | null;
63
80
  /** Test seam. Defaults to the adapter's production spawner. */
64
81
  spawn?: (harness: HarnessId, prompt: string, timeoutMs: number) => Promise<SpawnResult>;
65
82
  /** Test seam. Defaults to writing under the coord root. */
66
83
  persist?: (record: HarnessAttestation) => void;
84
+ /**
85
+ * Also probe `filesystemPolicyProjection` (ADR 0041). Off by default because
86
+ * it costs two extra turns per capable harness, against one for everything
87
+ * else: the observation needs a control run to be readable at all.
88
+ */
89
+ projection?: boolean;
90
+ /** Test seam for the projection probe. */
91
+ probeProjection?: typeof probeFilesystemProjection;
67
92
  now?: () => Date;
68
93
  }
69
94
 
@@ -74,6 +99,7 @@ export async function runHarnessAttestation(
74
99
  const ids = opts.harnesses?.length ? [...opts.harnesses] : registry.ids();
75
100
  const timeoutMs = opts.timeoutMs ?? DEFAULT_ATTESTATION_TIMEOUT_MS;
76
101
  const cwd = opts.cwd ?? process.cwd();
102
+ const subscriptionOnly = opts.subscriptionOnly === true;
77
103
  const versionProbe = opts.versionProbe ?? probeBinaryVersion;
78
104
  const now = opts.now ?? (() => new Date());
79
105
  const results: HarnessAttestationResult[] = [];
@@ -99,6 +125,7 @@ export async function runHarnessAttestation(
99
125
  timeoutMs,
100
126
  maxTurns: 1,
101
127
  cwd,
128
+ subscriptionOnly,
102
129
  });
103
130
  } catch (error) {
104
131
  results.push({
@@ -130,11 +157,25 @@ export async function runHarnessAttestation(
130
157
  cost: result.costUsd !== undefined ? "supported" : "unsupported",
131
158
  };
132
159
 
160
+ let projectionNote = "";
161
+ if (opts.projection) {
162
+ const probe = opts.probeProjection ?? probeFilesystemProjection;
163
+ const outcome = await probe(adapter, { timeoutMs, subscriptionOnly });
164
+ // An inconclusive probe records nothing for the dimension, leaving the
165
+ // declaration to stand on its own rather than dressing a non-observation
166
+ // as an observation.
167
+ if (outcome.observation !== "inconclusive") {
168
+ observations.filesystemPolicyProjection = outcome.observation;
169
+ }
170
+ projectionNote = `; projection ${outcome.observation}: ${outcome.detail}`;
171
+ }
172
+
133
173
  const record = sealAttestation({
134
- schema_version: 1,
174
+ schema_version: ATTESTATION_SCHEMA_VERSION,
135
175
  harness: id,
136
176
  binary_version: binaryVersion,
137
177
  profile_digest: profileDigest(adapter.profile),
178
+ subscription_only: subscriptionOnly,
138
179
  observed_at: now().toISOString(),
139
180
  observations,
140
181
  });
@@ -148,7 +189,7 @@ export async function runHarnessAttestation(
148
189
  binaryVersion,
149
190
  observations,
150
191
  durationMs: result.durationMs,
151
- note: `observed on ${binaryVersion}`,
192
+ note: `observed on ${binaryVersion}${projectionNote}`,
152
193
  });
153
194
  }
154
195
 
@@ -25,12 +25,18 @@ import { monorepoRoot } from "../agents/coord-client.ts";
25
25
  import { stableDigest } from "../workflow/durable-record.ts";
26
26
  import type { CapabilitySupport, HarnessId, HarnessProfile } from "./types.ts";
27
27
 
28
- export const ATTESTATION_SCHEMA_VERSION = 1;
28
+ export const ATTESTATION_SCHEMA_VERSION = 2;
29
29
 
30
30
  /** Dimensions one minimal live turn can honestly establish. Everything else
31
31
  * needs a purpose-built scenario and stays outside the record rather than
32
32
  * being guessed at. */
33
- export const ATTESTABLE_DIMENSIONS = ["invocation", "finalResult", "sessionId", "cost"] as const;
33
+ export const ATTESTABLE_DIMENSIONS = [
34
+ "invocation",
35
+ "finalResult",
36
+ "sessionId",
37
+ "cost",
38
+ "filesystemPolicyProjection",
39
+ ] as const;
34
40
 
35
41
  export type AttestableDimension = (typeof ATTESTABLE_DIMENSIONS)[number];
36
42
 
@@ -43,6 +49,10 @@ export interface HarnessAttestation {
43
49
  /** Digest of the capability declaration at record time, so an edited
44
50
  * declaration also invalidates the record. */
45
51
  profile_digest: string;
52
+ /** The billing policy the probe ran under. A child launched with API keys
53
+ * scrubbed can behave differently from one that can fall back to them, so an
54
+ * observation only speaks for the mode it was made in. */
55
+ subscription_only: boolean;
46
56
  observed_at: string;
47
57
  /** Only what the probe actually saw. A dimension absent from this map was
48
58
  * not observed, which is not the same as unsupported. */
@@ -164,9 +174,11 @@ export function isAttestationCurrent(
164
174
  record: HarnessAttestation | null,
165
175
  binaryVersion: string | null,
166
176
  profile: HarnessProfile,
177
+ subscriptionOnly?: boolean,
167
178
  ): record is HarnessAttestation {
168
179
  if (!record || !binaryVersion) return false;
169
180
  if (record.binary_version !== binaryVersion) return false;
181
+ if (subscriptionOnly !== undefined && record.subscription_only !== subscriptionOnly) return false;
170
182
  return record.profile_digest === profileDigest(profile);
171
183
  }
172
184
 
@@ -194,6 +206,9 @@ export function harnessProofInputs(
194
206
  profiles: readonly HarnessProfile[],
195
207
  opts: AttestationStoreOptions & {
196
208
  versionProbe: (binary: string) => string | null;
209
+ /** Billing policy this run will use, so a record made under the other mode
210
+ * is not cited as if it applied. */
211
+ subscriptionOnly?: boolean;
197
212
  },
198
213
  ): {
199
214
  harnessEvidence: Record<string, { toolEvidence: HarnessProfile["capabilities"]["toolEvidence"] }>;
@@ -220,7 +235,16 @@ export function harnessProofInputs(
220
235
  // No coord root or unreadable store: run unattested rather than fail.
221
236
  continue;
222
237
  }
223
- if (!isAttestationCurrent(record, opts.versionProbe(profile.binary), profile)) continue;
238
+ if (
239
+ !isAttestationCurrent(
240
+ record,
241
+ opts.versionProbe(profile.binary),
242
+ profile,
243
+ opts.subscriptionOnly,
244
+ )
245
+ ) {
246
+ continue;
247
+ }
224
248
  harnessAttestations[profile.id] = {
225
249
  binary_version: record.binary_version,
226
250
  observed_at: record.observed_at,
@@ -65,6 +65,9 @@ export interface HarnessBenchOptions {
65
65
  coordRoot?: string;
66
66
  /** Test seam. Defaults to reading the attestation store. */
67
67
  attestationReader?: (harness: HarnessId) => HarnessAttestation | null;
68
+ /** Only cite attestations recorded under this billing mode. Omitted means
69
+ * any mode is acceptable for reporting purposes. */
70
+ subscriptionOnly?: boolean;
68
71
  }
69
72
 
70
73
  const EMPTY_SUMMARY: Record<BenchVerdict, number> = {
@@ -101,7 +104,9 @@ function loadAttestation(
101
104
  // No coord root, unreadable store: the bench still runs, just unattested.
102
105
  return null;
103
106
  }
104
- return isAttestationCurrent(record, version, adapter.profile) ? record : null;
107
+ return isAttestationCurrent(record, version, adapter.profile, opts.subscriptionOnly)
108
+ ? record
109
+ : null;
105
110
  }
106
111
 
107
112
  /** First version-shaped token in a string, or null when there is none.
@@ -280,6 +285,21 @@ function observeAdapter(
280
285
  planningFailed = true;
281
286
  }
282
287
 
288
+ // Plan a second invocation carrying a filesystem policy. Offline this can only
289
+ // show whether the adapter *renders* the projection, never whether the vendor
290
+ // enforces it; enforcement needs the live probe (ADR 0041). An adapter that
291
+ // declares no projection throws here, which is the correct rendering.
292
+ let projectionRendered = false;
293
+ try {
294
+ const projected = adapter.buildInvocation(
295
+ { ...request, filesystemPolicy: { mode: "read-only" } },
296
+ "/harnery-bench/final.txt",
297
+ ).argv;
298
+ projectionRendered = projected.join(" ") !== argv.join(" ");
299
+ } catch {
300
+ projectionRendered = false;
301
+ }
302
+
283
303
  let normalized: SpawnResult | null = null;
284
304
  try {
285
305
  normalized = adapter.normalizeResult(adapter.fixture.raw);
@@ -331,6 +351,7 @@ function observeAdapter(
331
351
  normalized && "toolEvidence" in normalized ? "supported" : "unsupported",
332
352
  ),
333
353
  policyMapping: NOT_CHECKED,
354
+ filesystemPolicyProjection: fromAdapter(projectionRendered ? "supported" : "unsupported"),
334
355
  interruption: NOT_CHECKED,
335
356
  streaming: NOT_CHECKED,
336
357
  steering: NOT_CHECKED,
@@ -16,6 +16,7 @@ function capabilities(overrides: Partial<HarnessCapabilities>): HarnessCapabilit
16
16
  cost: unsupported(),
17
17
  toolEvidence: unsupported("The final-result adapter does not retain tool events."),
18
18
  policyMapping: unsupported("No ALLOW/DENY/ASK translation at the workflow boundary."),
19
+ filesystemPolicyProjection: unsupported("The adapter declares no sandbox projection."),
19
20
  interruption: partial("Timeout kills the subprocess; no caller-driven interrupt handle."),
20
21
  streaming: unsupported("Workflow children return one normalized final result."),
21
22
  steering: unsupported("One prompt is fixed at subprocess launch."),
@@ -43,7 +44,7 @@ export const BUILTIN_HARNESS_PROFILES = {
43
44
  authModel: "own-auth",
44
45
  modelFamily: "claude",
45
46
  effortValues: ["low", "medium", "high", "xhigh", "max"],
46
- verified: { date: "2026-07-20", version: "current CLI contract" },
47
+ verified: { date: "2026-07-25", version: "2.1.197 (Claude Code)" },
47
48
  capabilities: capabilities({
48
49
  effortSelection: supported("Mapped to `--effort <level>`."),
49
50
  maxTurns: supported("Mapped to `--max-turns <n>`."),
@@ -69,9 +70,19 @@ export const BUILTIN_HARNESS_PROFILES = {
69
70
  authModel: "own-auth",
70
71
  modelFamily: "gpt",
71
72
  effortValues: ["none", "minimal", "low", "medium", "high", "xhigh"],
72
- verified: { date: "2026-07-21", version: "codex-cli 0.145.0-alpha.18" },
73
+ verified: { date: "2026-07-25", version: "codex-cli 0.144.5" },
74
+ // Verified against codex-cli 0.144.5: `--sandbox <mode>` plus
75
+ // `sandbox_workspace_write.writable_roots`. Without the writable-root entry
76
+ // a workspace-write child still cannot write a repository's .git directory.
77
+ sandboxProjection: {
78
+ modes: { "read-only": "read-only", "workspace-write": "workspace-write" },
79
+ writableRoots: true,
80
+ },
73
81
  capabilities: capabilities({
74
82
  effortSelection: supported('Mapped to `-c model_reasoning_effort="<level>"`.'),
83
+ filesystemPolicyProjection: supported(
84
+ "Mode renders to --sandbox; writable roots to sandbox_workspace_write.writable_roots.",
85
+ ),
75
86
  maxTurns: unsupported("codex exec exposes no turn-ceiling flag."),
76
87
  sessionId: unsupported("--output-last-message carries no session id."),
77
88
  cost: unsupported("The final-message path carries no usage or cost."),
@@ -95,7 +106,7 @@ export const BUILTIN_HARNESS_PROFILES = {
95
106
  authModel: "own-auth",
96
107
  modelFamily: "multi",
97
108
  effortValues: [],
98
- verified: { date: "2026-07-21", version: "2026.07.16-899851b" },
109
+ verified: { date: "2026-07-25", version: "2026.07.23-e383d2b" },
99
110
  capabilities: capabilities({
100
111
  effortSelection: unsupported(
101
112
  "Cursor embeds effort in some parameterized model ids; Harnery does not rewrite model ids.",
@@ -16,6 +16,7 @@ export const HARNESS_CAPABILITY_DIMENSIONS = [
16
16
  "cost",
17
17
  "toolEvidence",
18
18
  "policyMapping",
19
+ "filesystemPolicyProjection",
19
20
  "interruption",
20
21
  "streaming",
21
22
  "steering",
@@ -38,6 +39,17 @@ export interface CapabilityClaim {
38
39
 
39
40
  export type HarnessCapabilities = Record<HarnessCapabilityDimension, CapabilityClaim>;
40
41
 
42
+ /** What one harness can represent of a filesystem-policy projection
43
+ * (ADR 0039). `null` means the harness does not distinguish that, which is a
44
+ * fact about the harness rather than something to paper over: a projection the
45
+ * adapter would silently drop must be refused instead. */
46
+ export interface HarnessSandboxProjection {
47
+ /** Native representation per canonical mode, or null when unrepresentable. */
48
+ modes: Record<"read-only" | "workspace-write", string | null>;
49
+ /** Whether the harness accepts an explicit writable-root set. */
50
+ writableRoots: boolean;
51
+ }
52
+
41
53
  export interface HarnessProfile {
42
54
  id: HarnessId;
43
55
  displayName: string;
@@ -52,6 +64,9 @@ export interface HarnessProfile {
52
64
  capabilities: HarnessCapabilities;
53
65
  /** The last real vendor CLI contract used to validate this declaration. */
54
66
  verified?: { date: string; version: string };
67
+ /** How this adapter projects host filesystem policy into the vendor's own
68
+ * sandbox (ADR 0039). Absent means it cannot project any of it. */
69
+ sandboxProjection?: HarnessSandboxProjection;
55
70
  }
56
71
 
57
72
  /** Fully planned child invocation. `resultFile` is used by adapters such as