harnery 0.24.0 → 0.26.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (55) hide show
  1. package/dist/commands/harness.d.ts +2 -1
  2. package/dist/commands/harness.d.ts.map +1 -1
  3. package/dist/commands/harness.js +113 -3
  4. package/dist/commands/supervisor.d.ts.map +1 -1
  5. package/dist/commands/supervisor.js +3 -13
  6. package/dist/commands/work.d.ts.map +1 -1
  7. package/dist/commands/work.js +2 -7
  8. package/dist/commands/workflow.d.ts.map +1 -1
  9. package/dist/commands/workflow.js +3 -13
  10. package/dist/core/harnesses/attest.d.ts +53 -0
  11. package/dist/core/harnesses/attest.d.ts.map +1 -0
  12. package/dist/core/harnesses/attest.js +120 -0
  13. package/dist/core/harnesses/attestation.d.ts +82 -0
  14. package/dist/core/harnesses/attestation.d.ts.map +1 -0
  15. package/dist/core/harnesses/attestation.js +171 -0
  16. package/dist/core/harnesses/bench.d.ts +26 -1
  17. package/dist/core/harnesses/bench.d.ts.map +1 -1
  18. package/dist/core/harnesses/bench.js +137 -32
  19. package/dist/core/harnesses/index.d.ts +6 -2
  20. package/dist/core/harnesses/index.d.ts.map +1 -1
  21. package/dist/core/harnesses/index.js +3 -1
  22. package/dist/core/harnesses/profiles.d.ts +6 -6
  23. package/dist/core/harnesses/profiles.js +3 -3
  24. package/dist/core/workflow/engine.js +3 -0
  25. package/dist/core/workflow/proof.d.ts +4 -1
  26. package/dist/core/workflow/proof.d.ts.map +1 -1
  27. package/dist/core/workflow/proof.js +3 -2
  28. package/dist/core/workflow/spawn-claude.d.ts.map +1 -1
  29. package/dist/core/workflow/spawn-claude.js +2 -1
  30. package/dist/core/workflow/spawn-codex.d.ts.map +1 -1
  31. package/dist/core/workflow/spawn-codex.js +2 -1
  32. package/dist/core/workflow/spawn-cursor.d.ts.map +1 -1
  33. package/dist/core/workflow/spawn-cursor.js +2 -1
  34. package/dist/core/workflow/spawn-failure.d.ts +18 -0
  35. package/dist/core/workflow/spawn-failure.d.ts.map +1 -0
  36. package/dist/core/workflow/spawn-failure.js +24 -0
  37. package/dist/core/workflow/types.d.ts +14 -0
  38. package/dist/core/workflow/types.d.ts.map +1 -1
  39. package/package.json +4 -2
  40. package/src/commands/harness.ts +135 -2
  41. package/src/commands/supervisor.ts +11 -15
  42. package/src/commands/work.ts +8 -8
  43. package/src/commands/workflow.ts +11 -15
  44. package/src/core/harnesses/attest.ts +181 -0
  45. package/src/core/harnesses/attestation.ts +249 -0
  46. package/src/core/harnesses/bench.ts +192 -32
  47. package/src/core/harnesses/index.ts +26 -1
  48. package/src/core/harnesses/profiles.ts +3 -3
  49. package/src/core/workflow/engine.ts +3 -0
  50. package/src/core/workflow/proof.ts +7 -1
  51. package/src/core/workflow/spawn-claude.ts +2 -1
  52. package/src/core/workflow/spawn-codex.ts +2 -1
  53. package/src/core/workflow/spawn-cursor.ts +2 -1
  54. package/src/core/workflow/spawn-failure.ts +28 -0
  55. package/src/core/workflow/types.ts +15 -0
@@ -1,6 +1,8 @@
1
1
  import { spawnSync } from "node:child_process";
2
2
  import { HARNESS_SPECS } from "../hooks/harness/events.ts";
3
3
  import type { SpawnRequest, SpawnResult } from "../workflow/types.ts";
4
+ import type { AttestableDimension, HarnessAttestation } from "./attestation.ts";
5
+ import { isAttestationCurrent, readAttestation } from "./attestation.ts";
4
6
  import type { HarnessRegistry } from "./registry.ts";
5
7
  import type {
6
8
  CapabilitySupport,
@@ -19,7 +21,16 @@ export type BenchVerdict =
19
21
  | "skipped"
20
22
  | "drift";
21
23
 
22
- export type BenchDimension = "registration" | "binary" | HarnessCapabilityDimension;
24
+ export type BenchDimension = "registration" | "binary" | "contract" | HarnessCapabilityDimension;
25
+
26
+ /** How a result was established (ADR 0037). Orthogonal to the verdict: this
27
+ * answers "how do we know", not "what is true".
28
+ *
29
+ * - `adapter`: checked against Harnery's own planner, normalizer, or fixture.
30
+ * Proves the adapter contract, says nothing about the installed vendor CLI.
31
+ * - `attested`: checked by observing the installed vendor CLI on this host.
32
+ * - `declared`: not checked. The value is the declaration, repeated. */
33
+ export type BenchBasis = "adapter" | "attested" | "declared";
23
34
 
24
35
  export interface BenchResult {
25
36
  harness: HarnessId;
@@ -27,6 +38,7 @@ export interface BenchResult {
27
38
  declared: CapabilitySupport | "not_applicable";
28
39
  observed: BenchVerdict;
29
40
  verdict: BenchVerdict;
41
+ basis: BenchBasis;
30
42
  note?: string;
31
43
  }
32
44
 
@@ -36,6 +48,9 @@ export interface HarnessBenchReport {
36
48
  harnesses: HarnessId[];
37
49
  results: BenchResult[];
38
50
  summary: Record<BenchVerdict, number>;
51
+ /** Result counts per basis, so a caller can tell a clean report from an
52
+ * unmeasured one without walking every result. */
53
+ basisSummary: Record<BenchBasis, number>;
39
54
  drift: boolean;
40
55
  skipped: boolean;
41
56
  }
@@ -45,6 +60,14 @@ export interface HarnessBenchOptions {
45
60
  dimensions?: readonly HarnessCapabilityDimension[];
46
61
  /** Test seam and alternate host probe. A null version means unavailable. */
47
62
  versionProbe?: (binary: string) => string | null;
63
+ /** Where to look for live attestations (ADR 0038). Defaults to the coord
64
+ * root. Reading them never runs a model turn. */
65
+ coordRoot?: string;
66
+ /** Test seam. Defaults to reading the attestation store. */
67
+ attestationReader?: (harness: HarnessId) => HarnessAttestation | null;
68
+ /** Only cite attestations recorded under this billing mode. Omitted means
69
+ * any mode is acceptable for reporting purposes. */
70
+ subscriptionOnly?: boolean;
48
71
  }
49
72
 
50
73
  const EMPTY_SUMMARY: Record<BenchVerdict, number> = {
@@ -57,6 +80,44 @@ const EMPTY_SUMMARY: Record<BenchVerdict, number> = {
57
80
  drift: 0,
58
81
  };
59
82
 
83
+ const EMPTY_BASIS_SUMMARY: Record<BenchBasis, number> = {
84
+ adapter: 0,
85
+ attested: 0,
86
+ declared: 0,
87
+ };
88
+
89
+ /** A stored attestation only counts when it still matches the installed
90
+ * version and the current declaration. A stale one is ignored rather than
91
+ * trusted, so an upgraded vendor CLI silently drops back to adapter basis. */
92
+ function loadAttestation(
93
+ id: HarnessId,
94
+ version: string | null,
95
+ adapter: HarnessAdapter,
96
+ opts: HarnessBenchOptions,
97
+ ): HarnessAttestation | null {
98
+ let record: HarnessAttestation | null = null;
99
+ try {
100
+ record = opts.attestationReader
101
+ ? opts.attestationReader(id)
102
+ : readAttestation(id, { coordRoot: opts.coordRoot });
103
+ } catch {
104
+ // No coord root, unreadable store: the bench still runs, just unattested.
105
+ return null;
106
+ }
107
+ return isAttestationCurrent(record, version, adapter.profile, opts.subscriptionOnly)
108
+ ? record
109
+ : null;
110
+ }
111
+
112
+ /** First version-shaped token in a string, or null when there is none.
113
+ * Tolerates the vendor prefixes and suffixes real CLIs print, so
114
+ * `codex-cli 0.144.5` and `2.1.197 (Claude Code)` both reduce to a token. */
115
+ function versionToken(value: string | null | undefined): string | null {
116
+ if (!value) return null;
117
+ const match = value.match(/\d+(?:\.\d+)+(?:[-.][0-9a-z]+)*/i);
118
+ return match ? match[0].toLowerCase() : null;
119
+ }
120
+
60
121
  export function runHarnessBench(
61
122
  registry: HarnessRegistry,
62
123
  opts: HarnessBenchOptions = {},
@@ -65,7 +126,7 @@ export function runHarnessBench(
65
126
  const dimensions = opts.dimensions?.length
66
127
  ? [...new Set(opts.dimensions)]
67
128
  : [...HARNESS_CAPABILITY_DIMENSIONS];
68
- const versionProbe = opts.versionProbe ?? probeVersion;
129
+ const versionProbe = opts.versionProbe ?? probeBinaryVersion;
69
130
  const results: BenchResult[] = [];
70
131
 
71
132
  for (const id of ids) {
@@ -76,6 +137,7 @@ export function runHarnessBench(
76
137
  declared: "supported",
77
138
  observed: "supported",
78
139
  verdict: "supported",
140
+ basis: "adapter",
79
141
  note: `${adapter.profile.integrationMode}; ${adapter.profile.authModel}`,
80
142
  });
81
143
 
@@ -86,38 +148,125 @@ export function runHarnessBench(
86
148
  declared: "supported",
87
149
  observed: version ? "supported" : "skipped",
88
150
  verdict: version ? "supported" : "skipped",
151
+ // A skip establishes nothing about the vendor, so it is not an attestation.
152
+ basis: version ? "attested" : "declared",
89
153
  note: version ?? `${adapter.profile.binary} not found on PATH`,
90
154
  });
91
155
 
156
+ results.push(contractResult(id, adapter, version));
157
+
92
158
  const observations = observeAdapter(adapter);
159
+ const attestation = loadAttestation(id, version, adapter, opts);
93
160
  for (const dimension of dimensions) {
94
161
  const claim = adapter.profile.capabilities[dimension];
95
- const observed = observations[dimension];
162
+ const live = attestation?.observations[dimension as AttestableDimension];
163
+ // A live observation of the installed CLI outranks a fixture check of
164
+ // Harnery's own normalizer. Everything else keeps its adapter basis.
165
+ const { observed, basis }: DimensionObservation = live
166
+ ? { observed: live, basis: "attested" }
167
+ : observations[dimension];
96
168
  results.push({
97
169
  harness: id,
98
170
  dimension,
99
171
  declared: claim.support,
100
172
  observed,
101
173
  verdict: reconcile(claim.support, observed),
102
- note: claim.note,
174
+ basis,
175
+ note: live ? `attested on ${attestation?.binary_version}` : claim.note,
103
176
  });
104
177
  }
105
178
  }
106
179
 
107
180
  const summary = { ...EMPTY_SUMMARY };
108
- for (const result of results) summary[result.verdict]++;
181
+ const basisSummary = { ...EMPTY_BASIS_SUMMARY };
182
+ for (const result of results) {
183
+ summary[result.verdict]++;
184
+ basisSummary[result.basis]++;
185
+ }
109
186
  return {
110
187
  generatedAt: new Date().toISOString(),
111
188
  mode: "offline",
112
189
  harnesses: ids,
113
190
  results,
114
191
  summary,
192
+ basisSummary,
115
193
  drift: summary.drift > 0,
116
194
  skipped: summary.skipped > 0,
117
195
  };
118
196
  }
119
197
 
120
- function observeAdapter(adapter: HarnessAdapter): Record<HarnessCapabilityDimension, BenchVerdict> {
198
+ /** Compare the vendor contract a declaration was validated against with the
199
+ * one actually installed (ADR 0037). This is the only dimension attested
200
+ * without a model turn, because a version string costs nothing to read.
201
+ *
202
+ * A mismatch is `drift`, not failure: a newer vendor CLI is not automatically
203
+ * broken, but the declaration is no longer backed by an observation. An
204
+ * unrecorded or unparseable declared version stays `unknown` rather than being
205
+ * inferred from the installed one. */
206
+ function contractResult(
207
+ id: HarnessId,
208
+ adapter: HarnessAdapter,
209
+ version: string | null,
210
+ ): BenchResult {
211
+ const recorded = adapter.profile.verified;
212
+ const base = { harness: id, dimension: "contract" as const, declared: "supported" as const };
213
+
214
+ if (!version) {
215
+ return {
216
+ ...base,
217
+ observed: "skipped",
218
+ verdict: "skipped",
219
+ basis: "declared",
220
+ note: `${adapter.profile.binary} not found on PATH; installed contract unobservable`,
221
+ };
222
+ }
223
+ const declaredToken = versionToken(recorded?.version);
224
+ if (!declaredToken) {
225
+ return {
226
+ ...base,
227
+ observed: "unknown",
228
+ verdict: "unknown",
229
+ basis: "declared",
230
+ note: recorded
231
+ ? `verified.version "${recorded.version}" is not a version; installed ${version}`
232
+ : `no verified vendor contract recorded; installed ${version}`,
233
+ };
234
+ }
235
+ const installedToken = versionToken(version) ?? version.trim().toLowerCase();
236
+ if (installedToken === declaredToken) {
237
+ return {
238
+ ...base,
239
+ observed: "supported",
240
+ verdict: "supported",
241
+ basis: "attested",
242
+ note: `declaration validated against ${recorded?.version} on ${recorded?.date}`,
243
+ };
244
+ }
245
+ return {
246
+ ...base,
247
+ observed: "drift",
248
+ verdict: "drift",
249
+ basis: "attested",
250
+ note: `declaration validated against ${recorded?.version} (${recorded?.date}); installed ${version}`,
251
+ };
252
+ }
253
+
254
+ interface DimensionObservation {
255
+ observed: BenchVerdict;
256
+ basis: BenchBasis;
257
+ }
258
+
259
+ /** Checked against Harnery's planner, normalizer, or fixture. */
260
+ function fromAdapter(observed: BenchVerdict): DimensionObservation {
261
+ return { observed, basis: "adapter" };
262
+ }
263
+
264
+ /** No check exists for this dimension yet. */
265
+ const NOT_CHECKED: DimensionObservation = { observed: "unknown", basis: "declared" };
266
+
267
+ function observeAdapter(
268
+ adapter: HarnessAdapter,
269
+ ): Record<HarnessCapabilityDimension, DimensionObservation> {
121
270
  const profile = adapter.profile;
122
271
  const effort = profile.effortValues[0];
123
272
  const request: SpawnRequest = {
@@ -154,49 +303,57 @@ function observeAdapter(adapter: HarnessAdapter): Record<HarnessCapabilityDimens
154
303
  const hookSubcommands = new Set(hookSpec?.events.map((event) => event.subcommand) ?? []);
155
304
 
156
305
  return {
157
- invocation:
306
+ invocation: fromAdapter(
158
307
  !planningFailed && argv[0] === profile.binary && argv.includes(request.prompt)
159
308
  ? "supported"
160
309
  : "unsupported",
161
- modelSelection:
310
+ ),
311
+ modelSelection: fromAdapter(
162
312
  !planningFailed && argv.includes(request.model ?? "") ? "supported" : "unsupported",
163
- effortSelection:
313
+ ),
314
+ effortSelection: fromAdapter(
164
315
  effort === undefined
165
316
  ? "unsupported"
166
317
  : !planningFailed && argv.some((arg) => arg.includes(effort))
167
318
  ? "supported"
168
319
  : "unsupported",
169
- maxTurns:
320
+ ),
321
+ maxTurns: fromAdapter(
170
322
  !planningFailed && argv.includes(String(request.maxTurns)) ? "supported" : "unsupported",
171
- finalResult: finalMatches ? "supported" : "unsupported",
172
- sessionId:
323
+ ),
324
+ finalResult: fromAdapter(finalMatches ? "supported" : "unsupported"),
325
+ sessionId: fromAdapter(
173
326
  sessionObserved === "supported" && normalized?.sessionId !== fixture.sessionId
174
327
  ? "unsupported"
175
328
  : sessionObserved,
176
- cost:
329
+ ),
330
+ cost: fromAdapter(
177
331
  costObserved === "supported" && normalized?.costUsd !== fixture.costUsd
178
332
  ? "unsupported"
179
333
  : costObserved,
180
- toolEvidence: normalized && "toolEvidence" in normalized ? "supported" : "unsupported",
181
- policyMapping: "unknown",
182
- interruption: "unknown",
183
- streaming: "unknown",
184
- steering: "unknown",
185
- resume: "unknown",
186
- images: "unknown",
187
- contextTelemetry: "unknown",
334
+ ),
335
+ toolEvidence: fromAdapter(
336
+ normalized && "toolEvidence" in normalized ? "supported" : "unsupported",
337
+ ),
338
+ policyMapping: NOT_CHECKED,
339
+ interruption: NOT_CHECKED,
340
+ streaming: NOT_CHECKED,
341
+ steering: NOT_CHECKED,
342
+ resume: NOT_CHECKED,
343
+ images: NOT_CHECKED,
344
+ contextTelemetry: NOT_CHECKED,
188
345
  preCompactionSignal: hookSpec
189
- ? hookSubcommands.has("pre-compact")
190
- ? "supported"
191
- : "unsupported"
192
- : "unknown",
346
+ ? fromAdapter(hookSubcommands.has("pre-compact") ? "supported" : "unsupported")
347
+ : NOT_CHECKED,
193
348
  postCompactionSignal: hookSpec
194
- ? hookSubcommands.has("post-compact") ||
195
- (adapter.profile.id === "claude-code" && hookSubcommands.has("session-start"))
196
- ? "supported"
197
- : "unsupported"
198
- : "unknown",
199
- compaction: "unknown",
349
+ ? fromAdapter(
350
+ hookSubcommands.has("post-compact") ||
351
+ (adapter.profile.id === "claude-code" && hookSubcommands.has("session-start"))
352
+ ? "supported"
353
+ : "unsupported",
354
+ )
355
+ : NOT_CHECKED,
356
+ compaction: NOT_CHECKED,
200
357
  };
201
358
  }
202
359
 
@@ -207,7 +364,10 @@ function reconcile(declared: CapabilitySupport, observed: BenchVerdict): BenchVe
207
364
  return declared === observed ? observed : "drift";
208
365
  }
209
366
 
210
- function probeVersion(binary: string): string | null {
367
+ /** Ask an installed vendor CLI what version it is. Null when it is absent or
368
+ * refuses to answer. Shared with the live attestation probe so both halves of
369
+ * the capability story key on the same string. */
370
+ export function probeBinaryVersion(binary: string): string | null {
211
371
  const result = spawnSync(binary, ["--version"], { encoding: "utf8", timeout: 5_000 });
212
372
  if (result.error || result.status !== 0) return null;
213
373
  return (result.stdout || result.stderr).trim().split("\n")[0] || "installed";
@@ -1,12 +1,37 @@
1
1
  export type { Spawner, SpawnRequest, SpawnResult } from "../workflow/types.ts";
2
2
  export type {
3
+ AttestationOutcome,
4
+ HarnessAttestationReport,
5
+ HarnessAttestationResult,
6
+ RunHarnessAttestationOptions,
7
+ } from "./attest.ts";
8
+ export { ATTESTATION_PROMPT, runHarnessAttestation } from "./attest.ts";
9
+ export type {
10
+ AttestableDimension,
11
+ AttestationStoreOptions,
12
+ HarnessAttestation,
13
+ } from "./attestation.ts";
14
+ export {
15
+ ATTESTABLE_DIMENSIONS,
16
+ ATTESTATION_SCHEMA_VERSION,
17
+ attestationsDir,
18
+ harnessProofInputs,
19
+ isAttestationCurrent,
20
+ listAttestations,
21
+ profileDigest,
22
+ readAttestation,
23
+ validateAttestation,
24
+ writeAttestation,
25
+ } from "./attestation.ts";
26
+ export type {
27
+ BenchBasis,
3
28
  BenchDimension,
4
29
  BenchResult,
5
30
  BenchVerdict,
6
31
  HarnessBenchOptions,
7
32
  HarnessBenchReport,
8
33
  } from "./bench.ts";
9
- export { runHarnessBench } from "./bench.ts";
34
+ export { probeBinaryVersion, runHarnessBench } from "./bench.ts";
10
35
  export type { BuiltinHarnessId } from "./profiles.ts";
11
36
  export {
12
37
  BUILTIN_HARNESS_IDS,
@@ -43,7 +43,7 @@ export const BUILTIN_HARNESS_PROFILES = {
43
43
  authModel: "own-auth",
44
44
  modelFamily: "claude",
45
45
  effortValues: ["low", "medium", "high", "xhigh", "max"],
46
- verified: { date: "2026-07-20", version: "current CLI contract" },
46
+ verified: { date: "2026-07-25", version: "2.1.197 (Claude Code)" },
47
47
  capabilities: capabilities({
48
48
  effortSelection: supported("Mapped to `--effort <level>`."),
49
49
  maxTurns: supported("Mapped to `--max-turns <n>`."),
@@ -69,7 +69,7 @@ export const BUILTIN_HARNESS_PROFILES = {
69
69
  authModel: "own-auth",
70
70
  modelFamily: "gpt",
71
71
  effortValues: ["none", "minimal", "low", "medium", "high", "xhigh"],
72
- verified: { date: "2026-07-21", version: "codex-cli 0.145.0-alpha.18" },
72
+ verified: { date: "2026-07-25", version: "codex-cli 0.144.5" },
73
73
  capabilities: capabilities({
74
74
  effortSelection: supported('Mapped to `-c model_reasoning_effort="<level>"`.'),
75
75
  maxTurns: unsupported("codex exec exposes no turn-ceiling flag."),
@@ -95,7 +95,7 @@ export const BUILTIN_HARNESS_PROFILES = {
95
95
  authModel: "own-auth",
96
96
  modelFamily: "multi",
97
97
  effortValues: [],
98
- verified: { date: "2026-07-21", version: "2026.07.16-899851b" },
98
+ verified: { date: "2026-07-25", version: "2026.07.23-e383d2b" },
99
99
  capabilities: capabilities({
100
100
  effortSelection: unsupported(
101
101
  "Cursor embeds effort in some parameterized model ids; Harnery does not rewrite model ids.",
@@ -316,6 +316,7 @@ async function executeWorkflow(
316
316
  agents: [],
317
317
  evidence: [],
318
318
  harnessEvidence: opts.harnessEvidence,
319
+ harnessAttestations: opts.harnessAttestations,
319
320
  policy: policy
320
321
  ? {
321
322
  config: policy,
@@ -1054,6 +1055,7 @@ async function executeWorkflow(
1054
1055
  agents: Array.from(agentProofs.values()),
1055
1056
  evidence: evidenceRecords,
1056
1057
  harnessEvidence: opts.harnessEvidence,
1058
+ harnessAttestations: opts.harnessAttestations,
1057
1059
  policy: policy
1058
1060
  ? {
1059
1061
  config: policy,
@@ -1133,6 +1135,7 @@ async function executeWorkflow(
1133
1135
  agents: Array.from(agentProofs.values()),
1134
1136
  evidence: evidenceRecords,
1135
1137
  harnessEvidence: opts.harnessEvidence,
1138
+ harnessAttestations: opts.harnessAttestations,
1136
1139
  policy: policy
1137
1140
  ? {
1138
1141
  config: policy,
@@ -16,6 +16,7 @@ import type {
16
16
  AcceptanceCriterion,
17
17
  AcceptanceResult,
18
18
  AcceptanceSummary,
19
+ HarnessAttestationCitation,
19
20
  HarnessEvidenceCapability,
20
21
  HarnessEvidenceCoverage,
21
22
  ResultDigest,
@@ -81,6 +82,9 @@ export interface BuildWorkflowProofInput {
81
82
  agents: WorkflowAgentProof[];
82
83
  evidence: WorkflowEvidenceRecord[];
83
84
  harnessEvidence?: Readonly<Record<string, HarnessEvidenceCapability | undefined>>;
85
+ /** Live attestations backing each harness's claims (ADR 0038). Injected by
86
+ * the caller so proof stays free of filesystem lookups. */
87
+ harnessAttestations?: Readonly<Record<string, HarnessAttestationCitation | undefined>>;
84
88
  policy?: {
85
89
  config: Readonly<NormalizedPolicy>;
86
90
  decisions: readonly PolicyDecision[];
@@ -222,7 +226,7 @@ export function buildWorkflowProof(input: BuildWorkflowProofInput): WorkflowProo
222
226
  session_id: clippedOptional(agent.session_id, MAX_REF_CHARS),
223
227
  error: clippedOptional(agent.error, MAX_SUMMARY_CHARS),
224
228
  }));
225
- const harnesses = buildHarnessCoverage(agents, input.harnessEvidence);
229
+ const harnesses = buildHarnessCoverage(agents, input.harnessEvidence, input.harnessAttestations);
226
230
  const unknowns = buildUnknowns(agents, harnesses, repository);
227
231
  const journal = readFileSync(input.journalPath);
228
232
  return {
@@ -455,6 +459,7 @@ function normalizeRepoSnapshot(snapshot: RepoSnapshot): WorkflowRepoSnapshot {
455
459
  function buildHarnessCoverage(
456
460
  agents: WorkflowAgentProof[],
457
461
  claims: Readonly<Record<string, HarnessEvidenceCapability | undefined>> | undefined,
462
+ attestations: Readonly<Record<string, HarnessAttestationCitation | undefined>> | undefined,
458
463
  ): HarnessEvidenceCoverage[] {
459
464
  return [...new Set(agents.map((agent) => agent.harness))].map((harness) => {
460
465
  const harnessAgents = agents.filter((agent) => agent.harness === harness);
@@ -469,6 +474,7 @@ function buildHarnessCoverage(
469
474
  session_ids: harnessAgents.filter((agent) => agent.session_id).length,
470
475
  costs: harnessAgents.filter((agent) => agent.cost_usd !== undefined).length,
471
476
  },
477
+ ...(attestations?.[harness] ? { attestation: attestations[harness] } : {}),
472
478
  };
473
479
  });
474
480
  }
@@ -23,6 +23,7 @@ import { validateHarnessEffort } from "../harnesses/profiles.ts";
23
23
  import type { HarnessInvocation, HarnessRawResult } from "../harnesses/types.ts";
24
24
  import { buildChildEnv } from "./child-env.ts";
25
25
  import { notFoundError } from "./harnesses.ts";
26
+ import { vendorFailureText } from "./spawn-failure.ts";
26
27
  import type { Spawner, SpawnRequest, SpawnResult } from "./types.ts";
27
28
 
28
29
  interface ClaudeEnvelope {
@@ -65,7 +66,7 @@ export function normalizeClaudeResult(raw: HarnessRawResult): SpawnResult {
65
66
  ok: false,
66
67
  text: "",
67
68
  durationMs: raw.durationMs,
68
- error: `claude exited ${raw.exitCode}: ${(raw.stderr || raw.stdout).slice(0, 500)}`,
69
+ error: `claude exited ${raw.exitCode}: ${vendorFailureText(raw)}`,
69
70
  };
70
71
  }
71
72
 
@@ -24,6 +24,7 @@ import { validateHarnessEffort } from "../harnesses/profiles.ts";
24
24
  import type { HarnessInvocation, HarnessRawResult } from "../harnesses/types.ts";
25
25
  import { buildChildEnv } from "./child-env.ts";
26
26
  import { notFoundError } from "./harnesses.ts";
27
+ import { vendorFailureText } from "./spawn-failure.ts";
27
28
  import type { Spawner, SpawnRequest, SpawnResult } from "./types.ts";
28
29
 
29
30
  export function buildCodexInvocation(req: SpawnRequest, resultFile?: string): HarnessInvocation {
@@ -53,7 +54,7 @@ export function normalizeCodexResult(raw: HarnessRawResult): SpawnResult {
53
54
  ok: false,
54
55
  text: "",
55
56
  durationMs: raw.durationMs,
56
- error: `codex exited ${raw.exitCode}: ${(raw.stderr || raw.stdout).slice(0, 500)}`,
57
+ error: `codex exited ${raw.exitCode}: ${vendorFailureText(raw)}`,
57
58
  };
58
59
  }
59
60
  return {
@@ -24,6 +24,7 @@ import { validateHarnessEffort } from "../harnesses/profiles.ts";
24
24
  import type { HarnessInvocation, HarnessRawResult } from "../harnesses/types.ts";
25
25
  import { buildChildEnv } from "./child-env.ts";
26
26
  import { notFoundError } from "./harnesses.ts";
27
+ import { vendorFailureText } from "./spawn-failure.ts";
27
28
  import type { Spawner, SpawnRequest, SpawnResult } from "./types.ts";
28
29
 
29
30
  interface CursorEnvelope {
@@ -72,7 +73,7 @@ export function normalizeCursorResult(raw: HarnessRawResult): SpawnResult {
72
73
  ok: false,
73
74
  text: "",
74
75
  durationMs: raw.durationMs,
75
- error: `cursor-agent exited ${raw.exitCode}: ${(raw.stderr || raw.stdout).slice(0, 500)}`,
76
+ error: `cursor-agent exited ${raw.exitCode}: ${vendorFailureText(raw)}`,
76
77
  };
77
78
  }
78
79
 
@@ -0,0 +1,28 @@
1
+ /**
2
+ * Shared failure-text extraction for the harness spawn adapters.
3
+ *
4
+ * A vendor CLI prints its banner, resolved config, and startup warnings first,
5
+ * and the reason it actually failed last. Every adapter used to keep the FIRST
6
+ * 500 characters of the transcript, which reliably preserved the banner and
7
+ * discarded the answer: a child that died with "your workspace is out of
8
+ * credits" reported a cosmetic startup warning instead.
9
+ *
10
+ * Two rules follow. Keep the tail. Include both streams, because which one
11
+ * carries the reason varies by vendor and by failure.
12
+ */
13
+
14
+ const DEFAULT_MAX_CHARS = 500;
15
+
16
+ /** Bounded failure text from a finished child, tail-preserving. */
17
+ export function vendorFailureText(
18
+ raw: { stdout?: string; stderr?: string },
19
+ maxChars = DEFAULT_MAX_CHARS,
20
+ ): string {
21
+ const parts = [raw.stderr, raw.stdout]
22
+ .map((part) => part?.trim())
23
+ .filter((part): part is string => !!part);
24
+ if (parts.length === 0) return "";
25
+ // Both streams, stderr first, so a reason on the unexpected stream survives.
26
+ const combined = parts.join("\n");
27
+ return combined.length > maxChars ? `…${combined.slice(-maxChars)}` : combined;
28
+ }
@@ -126,6 +126,14 @@ export interface WorkflowRepoEvidence {
126
126
  };
127
127
  }
128
128
 
129
+ /** Bounded pointer to the live attestation backing a harness's claims
130
+ * (ADR 0038). Structural facts only: no prompt text, no host paths. */
131
+ export interface HarnessAttestationCitation {
132
+ binary_version: string;
133
+ observed_at: string;
134
+ record_digest: string;
135
+ }
136
+
129
137
  export interface HarnessEvidenceCoverage {
130
138
  harness: HarnessName;
131
139
  tool_evidence: {
@@ -137,6 +145,10 @@ export interface HarnessEvidenceCoverage {
137
145
  session_ids: number;
138
146
  costs: number;
139
147
  };
148
+ /** What backs this harness's capability claims (ADR 0038). Absent when the
149
+ * host recorded no live attestation, which is the common case and is not by
150
+ * itself a proof unknown. */
151
+ attestation?: HarnessAttestationCitation;
140
152
  }
141
153
 
142
154
  export interface WorkflowProofUnknown {
@@ -429,6 +441,9 @@ export interface EngineOpts {
429
441
  /** Capability claims used to state whether adapter-native tool evidence was
430
442
  * available. Missing claims remain unknown. */
431
443
  harnessEvidence?: Readonly<Record<HarnessName, HarnessEvidenceCapability | undefined>>;
444
+ /** Live attestations backing each harness's claims (ADR 0038). Read once by
445
+ * the host and injected, so the engine performs no capability lookups. */
446
+ harnessAttestations?: Readonly<Record<HarnessName, HarnessAttestationCitation | undefined>>;
432
447
  /** Immutable host policy. Workflow scripts and model prompts cannot replace it. */
433
448
  policy?: PolicySpec | NormalizedPolicy;
434
449
  /** Host callback for ASK. Missing, invalid, throwing, or timed-out resolution denies. */