harnery 0.24.0 → 0.26.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/commands/harness.d.ts +2 -1
- package/dist/commands/harness.d.ts.map +1 -1
- package/dist/commands/harness.js +113 -3
- package/dist/commands/supervisor.d.ts.map +1 -1
- package/dist/commands/supervisor.js +3 -13
- package/dist/commands/work.d.ts.map +1 -1
- package/dist/commands/work.js +2 -7
- package/dist/commands/workflow.d.ts.map +1 -1
- package/dist/commands/workflow.js +3 -13
- package/dist/core/harnesses/attest.d.ts +53 -0
- package/dist/core/harnesses/attest.d.ts.map +1 -0
- package/dist/core/harnesses/attest.js +120 -0
- package/dist/core/harnesses/attestation.d.ts +82 -0
- package/dist/core/harnesses/attestation.d.ts.map +1 -0
- package/dist/core/harnesses/attestation.js +171 -0
- package/dist/core/harnesses/bench.d.ts +26 -1
- package/dist/core/harnesses/bench.d.ts.map +1 -1
- package/dist/core/harnesses/bench.js +137 -32
- package/dist/core/harnesses/index.d.ts +6 -2
- package/dist/core/harnesses/index.d.ts.map +1 -1
- package/dist/core/harnesses/index.js +3 -1
- package/dist/core/harnesses/profiles.d.ts +6 -6
- package/dist/core/harnesses/profiles.js +3 -3
- package/dist/core/workflow/engine.js +3 -0
- package/dist/core/workflow/proof.d.ts +4 -1
- package/dist/core/workflow/proof.d.ts.map +1 -1
- package/dist/core/workflow/proof.js +3 -2
- package/dist/core/workflow/spawn-claude.d.ts.map +1 -1
- package/dist/core/workflow/spawn-claude.js +2 -1
- package/dist/core/workflow/spawn-codex.d.ts.map +1 -1
- package/dist/core/workflow/spawn-codex.js +2 -1
- package/dist/core/workflow/spawn-cursor.d.ts.map +1 -1
- package/dist/core/workflow/spawn-cursor.js +2 -1
- package/dist/core/workflow/spawn-failure.d.ts +18 -0
- package/dist/core/workflow/spawn-failure.d.ts.map +1 -0
- package/dist/core/workflow/spawn-failure.js +24 -0
- package/dist/core/workflow/types.d.ts +14 -0
- package/dist/core/workflow/types.d.ts.map +1 -1
- package/package.json +4 -2
- package/src/commands/harness.ts +135 -2
- package/src/commands/supervisor.ts +11 -15
- package/src/commands/work.ts +8 -8
- package/src/commands/workflow.ts +11 -15
- package/src/core/harnesses/attest.ts +181 -0
- package/src/core/harnesses/attestation.ts +249 -0
- package/src/core/harnesses/bench.ts +192 -32
- package/src/core/harnesses/index.ts +26 -1
- package/src/core/harnesses/profiles.ts +3 -3
- package/src/core/workflow/engine.ts +3 -0
- package/src/core/workflow/proof.ts +7 -1
- package/src/core/workflow/spawn-claude.ts +2 -1
- package/src/core/workflow/spawn-codex.ts +2 -1
- package/src/core/workflow/spawn-cursor.ts +2 -1
- package/src/core/workflow/spawn-failure.ts +28 -0
- package/src/core/workflow/types.ts +15 -0
|
@@ -1,6 +1,8 @@
|
|
|
1
1
|
import { spawnSync } from "node:child_process";
|
|
2
2
|
import { HARNESS_SPECS } from "../hooks/harness/events.ts";
|
|
3
3
|
import type { SpawnRequest, SpawnResult } from "../workflow/types.ts";
|
|
4
|
+
import type { AttestableDimension, HarnessAttestation } from "./attestation.ts";
|
|
5
|
+
import { isAttestationCurrent, readAttestation } from "./attestation.ts";
|
|
4
6
|
import type { HarnessRegistry } from "./registry.ts";
|
|
5
7
|
import type {
|
|
6
8
|
CapabilitySupport,
|
|
@@ -19,7 +21,16 @@ export type BenchVerdict =
|
|
|
19
21
|
| "skipped"
|
|
20
22
|
| "drift";
|
|
21
23
|
|
|
22
|
-
export type BenchDimension = "registration" | "binary" | HarnessCapabilityDimension;
|
|
24
|
+
export type BenchDimension = "registration" | "binary" | "contract" | HarnessCapabilityDimension;
|
|
25
|
+
|
|
26
|
+
/** How a result was established (ADR 0037). Orthogonal to the verdict: this
|
|
27
|
+
* answers "how do we know", not "what is true".
|
|
28
|
+
*
|
|
29
|
+
* - `adapter`: checked against Harnery's own planner, normalizer, or fixture.
|
|
30
|
+
* Proves the adapter contract, says nothing about the installed vendor CLI.
|
|
31
|
+
* - `attested`: checked by observing the installed vendor CLI on this host.
|
|
32
|
+
* - `declared`: not checked. The value is the declaration, repeated. */
|
|
33
|
+
export type BenchBasis = "adapter" | "attested" | "declared";
|
|
23
34
|
|
|
24
35
|
export interface BenchResult {
|
|
25
36
|
harness: HarnessId;
|
|
@@ -27,6 +38,7 @@ export interface BenchResult {
|
|
|
27
38
|
declared: CapabilitySupport | "not_applicable";
|
|
28
39
|
observed: BenchVerdict;
|
|
29
40
|
verdict: BenchVerdict;
|
|
41
|
+
basis: BenchBasis;
|
|
30
42
|
note?: string;
|
|
31
43
|
}
|
|
32
44
|
|
|
@@ -36,6 +48,9 @@ export interface HarnessBenchReport {
|
|
|
36
48
|
harnesses: HarnessId[];
|
|
37
49
|
results: BenchResult[];
|
|
38
50
|
summary: Record<BenchVerdict, number>;
|
|
51
|
+
/** Result counts per basis, so a caller can tell a clean report from an
|
|
52
|
+
* unmeasured one without walking every result. */
|
|
53
|
+
basisSummary: Record<BenchBasis, number>;
|
|
39
54
|
drift: boolean;
|
|
40
55
|
skipped: boolean;
|
|
41
56
|
}
|
|
@@ -45,6 +60,14 @@ export interface HarnessBenchOptions {
|
|
|
45
60
|
dimensions?: readonly HarnessCapabilityDimension[];
|
|
46
61
|
/** Test seam and alternate host probe. A null version means unavailable. */
|
|
47
62
|
versionProbe?: (binary: string) => string | null;
|
|
63
|
+
/** Where to look for live attestations (ADR 0038). Defaults to the coord
|
|
64
|
+
* root. Reading them never runs a model turn. */
|
|
65
|
+
coordRoot?: string;
|
|
66
|
+
/** Test seam. Defaults to reading the attestation store. */
|
|
67
|
+
attestationReader?: (harness: HarnessId) => HarnessAttestation | null;
|
|
68
|
+
/** Only cite attestations recorded under this billing mode. Omitted means
|
|
69
|
+
* any mode is acceptable for reporting purposes. */
|
|
70
|
+
subscriptionOnly?: boolean;
|
|
48
71
|
}
|
|
49
72
|
|
|
50
73
|
const EMPTY_SUMMARY: Record<BenchVerdict, number> = {
|
|
@@ -57,6 +80,44 @@ const EMPTY_SUMMARY: Record<BenchVerdict, number> = {
|
|
|
57
80
|
drift: 0,
|
|
58
81
|
};
|
|
59
82
|
|
|
83
|
+
const EMPTY_BASIS_SUMMARY: Record<BenchBasis, number> = {
|
|
84
|
+
adapter: 0,
|
|
85
|
+
attested: 0,
|
|
86
|
+
declared: 0,
|
|
87
|
+
};
|
|
88
|
+
|
|
89
|
+
/** A stored attestation only counts when it still matches the installed
|
|
90
|
+
* version and the current declaration. A stale one is ignored rather than
|
|
91
|
+
* trusted, so an upgraded vendor CLI silently drops back to adapter basis. */
|
|
92
|
+
function loadAttestation(
|
|
93
|
+
id: HarnessId,
|
|
94
|
+
version: string | null,
|
|
95
|
+
adapter: HarnessAdapter,
|
|
96
|
+
opts: HarnessBenchOptions,
|
|
97
|
+
): HarnessAttestation | null {
|
|
98
|
+
let record: HarnessAttestation | null = null;
|
|
99
|
+
try {
|
|
100
|
+
record = opts.attestationReader
|
|
101
|
+
? opts.attestationReader(id)
|
|
102
|
+
: readAttestation(id, { coordRoot: opts.coordRoot });
|
|
103
|
+
} catch {
|
|
104
|
+
// No coord root, unreadable store: the bench still runs, just unattested.
|
|
105
|
+
return null;
|
|
106
|
+
}
|
|
107
|
+
return isAttestationCurrent(record, version, adapter.profile, opts.subscriptionOnly)
|
|
108
|
+
? record
|
|
109
|
+
: null;
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
/** First version-shaped token in a string, or null when there is none.
|
|
113
|
+
* Tolerates the vendor prefixes and suffixes real CLIs print, so
|
|
114
|
+
* `codex-cli 0.144.5` and `2.1.197 (Claude Code)` both reduce to a token. */
|
|
115
|
+
function versionToken(value: string | null | undefined): string | null {
|
|
116
|
+
if (!value) return null;
|
|
117
|
+
const match = value.match(/\d+(?:\.\d+)+(?:[-.][0-9a-z]+)*/i);
|
|
118
|
+
return match ? match[0].toLowerCase() : null;
|
|
119
|
+
}
|
|
120
|
+
|
|
60
121
|
export function runHarnessBench(
|
|
61
122
|
registry: HarnessRegistry,
|
|
62
123
|
opts: HarnessBenchOptions = {},
|
|
@@ -65,7 +126,7 @@ export function runHarnessBench(
|
|
|
65
126
|
const dimensions = opts.dimensions?.length
|
|
66
127
|
? [...new Set(opts.dimensions)]
|
|
67
128
|
: [...HARNESS_CAPABILITY_DIMENSIONS];
|
|
68
|
-
const versionProbe = opts.versionProbe ??
|
|
129
|
+
const versionProbe = opts.versionProbe ?? probeBinaryVersion;
|
|
69
130
|
const results: BenchResult[] = [];
|
|
70
131
|
|
|
71
132
|
for (const id of ids) {
|
|
@@ -76,6 +137,7 @@ export function runHarnessBench(
|
|
|
76
137
|
declared: "supported",
|
|
77
138
|
observed: "supported",
|
|
78
139
|
verdict: "supported",
|
|
140
|
+
basis: "adapter",
|
|
79
141
|
note: `${adapter.profile.integrationMode}; ${adapter.profile.authModel}`,
|
|
80
142
|
});
|
|
81
143
|
|
|
@@ -86,38 +148,125 @@ export function runHarnessBench(
|
|
|
86
148
|
declared: "supported",
|
|
87
149
|
observed: version ? "supported" : "skipped",
|
|
88
150
|
verdict: version ? "supported" : "skipped",
|
|
151
|
+
// A skip establishes nothing about the vendor, so it is not an attestation.
|
|
152
|
+
basis: version ? "attested" : "declared",
|
|
89
153
|
note: version ?? `${adapter.profile.binary} not found on PATH`,
|
|
90
154
|
});
|
|
91
155
|
|
|
156
|
+
results.push(contractResult(id, adapter, version));
|
|
157
|
+
|
|
92
158
|
const observations = observeAdapter(adapter);
|
|
159
|
+
const attestation = loadAttestation(id, version, adapter, opts);
|
|
93
160
|
for (const dimension of dimensions) {
|
|
94
161
|
const claim = adapter.profile.capabilities[dimension];
|
|
95
|
-
const
|
|
162
|
+
const live = attestation?.observations[dimension as AttestableDimension];
|
|
163
|
+
// A live observation of the installed CLI outranks a fixture check of
|
|
164
|
+
// Harnery's own normalizer. Everything else keeps its adapter basis.
|
|
165
|
+
const { observed, basis }: DimensionObservation = live
|
|
166
|
+
? { observed: live, basis: "attested" }
|
|
167
|
+
: observations[dimension];
|
|
96
168
|
results.push({
|
|
97
169
|
harness: id,
|
|
98
170
|
dimension,
|
|
99
171
|
declared: claim.support,
|
|
100
172
|
observed,
|
|
101
173
|
verdict: reconcile(claim.support, observed),
|
|
102
|
-
|
|
174
|
+
basis,
|
|
175
|
+
note: live ? `attested on ${attestation?.binary_version}` : claim.note,
|
|
103
176
|
});
|
|
104
177
|
}
|
|
105
178
|
}
|
|
106
179
|
|
|
107
180
|
const summary = { ...EMPTY_SUMMARY };
|
|
108
|
-
|
|
181
|
+
const basisSummary = { ...EMPTY_BASIS_SUMMARY };
|
|
182
|
+
for (const result of results) {
|
|
183
|
+
summary[result.verdict]++;
|
|
184
|
+
basisSummary[result.basis]++;
|
|
185
|
+
}
|
|
109
186
|
return {
|
|
110
187
|
generatedAt: new Date().toISOString(),
|
|
111
188
|
mode: "offline",
|
|
112
189
|
harnesses: ids,
|
|
113
190
|
results,
|
|
114
191
|
summary,
|
|
192
|
+
basisSummary,
|
|
115
193
|
drift: summary.drift > 0,
|
|
116
194
|
skipped: summary.skipped > 0,
|
|
117
195
|
};
|
|
118
196
|
}
|
|
119
197
|
|
|
120
|
-
|
|
198
|
+
/** Compare the vendor contract a declaration was validated against with the
|
|
199
|
+
* one actually installed (ADR 0037). This is the only dimension attested
|
|
200
|
+
* without a model turn, because a version string costs nothing to read.
|
|
201
|
+
*
|
|
202
|
+
* A mismatch is `drift`, not failure: a newer vendor CLI is not automatically
|
|
203
|
+
* broken, but the declaration is no longer backed by an observation. An
|
|
204
|
+
* unrecorded or unparseable declared version stays `unknown` rather than being
|
|
205
|
+
* inferred from the installed one. */
|
|
206
|
+
function contractResult(
|
|
207
|
+
id: HarnessId,
|
|
208
|
+
adapter: HarnessAdapter,
|
|
209
|
+
version: string | null,
|
|
210
|
+
): BenchResult {
|
|
211
|
+
const recorded = adapter.profile.verified;
|
|
212
|
+
const base = { harness: id, dimension: "contract" as const, declared: "supported" as const };
|
|
213
|
+
|
|
214
|
+
if (!version) {
|
|
215
|
+
return {
|
|
216
|
+
...base,
|
|
217
|
+
observed: "skipped",
|
|
218
|
+
verdict: "skipped",
|
|
219
|
+
basis: "declared",
|
|
220
|
+
note: `${adapter.profile.binary} not found on PATH; installed contract unobservable`,
|
|
221
|
+
};
|
|
222
|
+
}
|
|
223
|
+
const declaredToken = versionToken(recorded?.version);
|
|
224
|
+
if (!declaredToken) {
|
|
225
|
+
return {
|
|
226
|
+
...base,
|
|
227
|
+
observed: "unknown",
|
|
228
|
+
verdict: "unknown",
|
|
229
|
+
basis: "declared",
|
|
230
|
+
note: recorded
|
|
231
|
+
? `verified.version "${recorded.version}" is not a version; installed ${version}`
|
|
232
|
+
: `no verified vendor contract recorded; installed ${version}`,
|
|
233
|
+
};
|
|
234
|
+
}
|
|
235
|
+
const installedToken = versionToken(version) ?? version.trim().toLowerCase();
|
|
236
|
+
if (installedToken === declaredToken) {
|
|
237
|
+
return {
|
|
238
|
+
...base,
|
|
239
|
+
observed: "supported",
|
|
240
|
+
verdict: "supported",
|
|
241
|
+
basis: "attested",
|
|
242
|
+
note: `declaration validated against ${recorded?.version} on ${recorded?.date}`,
|
|
243
|
+
};
|
|
244
|
+
}
|
|
245
|
+
return {
|
|
246
|
+
...base,
|
|
247
|
+
observed: "drift",
|
|
248
|
+
verdict: "drift",
|
|
249
|
+
basis: "attested",
|
|
250
|
+
note: `declaration validated against ${recorded?.version} (${recorded?.date}); installed ${version}`,
|
|
251
|
+
};
|
|
252
|
+
}
|
|
253
|
+
|
|
254
|
+
interface DimensionObservation {
|
|
255
|
+
observed: BenchVerdict;
|
|
256
|
+
basis: BenchBasis;
|
|
257
|
+
}
|
|
258
|
+
|
|
259
|
+
/** Checked against Harnery's planner, normalizer, or fixture. */
|
|
260
|
+
function fromAdapter(observed: BenchVerdict): DimensionObservation {
|
|
261
|
+
return { observed, basis: "adapter" };
|
|
262
|
+
}
|
|
263
|
+
|
|
264
|
+
/** No check exists for this dimension yet. */
|
|
265
|
+
const NOT_CHECKED: DimensionObservation = { observed: "unknown", basis: "declared" };
|
|
266
|
+
|
|
267
|
+
function observeAdapter(
|
|
268
|
+
adapter: HarnessAdapter,
|
|
269
|
+
): Record<HarnessCapabilityDimension, DimensionObservation> {
|
|
121
270
|
const profile = adapter.profile;
|
|
122
271
|
const effort = profile.effortValues[0];
|
|
123
272
|
const request: SpawnRequest = {
|
|
@@ -154,49 +303,57 @@ function observeAdapter(adapter: HarnessAdapter): Record<HarnessCapabilityDimens
|
|
|
154
303
|
const hookSubcommands = new Set(hookSpec?.events.map((event) => event.subcommand) ?? []);
|
|
155
304
|
|
|
156
305
|
return {
|
|
157
|
-
invocation:
|
|
306
|
+
invocation: fromAdapter(
|
|
158
307
|
!planningFailed && argv[0] === profile.binary && argv.includes(request.prompt)
|
|
159
308
|
? "supported"
|
|
160
309
|
: "unsupported",
|
|
161
|
-
|
|
310
|
+
),
|
|
311
|
+
modelSelection: fromAdapter(
|
|
162
312
|
!planningFailed && argv.includes(request.model ?? "") ? "supported" : "unsupported",
|
|
163
|
-
|
|
313
|
+
),
|
|
314
|
+
effortSelection: fromAdapter(
|
|
164
315
|
effort === undefined
|
|
165
316
|
? "unsupported"
|
|
166
317
|
: !planningFailed && argv.some((arg) => arg.includes(effort))
|
|
167
318
|
? "supported"
|
|
168
319
|
: "unsupported",
|
|
169
|
-
|
|
320
|
+
),
|
|
321
|
+
maxTurns: fromAdapter(
|
|
170
322
|
!planningFailed && argv.includes(String(request.maxTurns)) ? "supported" : "unsupported",
|
|
171
|
-
|
|
172
|
-
|
|
323
|
+
),
|
|
324
|
+
finalResult: fromAdapter(finalMatches ? "supported" : "unsupported"),
|
|
325
|
+
sessionId: fromAdapter(
|
|
173
326
|
sessionObserved === "supported" && normalized?.sessionId !== fixture.sessionId
|
|
174
327
|
? "unsupported"
|
|
175
328
|
: sessionObserved,
|
|
176
|
-
|
|
329
|
+
),
|
|
330
|
+
cost: fromAdapter(
|
|
177
331
|
costObserved === "supported" && normalized?.costUsd !== fixture.costUsd
|
|
178
332
|
? "unsupported"
|
|
179
333
|
: costObserved,
|
|
180
|
-
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
|
|
334
|
+
),
|
|
335
|
+
toolEvidence: fromAdapter(
|
|
336
|
+
normalized && "toolEvidence" in normalized ? "supported" : "unsupported",
|
|
337
|
+
),
|
|
338
|
+
policyMapping: NOT_CHECKED,
|
|
339
|
+
interruption: NOT_CHECKED,
|
|
340
|
+
streaming: NOT_CHECKED,
|
|
341
|
+
steering: NOT_CHECKED,
|
|
342
|
+
resume: NOT_CHECKED,
|
|
343
|
+
images: NOT_CHECKED,
|
|
344
|
+
contextTelemetry: NOT_CHECKED,
|
|
188
345
|
preCompactionSignal: hookSpec
|
|
189
|
-
? hookSubcommands.has("pre-compact")
|
|
190
|
-
|
|
191
|
-
: "unsupported"
|
|
192
|
-
: "unknown",
|
|
346
|
+
? fromAdapter(hookSubcommands.has("pre-compact") ? "supported" : "unsupported")
|
|
347
|
+
: NOT_CHECKED,
|
|
193
348
|
postCompactionSignal: hookSpec
|
|
194
|
-
?
|
|
195
|
-
|
|
196
|
-
|
|
197
|
-
|
|
198
|
-
|
|
199
|
-
|
|
349
|
+
? fromAdapter(
|
|
350
|
+
hookSubcommands.has("post-compact") ||
|
|
351
|
+
(adapter.profile.id === "claude-code" && hookSubcommands.has("session-start"))
|
|
352
|
+
? "supported"
|
|
353
|
+
: "unsupported",
|
|
354
|
+
)
|
|
355
|
+
: NOT_CHECKED,
|
|
356
|
+
compaction: NOT_CHECKED,
|
|
200
357
|
};
|
|
201
358
|
}
|
|
202
359
|
|
|
@@ -207,7 +364,10 @@ function reconcile(declared: CapabilitySupport, observed: BenchVerdict): BenchVe
|
|
|
207
364
|
return declared === observed ? observed : "drift";
|
|
208
365
|
}
|
|
209
366
|
|
|
210
|
-
|
|
367
|
+
/** Ask an installed vendor CLI what version it is. Null when it is absent or
|
|
368
|
+
* refuses to answer. Shared with the live attestation probe so both halves of
|
|
369
|
+
* the capability story key on the same string. */
|
|
370
|
+
export function probeBinaryVersion(binary: string): string | null {
|
|
211
371
|
const result = spawnSync(binary, ["--version"], { encoding: "utf8", timeout: 5_000 });
|
|
212
372
|
if (result.error || result.status !== 0) return null;
|
|
213
373
|
return (result.stdout || result.stderr).trim().split("\n")[0] || "installed";
|
|
@@ -1,12 +1,37 @@
|
|
|
1
1
|
export type { Spawner, SpawnRequest, SpawnResult } from "../workflow/types.ts";
|
|
2
2
|
export type {
|
|
3
|
+
AttestationOutcome,
|
|
4
|
+
HarnessAttestationReport,
|
|
5
|
+
HarnessAttestationResult,
|
|
6
|
+
RunHarnessAttestationOptions,
|
|
7
|
+
} from "./attest.ts";
|
|
8
|
+
export { ATTESTATION_PROMPT, runHarnessAttestation } from "./attest.ts";
|
|
9
|
+
export type {
|
|
10
|
+
AttestableDimension,
|
|
11
|
+
AttestationStoreOptions,
|
|
12
|
+
HarnessAttestation,
|
|
13
|
+
} from "./attestation.ts";
|
|
14
|
+
export {
|
|
15
|
+
ATTESTABLE_DIMENSIONS,
|
|
16
|
+
ATTESTATION_SCHEMA_VERSION,
|
|
17
|
+
attestationsDir,
|
|
18
|
+
harnessProofInputs,
|
|
19
|
+
isAttestationCurrent,
|
|
20
|
+
listAttestations,
|
|
21
|
+
profileDigest,
|
|
22
|
+
readAttestation,
|
|
23
|
+
validateAttestation,
|
|
24
|
+
writeAttestation,
|
|
25
|
+
} from "./attestation.ts";
|
|
26
|
+
export type {
|
|
27
|
+
BenchBasis,
|
|
3
28
|
BenchDimension,
|
|
4
29
|
BenchResult,
|
|
5
30
|
BenchVerdict,
|
|
6
31
|
HarnessBenchOptions,
|
|
7
32
|
HarnessBenchReport,
|
|
8
33
|
} from "./bench.ts";
|
|
9
|
-
export { runHarnessBench } from "./bench.ts";
|
|
34
|
+
export { probeBinaryVersion, runHarnessBench } from "./bench.ts";
|
|
10
35
|
export type { BuiltinHarnessId } from "./profiles.ts";
|
|
11
36
|
export {
|
|
12
37
|
BUILTIN_HARNESS_IDS,
|
|
@@ -43,7 +43,7 @@ export const BUILTIN_HARNESS_PROFILES = {
|
|
|
43
43
|
authModel: "own-auth",
|
|
44
44
|
modelFamily: "claude",
|
|
45
45
|
effortValues: ["low", "medium", "high", "xhigh", "max"],
|
|
46
|
-
verified: { date: "2026-07-
|
|
46
|
+
verified: { date: "2026-07-25", version: "2.1.197 (Claude Code)" },
|
|
47
47
|
capabilities: capabilities({
|
|
48
48
|
effortSelection: supported("Mapped to `--effort <level>`."),
|
|
49
49
|
maxTurns: supported("Mapped to `--max-turns <n>`."),
|
|
@@ -69,7 +69,7 @@ export const BUILTIN_HARNESS_PROFILES = {
|
|
|
69
69
|
authModel: "own-auth",
|
|
70
70
|
modelFamily: "gpt",
|
|
71
71
|
effortValues: ["none", "minimal", "low", "medium", "high", "xhigh"],
|
|
72
|
-
verified: { date: "2026-07-
|
|
72
|
+
verified: { date: "2026-07-25", version: "codex-cli 0.144.5" },
|
|
73
73
|
capabilities: capabilities({
|
|
74
74
|
effortSelection: supported('Mapped to `-c model_reasoning_effort="<level>"`.'),
|
|
75
75
|
maxTurns: unsupported("codex exec exposes no turn-ceiling flag."),
|
|
@@ -95,7 +95,7 @@ export const BUILTIN_HARNESS_PROFILES = {
|
|
|
95
95
|
authModel: "own-auth",
|
|
96
96
|
modelFamily: "multi",
|
|
97
97
|
effortValues: [],
|
|
98
|
-
verified: { date: "2026-07-
|
|
98
|
+
verified: { date: "2026-07-25", version: "2026.07.23-e383d2b" },
|
|
99
99
|
capabilities: capabilities({
|
|
100
100
|
effortSelection: unsupported(
|
|
101
101
|
"Cursor embeds effort in some parameterized model ids; Harnery does not rewrite model ids.",
|
|
@@ -316,6 +316,7 @@ async function executeWorkflow(
|
|
|
316
316
|
agents: [],
|
|
317
317
|
evidence: [],
|
|
318
318
|
harnessEvidence: opts.harnessEvidence,
|
|
319
|
+
harnessAttestations: opts.harnessAttestations,
|
|
319
320
|
policy: policy
|
|
320
321
|
? {
|
|
321
322
|
config: policy,
|
|
@@ -1054,6 +1055,7 @@ async function executeWorkflow(
|
|
|
1054
1055
|
agents: Array.from(agentProofs.values()),
|
|
1055
1056
|
evidence: evidenceRecords,
|
|
1056
1057
|
harnessEvidence: opts.harnessEvidence,
|
|
1058
|
+
harnessAttestations: opts.harnessAttestations,
|
|
1057
1059
|
policy: policy
|
|
1058
1060
|
? {
|
|
1059
1061
|
config: policy,
|
|
@@ -1133,6 +1135,7 @@ async function executeWorkflow(
|
|
|
1133
1135
|
agents: Array.from(agentProofs.values()),
|
|
1134
1136
|
evidence: evidenceRecords,
|
|
1135
1137
|
harnessEvidence: opts.harnessEvidence,
|
|
1138
|
+
harnessAttestations: opts.harnessAttestations,
|
|
1136
1139
|
policy: policy
|
|
1137
1140
|
? {
|
|
1138
1141
|
config: policy,
|
|
@@ -16,6 +16,7 @@ import type {
|
|
|
16
16
|
AcceptanceCriterion,
|
|
17
17
|
AcceptanceResult,
|
|
18
18
|
AcceptanceSummary,
|
|
19
|
+
HarnessAttestationCitation,
|
|
19
20
|
HarnessEvidenceCapability,
|
|
20
21
|
HarnessEvidenceCoverage,
|
|
21
22
|
ResultDigest,
|
|
@@ -81,6 +82,9 @@ export interface BuildWorkflowProofInput {
|
|
|
81
82
|
agents: WorkflowAgentProof[];
|
|
82
83
|
evidence: WorkflowEvidenceRecord[];
|
|
83
84
|
harnessEvidence?: Readonly<Record<string, HarnessEvidenceCapability | undefined>>;
|
|
85
|
+
/** Live attestations backing each harness's claims (ADR 0038). Injected by
|
|
86
|
+
* the caller so proof stays free of filesystem lookups. */
|
|
87
|
+
harnessAttestations?: Readonly<Record<string, HarnessAttestationCitation | undefined>>;
|
|
84
88
|
policy?: {
|
|
85
89
|
config: Readonly<NormalizedPolicy>;
|
|
86
90
|
decisions: readonly PolicyDecision[];
|
|
@@ -222,7 +226,7 @@ export function buildWorkflowProof(input: BuildWorkflowProofInput): WorkflowProo
|
|
|
222
226
|
session_id: clippedOptional(agent.session_id, MAX_REF_CHARS),
|
|
223
227
|
error: clippedOptional(agent.error, MAX_SUMMARY_CHARS),
|
|
224
228
|
}));
|
|
225
|
-
const harnesses = buildHarnessCoverage(agents, input.harnessEvidence);
|
|
229
|
+
const harnesses = buildHarnessCoverage(agents, input.harnessEvidence, input.harnessAttestations);
|
|
226
230
|
const unknowns = buildUnknowns(agents, harnesses, repository);
|
|
227
231
|
const journal = readFileSync(input.journalPath);
|
|
228
232
|
return {
|
|
@@ -455,6 +459,7 @@ function normalizeRepoSnapshot(snapshot: RepoSnapshot): WorkflowRepoSnapshot {
|
|
|
455
459
|
function buildHarnessCoverage(
|
|
456
460
|
agents: WorkflowAgentProof[],
|
|
457
461
|
claims: Readonly<Record<string, HarnessEvidenceCapability | undefined>> | undefined,
|
|
462
|
+
attestations: Readonly<Record<string, HarnessAttestationCitation | undefined>> | undefined,
|
|
458
463
|
): HarnessEvidenceCoverage[] {
|
|
459
464
|
return [...new Set(agents.map((agent) => agent.harness))].map((harness) => {
|
|
460
465
|
const harnessAgents = agents.filter((agent) => agent.harness === harness);
|
|
@@ -469,6 +474,7 @@ function buildHarnessCoverage(
|
|
|
469
474
|
session_ids: harnessAgents.filter((agent) => agent.session_id).length,
|
|
470
475
|
costs: harnessAgents.filter((agent) => agent.cost_usd !== undefined).length,
|
|
471
476
|
},
|
|
477
|
+
...(attestations?.[harness] ? { attestation: attestations[harness] } : {}),
|
|
472
478
|
};
|
|
473
479
|
});
|
|
474
480
|
}
|
|
@@ -23,6 +23,7 @@ import { validateHarnessEffort } from "../harnesses/profiles.ts";
|
|
|
23
23
|
import type { HarnessInvocation, HarnessRawResult } from "../harnesses/types.ts";
|
|
24
24
|
import { buildChildEnv } from "./child-env.ts";
|
|
25
25
|
import { notFoundError } from "./harnesses.ts";
|
|
26
|
+
import { vendorFailureText } from "./spawn-failure.ts";
|
|
26
27
|
import type { Spawner, SpawnRequest, SpawnResult } from "./types.ts";
|
|
27
28
|
|
|
28
29
|
interface ClaudeEnvelope {
|
|
@@ -65,7 +66,7 @@ export function normalizeClaudeResult(raw: HarnessRawResult): SpawnResult {
|
|
|
65
66
|
ok: false,
|
|
66
67
|
text: "",
|
|
67
68
|
durationMs: raw.durationMs,
|
|
68
|
-
error: `claude exited ${raw.exitCode}: ${(raw
|
|
69
|
+
error: `claude exited ${raw.exitCode}: ${vendorFailureText(raw)}`,
|
|
69
70
|
};
|
|
70
71
|
}
|
|
71
72
|
|
|
@@ -24,6 +24,7 @@ import { validateHarnessEffort } from "../harnesses/profiles.ts";
|
|
|
24
24
|
import type { HarnessInvocation, HarnessRawResult } from "../harnesses/types.ts";
|
|
25
25
|
import { buildChildEnv } from "./child-env.ts";
|
|
26
26
|
import { notFoundError } from "./harnesses.ts";
|
|
27
|
+
import { vendorFailureText } from "./spawn-failure.ts";
|
|
27
28
|
import type { Spawner, SpawnRequest, SpawnResult } from "./types.ts";
|
|
28
29
|
|
|
29
30
|
export function buildCodexInvocation(req: SpawnRequest, resultFile?: string): HarnessInvocation {
|
|
@@ -53,7 +54,7 @@ export function normalizeCodexResult(raw: HarnessRawResult): SpawnResult {
|
|
|
53
54
|
ok: false,
|
|
54
55
|
text: "",
|
|
55
56
|
durationMs: raw.durationMs,
|
|
56
|
-
error: `codex exited ${raw.exitCode}: ${(raw
|
|
57
|
+
error: `codex exited ${raw.exitCode}: ${vendorFailureText(raw)}`,
|
|
57
58
|
};
|
|
58
59
|
}
|
|
59
60
|
return {
|
|
@@ -24,6 +24,7 @@ import { validateHarnessEffort } from "../harnesses/profiles.ts";
|
|
|
24
24
|
import type { HarnessInvocation, HarnessRawResult } from "../harnesses/types.ts";
|
|
25
25
|
import { buildChildEnv } from "./child-env.ts";
|
|
26
26
|
import { notFoundError } from "./harnesses.ts";
|
|
27
|
+
import { vendorFailureText } from "./spawn-failure.ts";
|
|
27
28
|
import type { Spawner, SpawnRequest, SpawnResult } from "./types.ts";
|
|
28
29
|
|
|
29
30
|
interface CursorEnvelope {
|
|
@@ -72,7 +73,7 @@ export function normalizeCursorResult(raw: HarnessRawResult): SpawnResult {
|
|
|
72
73
|
ok: false,
|
|
73
74
|
text: "",
|
|
74
75
|
durationMs: raw.durationMs,
|
|
75
|
-
error: `cursor-agent exited ${raw.exitCode}: ${(raw
|
|
76
|
+
error: `cursor-agent exited ${raw.exitCode}: ${vendorFailureText(raw)}`,
|
|
76
77
|
};
|
|
77
78
|
}
|
|
78
79
|
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Shared failure-text extraction for the harness spawn adapters.
|
|
3
|
+
*
|
|
4
|
+
* A vendor CLI prints its banner, resolved config, and startup warnings first,
|
|
5
|
+
* and the reason it actually failed last. Every adapter used to keep the FIRST
|
|
6
|
+
* 500 characters of the transcript, which reliably preserved the banner and
|
|
7
|
+
* discarded the answer: a child that died with "your workspace is out of
|
|
8
|
+
* credits" reported a cosmetic startup warning instead.
|
|
9
|
+
*
|
|
10
|
+
* Two rules follow. Keep the tail. Include both streams, because which one
|
|
11
|
+
* carries the reason varies by vendor and by failure.
|
|
12
|
+
*/
|
|
13
|
+
|
|
14
|
+
const DEFAULT_MAX_CHARS = 500;
|
|
15
|
+
|
|
16
|
+
/** Bounded failure text from a finished child, tail-preserving. */
|
|
17
|
+
export function vendorFailureText(
|
|
18
|
+
raw: { stdout?: string; stderr?: string },
|
|
19
|
+
maxChars = DEFAULT_MAX_CHARS,
|
|
20
|
+
): string {
|
|
21
|
+
const parts = [raw.stderr, raw.stdout]
|
|
22
|
+
.map((part) => part?.trim())
|
|
23
|
+
.filter((part): part is string => !!part);
|
|
24
|
+
if (parts.length === 0) return "";
|
|
25
|
+
// Both streams, stderr first, so a reason on the unexpected stream survives.
|
|
26
|
+
const combined = parts.join("\n");
|
|
27
|
+
return combined.length > maxChars ? `…${combined.slice(-maxChars)}` : combined;
|
|
28
|
+
}
|
|
@@ -126,6 +126,14 @@ export interface WorkflowRepoEvidence {
|
|
|
126
126
|
};
|
|
127
127
|
}
|
|
128
128
|
|
|
129
|
+
/** Bounded pointer to the live attestation backing a harness's claims
|
|
130
|
+
* (ADR 0038). Structural facts only: no prompt text, no host paths. */
|
|
131
|
+
export interface HarnessAttestationCitation {
|
|
132
|
+
binary_version: string;
|
|
133
|
+
observed_at: string;
|
|
134
|
+
record_digest: string;
|
|
135
|
+
}
|
|
136
|
+
|
|
129
137
|
export interface HarnessEvidenceCoverage {
|
|
130
138
|
harness: HarnessName;
|
|
131
139
|
tool_evidence: {
|
|
@@ -137,6 +145,10 @@ export interface HarnessEvidenceCoverage {
|
|
|
137
145
|
session_ids: number;
|
|
138
146
|
costs: number;
|
|
139
147
|
};
|
|
148
|
+
/** What backs this harness's capability claims (ADR 0038). Absent when the
|
|
149
|
+
* host recorded no live attestation, which is the common case and is not by
|
|
150
|
+
* itself a proof unknown. */
|
|
151
|
+
attestation?: HarnessAttestationCitation;
|
|
140
152
|
}
|
|
141
153
|
|
|
142
154
|
export interface WorkflowProofUnknown {
|
|
@@ -429,6 +441,9 @@ export interface EngineOpts {
|
|
|
429
441
|
/** Capability claims used to state whether adapter-native tool evidence was
|
|
430
442
|
* available. Missing claims remain unknown. */
|
|
431
443
|
harnessEvidence?: Readonly<Record<HarnessName, HarnessEvidenceCapability | undefined>>;
|
|
444
|
+
/** Live attestations backing each harness's claims (ADR 0038). Read once by
|
|
445
|
+
* the host and injected, so the engine performs no capability lookups. */
|
|
446
|
+
harnessAttestations?: Readonly<Record<HarnessName, HarnessAttestationCitation | undefined>>;
|
|
432
447
|
/** Immutable host policy. Workflow scripts and model prompts cannot replace it. */
|
|
433
448
|
policy?: PolicySpec | NormalizedPolicy;
|
|
434
449
|
/** Host callback for ASK. Missing, invalid, throwing, or timed-out resolution denies. */
|